From 7dde28ea9b2721c644b08b08ad973c85beb28649 Mon Sep 17 00:00:00 2001 From: Richard O'Shaughnessy Date: Thu, 17 Sep 2026 13:20:34 -0700 Subject: [PATCH 1/6] close five review findings from PR #363, and measure the sixth away F1 The defaults reader returned on the FIRST ast.Dict it walked into, so a second construction site with wrong defaults was never looked at. Pin the construction-site count exactly, and check every site's dict. F3 Normalize both sides of that comparison through ast.unparse instead of deleting every space from one side, and refuse an unresolvable dict key with a named assertion rather than an AttributeError on k.value. F4 The drop-the-factor test pins a partition of six parameter TUPLES, not of the eight definitions; say so, since the claim belongs to it and _EXPECTED_SIGNATURES together. Also assert the calling and dropping sets are disjoint. F5 make_e2e_calibration: refuse a bad --lane before mkdtemp and build_event, which cost a fixture build and leaked a directory on /. F6 Tag lanes by enumerate rather than LANES.index, a first-match lookup; correct the field-count comment above LANES. F2 is NOT a defect. xpy.asarray(x, dtype=float) does land on the device AND take object dtype: cupy refuses object dtype only when no dtype is passed, and the reviewer's suggested xpy.asarray(np.asarray(x, dtype=float)) would raise on every cupy input. Measured on cupy 12.0.0 on two hosts and recorded on the line; pinned by a new 5 s test on the dtype keyword. Each new guard was mutation-checked against the mutation it exists for, with the pre-change file as the control arm. Co-Authored-By: Claude Opus 5 --- .travis/test-integrate.sh | 2 +- .../Code/test/analytic_supplement_for_e2e.py | 20 ++- .../integrators/make_e2e_calibration.py | 35 ++-- .../Code/test/test_zero_likelihood_standin.py | 151 +++++++++++++++--- 4 files changed, 170 insertions(+), 38 deletions(-) diff --git a/.travis/test-integrate.sh b/.travis/test-integrate.sh index 8536f0456..7380ca984 100755 --- a/.travis/test-integrate.sh +++ b/.travis/test-integrate.sh @@ -127,7 +127,7 @@ python -m pytest -q "$_PORTDENS_TESTS" # (one parametrized case each, currently 8); if a signature is added, look at the new one and # update the number. _ZLSTANDIN_TESTS=MonteCarloMarginalizeCode/Code/test/test_zero_likelihood_standin.py -_ZLSTANDIN_EXPECTED=23 +_ZLSTANDIN_EXPECTED=24 _ZLSTANDIN_FOUND=$(python -m pytest -q --collect-only "$_ZLSTANDIN_TESTS" 2>/dev/null | grep -c '::' || true) if [ "$_ZLSTANDIN_FOUND" -ne "$_ZLSTANDIN_EXPECTED" ]; then echo "zero-likelihood stand-in gate: collected $_ZLSTANDIN_FOUND tests, expected $_ZLSTANDIN_EXPECTED" >&2 diff --git a/MonteCarloMarginalizeCode/Code/test/analytic_supplement_for_e2e.py b/MonteCarloMarginalizeCode/Code/test/analytic_supplement_for_e2e.py index 964bc6007..dd6785195 100644 --- a/MonteCarloMarginalizeCode/Code/test/analytic_supplement_for_e2e.py +++ b/MonteCarloMarginalizeCode/Code/test/analytic_supplement_for_e2e.py @@ -73,10 +73,22 @@ def ln_analytic_factor(right_ascension, declination, phi_orb, inclination, psi, # cannot check it, because every lane here pins CUDA_VISIBLE_DEVICES="" and the CI runners # have no cupy. # - # The CAST is not decoration either. mcsampler (--sampler-method adaptive_cartesian) hands - # its integrand object-dtype draws, on which np.cos raises "loop of ufunc does not support - # argument 0 of type float"; the driver's own non-vectorized likelihood casts for the same - # reason ("get rid of 'object'"). Any real supplementary factor needs it. + # The CAST is not decoration, and the EXPLICIT dtype is the whole of it. mcsampler + # (--sampler-method adaptive_cartesian) hands its integrand object-dtype draws, on which + # np.cos raises "loop of ufunc does not support argument 0 of type float"; the driver's own + # non-vectorized likelihood casts for the same reason ("get rid of 'object'"). + # + # ONE line does BOTH jobs -- object dtype to float, and host to device -- and it does so + # only because `dtype=` is passed. Measured on cupy 12.0.0, 2026-09-17, on ldas-pcdev2 + # (A100, cc 8.0) and ldas-pcdev13 (RTX 2080 Ti, cc 7.5): + # cupy.asarray(obj_arr) -> ValueError: Unsupported dtype object + # cupy.asarray(obj_arr, dtype=float) -> float64 cupy.ndarray, values preserved + # cupy.asarray(device_arr, dtype=float) -> float64 cupy.ndarray + # A review read the first of those three and concluded this line could not be doing both. + # Do NOT act on that by rewriting it as xpy.asarray(np.asarray(x, dtype=float)): np.asarray + # of a cupy array raises TypeError ("Implicit conversion to a NumPy array is not allowed"), + # so that form breaks every GPU run -- the exact path the line exists for. Re-check by + # running those three calls; it is a three-line probe, not an argument. phi_orb = xpy.asarray(phi_orb, dtype=float) out = A_COEFF * xpy.cos(phi_orb) if B_COEFF: diff --git a/MonteCarloMarginalizeCode/Code/test/expensive_before_merging/integrators/make_e2e_calibration.py b/MonteCarloMarginalizeCode/Code/test/expensive_before_merging/integrators/make_e2e_calibration.py index e4356ea69..0e782d9e0 100644 --- a/MonteCarloMarginalizeCode/Code/test/expensive_before_merging/integrators/make_e2e_calibration.py +++ b/MonteCarloMarginalizeCode/Code/test/expensive_before_merging/integrators/make_e2e_calibration.py @@ -31,7 +31,7 @@ _AV, _PORTFOLIO, _GMM = gate._AV, gate._PORTFOLIO, gate._GMM -# (label, sampler argv, A, B, exact-A, exact-B, extra kwargs for _run_ile) +# (label, sampler argv, A, B, extra kwargs for _run_ile). A is None for a prior-only lane. LANES = [ ("prior-only, AV", _AV, None, 0.0, {}), ("prior-only, portfolio", _PORTFOLIO, None, 0.0, {}), @@ -79,31 +79,44 @@ def main(): ap.add_argument("--lane", default=None, help="substring; measure only matching lanes") args = ap.parse_args() + # Lane selection BEFORE anything expensive or on-disk. It depends on nothing but argv, and + # running it after mkdtemp/build_event meant a --lane typo cost a full fixture build and + # then leaked the directory: the refusal path below neither prints nor removes it, and on + # the CIT nodes that directory is on /, which is 20-22 GB. + # + # The index is carried alongside each lane, and it is the index into LANES rather than into + # this filtered list, so --lane does not renumber the output tags. + selected = [(i, L) for i, L in enumerate(LANES) if not args.lane or args.lane in L[0]] + if not selected: + # Exiting 0 having measured nothing, under a summary line that reads like a clean + # result, is the exact shape this whole branch exists to remove. + raise SystemExit("--lane %r matched none of:\n %s" + % (args.lane, "\n ".join(L[0].strip() for L in LANES))) + import pathlib out = pathlib.Path(tempfile.mkdtemp(prefix="e2e_calibration_")) event = gate.build_event(out) if event is None: - raise SystemExit("could not build the fixture (lal_path2cache missing or failed)") + raise SystemExit( + "could not build the fixture (lal_path2cache missing or failed); the empty " + "directory is at %s" % out) seeds = [args.first_seed + i for i in range(args.seeds)] - selected = [L for L in LANES if not args.lane or args.lane in L[0]] - if not selected: - # Exiting 0 having measured nothing, under a summary line that reads like a clean - # result, is the exact shape this whole branch exists to remove. - raise SystemExit("--lane %r matched none of:\n %s" - % (args.lane, "\n ".join(L[0].strip() for L in LANES))) print("# lane max |z| max sigma min n_eff") worst_z = worst_s = 0.0 least_n = float("inf") - for label, sampler, a, b, kw in selected: - kw0 = kw + for idx, (label, sampler, a, b, kw) in selected: kw = dict(kw) if kw.pop("_needs_dmarg", False): kw.update(_dmarg_extra(event, out)) exact = 0.0 if a is None else gate._exact(a, b) zs, sigs, neffs = [], [], [] for seed in seeds: - tag = "cal_%d_%d" % (LANES.index((label, sampler, a, b, kw0)), seed) + # `idx` from enumerate, NOT LANES.index(...): index is a first-match lookup, so two + # identical lane rows would silently share a tag and overwrite each other's ILE + # output directory. Correct for today's 17 distinct rows; wrong the moment one is + # duplicated, which is exactly the kind of edit this table invites. + tag = "cal_%d_%d" % (idx, seed) lnL, sigma, neff = gate._run_ile(event, tag, sampler, a_coeff=a, b_coeff=b, seed=seed, **kw) zs.append(abs((lnL - exact) / sigma)) diff --git a/MonteCarloMarginalizeCode/Code/test/test_zero_likelihood_standin.py b/MonteCarloMarginalizeCode/Code/test/test_zero_likelihood_standin.py index 055cbbda0..bf926530d 100644 --- a/MonteCarloMarginalizeCode/Code/test/test_zero_likelihood_standin.py +++ b/MonteCarloMarginalizeCode/Code/test/test_zero_likelihood_standin.py @@ -230,6 +230,45 @@ def test_the_shipped_example_factor_accepts_the_real_call_shapes(): ex.ln_analytic_factor(*([a] * len(order)), **{kw: _np}) +def test_the_shipped_example_survives_object_dtype_draws(): + """The example's cast, pinned on the keyword that is easy to lose. + + mcsampler (--sampler-method adaptive_cartesian) hands its integrand OBJECT-dtype draws, on + which np.cos raises "loop of ufunc does not support argument 0 of type float". What makes + the example work on them is the EXPLICIT `dtype=float` on its asarray -- and that same + explicit dtype is what lets cupy.asarray accept an object array at all, since + cupy.asarray(obj) without one raises ValueError: Unsupported dtype object (measured, cupy + 12.0.0; see the comment on the line itself). One keyword carries the host path and the + device path together. + + WHAT THIS ADDS. The end-to-end gate's adaptive_cartesian lane runs with B = 2, so it does + reach both casts and does redden if either loses its dtype -- but at ~3 minutes and a full + ILE run, reported as a ln Z failure rather than as a cast. This is the same check in ~5 s, + in the file that already reads this contract, naming the cause. It does NOT cover the + device half: there is no cupy on the runners, and that half is an argument in the comment, + not a test. + + Verified by mutation: deleting `dtype=float` from either asarray reddens this test.""" + if HERE not in sys.path: + sys.path.insert(0, HERE) + import analytic_supplement_for_e2e as ex + obj = np.array([0.1, 0.2, 0.3], dtype=object) + flt = np.asarray(obj, dtype=float) + b_was = ex.B_COEFF + try: + # B = 0 exercises only the phi_orb cast; B != 0 also reaches the inclination one, which + # the gate's default lane never does. Restored, because this is module state. + for b in (0.0, 2.0): + ex.B_COEFF = b + want = ex.ln_analytic_factor(*([flt] * 6)) + got = ex.ln_analytic_factor(*([obj] * 6)) + assert np.allclose(np.asarray(got, dtype=float), np.asarray(want, dtype=float)), ( + "B_COEFF=%r: the example gives a different answer on object-dtype draws (%r) " + "than on the same values as float64 (%r)" % (b, got, want)) + finally: + ex.B_COEFF = b_was + + def test_the_stand_in_passes_a_keyword_only_to_a_factor_that_takes_it(): """A factor without xpy must not be handed one, or fixing the contract would break every plugin written for the two sites that never passed it.""" @@ -297,10 +336,29 @@ def test_the_paths_that_silently_drop_the_factor_are_the_known_ones(): validated under --zero-likelihood and then run in production WITHOUT --time-marginalization is silently ignored, with the startup banner still announcing it. - The hole is pre-existing; this pins its shape. If it widens, that is a new silently-ignored - configuration. If it shrinks, someone fixed it and should delete the entry rather than let - this file keep describing a hole that closed.""" - _calls, drops = _signatures_by_whether_they_call_the_factor() + The hole is pre-existing; this pins its shape. WHAT IT PINS IS A PARTITION OF PARAMETER + TUPLES, NOT OF DEFINITIONS. The driver's eight likelihood_function defs carry only six + distinct tuples -- three of them share (right_ascension, declination, phi_orb, inclination, + psi, distance) -- so a NEW dropping definition whose tuple is already recorded here is + invisible to this test. What catches that one is _EXPECTED_SIGNATURES, pinned exactly in + test_the_stand_in_reproduces_every_live_likelihood_signature. The claim "a new + silently-ignored configuration reddens something" belongs to the two tests TOGETHER; + neither states it alone. + + So: if this set widens, that is a new silently-ignored configuration. If it shrinks, + someone fixed a path and should delete the entry rather than let this file keep describing a + hole that closed.""" + calls, drops = _signatures_by_whether_they_call_the_factor() + # A tuple in BOTH sets means two definitions share a signature and disagree about calling + # the factor. That is the one shape the partition below cannot express -- "this signature + # drops the factor" stops being a well-formed statement -- and it would otherwise surface + # only as a confusing widening of `drops`. + assert calls.isdisjoint(drops), ( + "signature(s) %r have one likelihood_function definition that calls the supplementary " + "factor and another that does not, so whether the factor applies no longer follows from " + "the signature. The recorded set below cannot describe that; give the two definitions " + "distinguishable signatures, or record the hole some other way." + % sorted(calls & drops)) assert drops == _SIGNATURES_THAT_DROP_THE_FACTOR, ( "the set of likelihood_function signatures that never call the supplementary factor " "changed.\n now dropping: %r\n recorded: %r\nIf a path was fixed, remove it here " @@ -319,9 +377,11 @@ def test_the_help_warns_that_some_paths_drop_the_factor(): "drop the factor (missing %r)" % phrase) -# What the DRIVER's own call site passes as supplement_defaults. Read out of the source, not +# What the DRIVER's own call sites pass as supplement_defaults. Read out of the source, not # reproduced here from memory: the values below were unpinned entirely until a review mutated -# them to nonsense and every test still passed. +# them to nonsense and every test still passed. Written as ordinary source text; both sides of +# the comparison are normalized through ast.unparse, so a multi-token value can be spelled here +# the way it would be written in the driver. _EXPECTED_DEFAULT_SOURCES = { "right_ascension": "P.phi", "declination": "P.theta", @@ -331,19 +391,57 @@ def test_the_help_warns_that_some_paths_drop_the_factor(): "distance": "0.0", } +# How many times the driver CONSTRUCTS the stand-in. Pinned EXACTLY, for the same reason as +# _EXPECTED_CALL_SITES and _EXPECTED_SIGNATURES: the check below reads every site it is given, +# but with the count unpinned a SECOND site could appear and simply never be looked at. +# Demonstrated by mutation: the reader used to return on the first ast.Dict it walked into, so +# splitting the one site into two branches and corrupting the second branch's defaults left +# every test in this file green (all 23 of them, as it then stood) -- coverage decided by walk +# order rather than by the driver. +# If this number changes, read the new site's defaults by hand before updating it. +_EXPECTED_STANDIN_CONSTRUCTIONS = 1 + + +def _normalize_expr(text): + """Canonical source text for an expression, for comparing a hand-written value to an + unparsed one. + + `ast.unparse(v).replace(" ", "")` normalized only the driver's side, and did it by deleting + every space -- which collapses 'a b' with 'ab' and f"{x} {y}" with f"{x}{y}", and means a + multi-token expected value written naturally above ("P.phi if opts.q else P.psi") could + never match anything. Round-tripping BOTH sides through ast.unparse normalizes spacing + without reaching inside a string literal.""" + return ast.unparse(ast.parse(text, mode="eval").body) + def _driver_supplement_defaults(): - """The dict literal the driver hands make_zero_likelihood_standin, as {name: source text}.""" - tree = ast.parse(_driver_source()) - for n in ast.walk(tree): + """Every supplement_defaults dict the driver hands make_zero_likelihood_standin. + + Returns [(line, {name: source text}), ...] -- EVERY construction site, not whichever one + ast.walk reached first.""" + out = [] + for n in ast.walk(ast.parse(_driver_source())): if not (isinstance(n, ast.Call) and isinstance(n.func, ast.Name) and n.func.id == "make_zero_likelihood_standin"): continue - for a in n.args: - if isinstance(a, ast.Dict): - return {k.value: ast.unparse(v).replace(" ", "") - for k, v in zip(a.keys, a.values)} - raise AssertionError("no dict literal in the make_zero_likelihood_standin call") + dicts = [a for a in n.args if isinstance(a, ast.Dict)] + assert len(dicts) == 1, ( + "the make_zero_likelihood_standin call at line %d passes %d dict literals; this " + "reader cannot say which one is supplement_defaults." % (n.lineno, len(dicts))) + got = {} + for k, v in zip(dicts[0].keys, dicts[0].values): + # k is None for `**spread`, and a computed key is an expression this reader cannot + # resolve. Both used to be an AttributeError on k.value, which reads like a broken + # test rather than like a contract the test can no longer check. + assert isinstance(k, ast.Constant), ( + "the supplement_defaults dict at line %d has a key this reader cannot resolve " + "(%s). Spell the defaults out with literal keys, or teach this reader the new " + "shape -- do not leave them unchecked." + % (n.lineno, "**spread" if k is None else ast.dump(k))) + got[k.value] = ast.unparse(v) + out.append((n.lineno, got)) + assert out, "no make_zero_likelihood_standin call site found in the ILE" + return out def test_the_driver_supplies_the_defaults_it_documents(): @@ -352,14 +450,23 @@ def test_the_driver_supplies_the_defaults_it_documents(): wiring tests below use their own _DEFAULTS, and the only gate lane that reaches a default is the distance-marginalized one, whose factor ignores distance. Mutating the driver's literal to P.dist (SI, ~1e24) and P.phiref left all twenty tests green.""" - got = _driver_supplement_defaults() - assert got == _EXPECTED_DEFAULT_SOURCES, ( - "the driver's supplement_defaults changed.\n now: %r\n was: %r\ndistance must stay " - "0.0, which is what the two distance-marginalized call sites pass the factor; the angles " - "must stay the template's own value for that parameter." - % (got, _EXPECTED_DEFAULT_SOURCES)) - assert set(got) == set(_driver_ns()["_SUPPLEMENT_ARG_ORDER"]), \ - "the defaults dict no longer covers exactly the factor's arguments: %r" % sorted(got) + sites = _driver_supplement_defaults() + assert len(sites) == _EXPECTED_STANDIN_CONSTRUCTIONS, ( + "the driver constructs the stand-in at %d sites (lines %r), expected %d. Every site is " + "checked below, so this is not a formality: read the new one's defaults by hand before " + "updating _EXPECTED_STANDIN_CONSTRUCTIONS." + % (len(sites), [ln for ln, _ in sites], _EXPECTED_STANDIN_CONSTRUCTIONS)) + want = {k: _normalize_expr(v) for k, v in _EXPECTED_DEFAULT_SOURCES.items()} + order = set(_driver_ns()["_SUPPLEMENT_ARG_ORDER"]) + for lineno, got in sites: + assert got == want, ( + "the driver's supplement_defaults at line %d changed.\n now: %r\n was: %r\n" + "distance must stay 0.0, which is what the two distance-marginalized call sites pass " + "the factor; the angles must stay the template's own value for that parameter." + % (lineno, got, want)) + assert set(got) == order, ( + "the defaults dict at line %d no longer covers exactly the factor's arguments: %r" + % (lineno, sorted(got))) def test_the_argument_order_is_the_documented_one(): From 5bb8da02b19196979f8c0be52028575ae5f85c5c Mon Sep 17 00:00:00 2001 From: Richard O'Shaughnessy Date: Thu, 17 Sep 2026 13:27:35 -0700 Subject: [PATCH 2/6] run the device claims on a real GPU instead of asserting them The F2 conclusion rested on a probe I ran by hand, and the test I shipped for it covered only the numpy path; its own docstring said the device half was "an argument in the comment, not a test". That is the read-and-trust shape. Fixed: New section 4 runs the shipped example and the driver's own stand-in factory against REAL cupy -- both input kinds (device float64, host object dtype), both lnL conventions, a fully-sampled signature and one falling back to a scalar default -- and checks the result is ON the device and equals the numpy arm. _FakeXpy stays: it runs everywhere, and a fake can be more permissive than what it stands for, so the two cover each other. _cupy_or_skip separates the two failures that look alike. `import cupy` fails on a host with no CUDA runtime; it SUCCEEDS on the CIT GPU head nodes while the visible device is one this cupy cannot build a kernel for (pcdev13 slots 0-2, all of pcdev11: Blackwell cc 12.0, cupy 12.0.0 -> "invalid value for --gpu-architecture"). So it runs a kernel rather than trusting the import, probes at dispatch because the slot map moves, and says in the skip reason that a skip is not a pass. Measured, ldas-pcdev2 slot 0 (A100, cc 8.0) and ldas-pcdev13 slot 3 (RTX 2080 Ti, cc 7.5), cupy 12.0.0, same HEAD and same md5s as the local tree: 27 passed on both. Mutation on the A100: the rewrite the comment warns against, xpy.asarray(np.asarray(x, dtype=float)), now fails 3 tests instead of being refused by an argument; dropping dtype= fails 3; a factor that ignores xpy and returns numpy fails 3. Also corrects the module docstring and the CI comment, which both said this file needs no GPU. Co-Authored-By: Claude Opus 5 --- .travis/test-integrate.sh | 9 +- .../Code/test/test_zero_likelihood_standin.py | 129 +++++++++++++++++- 2 files changed, 131 insertions(+), 7 deletions(-) diff --git a/.travis/test-integrate.sh b/.travis/test-integrate.sh index 7380ca984..686786696 100755 --- a/.travis/test-integrate.sh +++ b/.travis/test-integrate.sh @@ -122,12 +122,17 @@ python -m pytest -q "$_PORTDENS_TESTS" # every live likelihood_function signature, and its array module. This is the half the # end-to-end gate below is structurally blind to -- right_ascension, phi_orb and psi are iid # uniform on [0, 2pi), so NO marginal can tell a permutation of the three apart, and the runners -# have no cupy, so nothing there sees a host/device mistake. Pure AST + exec, ~4 s. +# have no cupy, so nothing there sees a host/device mistake. Mostly AST + exec, ~5 s. +# Its section 4 runs the same device paths against REAL cupy and SKIPS here, because the runners +# have none. Skipped tests are still collected, so the count below is the same everywhere; but +# a green run here has NOT exercised the device half. That half is run by hand on a CIT GPU +# node with CUDA_VISIBLE_DEVICES pinned to a slot the installed cupy supports -- see +# _cupy_or_skip in the test file. # The collected count tracks the number of `def likelihood_function` signatures in the driver # (one parametrized case each, currently 8); if a signature is added, look at the new one and # update the number. _ZLSTANDIN_TESTS=MonteCarloMarginalizeCode/Code/test/test_zero_likelihood_standin.py -_ZLSTANDIN_EXPECTED=24 +_ZLSTANDIN_EXPECTED=27 _ZLSTANDIN_FOUND=$(python -m pytest -q --collect-only "$_ZLSTANDIN_TESTS" 2>/dev/null | grep -c '::' || true) if [ "$_ZLSTANDIN_FOUND" -ne "$_ZLSTANDIN_EXPECTED" ]; then echo "zero-likelihood stand-in gate: collected $_ZLSTANDIN_FOUND tests, expected $_ZLSTANDIN_EXPECTED" >&2 diff --git a/MonteCarloMarginalizeCode/Code/test/test_zero_likelihood_standin.py b/MonteCarloMarginalizeCode/Code/test/test_zero_likelihood_standin.py index bf926530d..fe7e1db4e 100644 --- a/MonteCarloMarginalizeCode/Code/test/test_zero_likelihood_standin.py +++ b/MonteCarloMarginalizeCode/Code/test/test_zero_likelihood_standin.py @@ -9,13 +9,17 @@ So this file checks the WIRING instead of the answer: the argument order against the driver's own real call sites, the generated signature against the live likelihood_function signatures, -and that the values handed to the factor are the RAW sampled ones. It is pure AST + exec, runs -in under a second, and needs no data, no network and no GPU. +and that the values handed to the factor are the RAW sampled ones. That part is pure AST + +exec, runs in a few seconds, and needs no data, no network and no GPU. It also covers the device question the end-to-end gate is structurally blind to: that gate pins -CUDA_VISIBLE_DEVICES="" and the CI runners have no cupy, so nothing there can catch the -stand-in reaching for numpy on a host where the integrand is on a device. Here the array module -is INJECTED, and a fake one records every call. +CUDA_VISIBLE_DEVICES="" for its children by design, and the CI runners have no cupy, so nothing +there can catch the stand-in reaching for numpy on a host where the integrand is on a device. +That is covered twice here, deliberately. Section 3 INJECTS a fake array module that records +every call, which runs everywhere. Section 4 runs the same paths against REAL cupy, because a +fake can be more permissive than the thing it stands for and a device claim checked only +against a fake is a claim nobody ran. Section 4 skips where there is no usable GPU -- a skip +is not a pass; see _cupy_or_skip for how to make it run. """ import ast import inspect @@ -618,3 +622,118 @@ def supplement(*args): # and with no factor at all, the exact zero / one the option promises assert np.allclose(make(sig, True, None, _DEFAULTS, np)(**kw), 0.0) assert np.allclose(make(sig, False, None, _DEFAULTS, np)(**kw), 1.0) + + +# --------------------------------------------------------------------------------------- +# 4. the device path, on a REAL GPU + +def _cupy_or_skip(): + """Real cupy, on a device this cupy can actually build a kernel for, or a skip saying why. + + TWO distinct failures, and conflating them is how a device claim goes unchecked. `import + cupy` fails outright on a host with no CUDA runtime (ldas-grid, and every CI runner). It + SUCCEEDS on the CIT GPU head nodes while the visible device is one this cupy cannot compile + for: ldas-pcdev13 slots 0-2 and all of ldas-pcdev11 are Blackwell cc 12.0, and cupy 12.0.0 + answers `nvrtc: error: invalid value for --gpu-architecture`. So the probe RUNS a kernel + rather than trusting the import, and the slot map moves, so it probes at dispatch. + + A SKIP HERE IS NOT A PASS. Pin CUDA_VISIBLE_DEVICES to a slot this cupy supports and run + the file again; as of 2026-09-17 ldas-pcdev2 slot 0 (A100, cc 8.0) and ldas-pcdev13 slot 3 + (RTX 2080 Ti, cc 7.5) work. The reason string says which of the two failures happened.""" + try: + import cupy + except Exception as exc: + pytest.skip("no usable cupy on this host (%s: %s)" % (type(exc).__name__, + str(exc)[:80])) + try: + cupy.asnumpy(cupy.cos(cupy.asarray(np.zeros(2), dtype=float))) + except Exception as exc: + pytest.skip( + "cupy %s imports but cannot run a kernel on the visible device (%s: %s). Pin " + "CUDA_VISIBLE_DEVICES to a slot this cupy supports; this is NOT a pass." + % (cupy.__version__, type(exc).__name__, str(exc)[:80])) + return cupy + + +@pytest.mark.parametrize("b_coeff", [0.0, 2.0]) +def test_the_shipped_example_really_runs_on_a_gpu(b_coeff): + """THE DEVICE HALF, RUN RATHER THAN ARGUED. + + The example's one cast has to do two jobs: object dtype to float, for the draws mcsampler + hands --sampler-method adaptive_cartesian, and host to device, because the three call sites + that pass xpy do `lnL += factor(...)` with lnL on the device. Everything else that touches + this is blind to the second job: the end-to-end gate pins CUDA_VISIBLE_DEVICES="" for its + children by design, the CI runners have no cupy, and _FakeXpy above is a stand-in that can + be more permissive than the thing it stands for. + + So this runs the real module with xpy=cupy, on BOTH input kinds, and checks the result is + on the device and equals the numpy arm. It is also the test that refuses the rewrite the + comment on that line warns about: xpy.asarray(np.asarray(x, dtype=float)) raises TypeError + on a cupy argument, so the device-float64 case below goes red. Verified by mutation on + ldas-pcdev2. + + b_coeff = 0 exercises only the phi_orb cast; b_coeff != 0 also reaches the inclination one, + which the default B = 0 never does.""" + cupy = _cupy_or_skip() + if HERE not in sys.path: + sys.path.insert(0, HERE) + import analytic_supplement_for_e2e as ex + vals = np.array([0.1, 1.2, 2.3, 3.4]) + b_was = ex.B_COEFF + try: + ex.B_COEFF = b_coeff + want = np.asarray(ex.ln_analytic_factor(*([vals] * 6), xpy=np), dtype=float) + cases = (("device float64", cupy.asarray(vals, dtype=float)), + ("host object dtype", np.asarray(vals, dtype=object))) + for label, arg in cases: + got = ex.ln_analytic_factor(*([arg] * 6), xpy=cupy) + assert isinstance(got, cupy.ndarray), ( + "B=%r, %s input: the factor returned %r, not a device array. The three call " + "sites that pass xpy add this to a device lnL and would raise." + % (b_coeff, label, type(got))) + assert np.allclose(cupy.asnumpy(got), want), ( + "B=%r, %s input: the device answer %r disagrees with the numpy arm %r" + % (b_coeff, label, cupy.asnumpy(got), want)) + finally: + ex.B_COEFF = b_was + + +def test_the_stand_in_builds_its_array_on_a_real_device(): + """The same question as test_the_base_array_comes_from_the_injected_module_not_numpy, asked + of real cupy instead of _FakeXpy. + + _FakeXpy implements zeros/ones/exp over numpy and tags the result. It therefore accepts + things cupy refuses -- an object-dtype array, a numpy array handed to a ufunc -- so it can + report a device path working that would raise on a GPU. This runs the driver's own factory + with xpy=cupy and the shipped factor, on a fully-sampled signature and on one that has to + fall back to a scalar default, and checks the result is on the device and matches numpy.""" + cupy = _cupy_or_skip() + if HERE not in sys.path: + sys.path.insert(0, HERE) + import analytic_supplement_for_e2e as ex + make = _driver_ns()["make_zero_likelihood_standin"] + full = ("right_ascension", "declination", "phi_orb", "inclination", "psi", "distance") + # distance is not sampled here, so the factory has to supply _DEFAULTS["distance"] = 0.0 -- + # a PYTHON SCALAR reaching the factor's cast alongside device arrays. + partial = ("right_ascension", "declination", "phi_orb", "inclination", "psi") + b_was = ex.B_COEFF + try: + ex.B_COEFF = 2.0 + for sig in (full, partial): + host = {nm: np.linspace(0.1, 1.0, 4) for nm in sig} + dev = {nm: cupy.asarray(v, dtype=float) for nm, v in host.items()} + for return_lnL in (True, False): + want = make(sig, return_lnL, ex.ln_analytic_factor, _DEFAULTS, np)(**host) + got = make(sig, return_lnL, ex.ln_analytic_factor, _DEFAULTS, cupy)(**dev) + assert isinstance(got, cupy.ndarray), ( + "signature %r, return_lnL=%r: the stand-in returned %r, not a device array" + % (sig, return_lnL, type(got))) + assert np.allclose(cupy.asnumpy(got), np.asarray(want, dtype=float)), ( + "signature %r, return_lnL=%r: device %r != numpy %r" + % (sig, return_lnL, cupy.asnumpy(got), want)) + # and with no factor at all, the exact zero / one the option promises, on the device + kw = {nm: cupy.asarray(np.zeros(4), dtype=float) for nm in full} + assert np.allclose(cupy.asnumpy(make(full, True, None, _DEFAULTS, cupy)(**kw)), 0.0) + assert np.allclose(cupy.asnumpy(make(full, False, None, _DEFAULTS, cupy)(**kw)), 1.0) + finally: + ex.B_COEFF = b_was From 8ded3f089d4f0ce5594b9a6d631fb39b4387592c Mon Sep 17 00:00:00 2001 From: Richard O'Shaughnessy Date: Thu, 17 Sep 2026 16:42:42 -0700 Subject: [PATCH 3/6] e2e gate: run the pipeline on a real GPU, and record the two samplers that cannot The whole file pinned CUDA_VISIBLE_DEVICES="" and so said nothing about the device path, which is the path production takes by default on a GPU node: RIFT binds its array module at import from whether cupy imports, not from --gpu. Section 5 runs the prior-only and closed-form answers with a device visible and asserts the child reached one. WHAT IT COVERS: the extrinsic sampler on device, the --zero-likelihood stand-in building its base array with xpy_default=cupy, and the supplementary factor evaluating on device. NOT the GPU signal likelihood -- --zero-likelihood replaces likelihood_function outright, so NoLoop never runs. Measured, not assumed: AV with and without --vectorized --gpu --force-xpy returns bit-identical lnZ, and that inertness is now its own test. THE BLOCKER DID NOT REPRODUCE. "CUDA_VISIBLE_DEVICES='' is required, the scalar AV path raises Unsupported dtype float128" is true where it was written (test_psi_marginalization.py) and false here: under --zero-likelihood the stand-in builds lnL with xpy.zeros, float64 on device, and the scalar likelihood that would have built a RiftFloat never runs. AV on an A100 completes and lands 1.5 sigma from the closed form. TWO SAMPLERS CANNOT RUN ON A DEVICE, both pre-existing, both reproduced with no supplementary factor, both `TypeError: Unsupported type ` from a cupy ufunc, and both invisible in production because the driver prints FAILED ANALYSIS and exits 0: portfolio RIFT/likelihood/vectorized_general_tools.py histogram(), where xpy.maximum gets a device blank_array and numpy samples adaptive_cartesian RIFT/integrators/mcsampler.py, fval*joint_p_prior/joint_p_s, device integrand against numpy priors They are RECORDED lanes, pinned by site and exception, not skips: skipping is what hid them. The test reddens if one starts working and if the failure moves. The GMM device lane is A=0.75 B=3, not A=8 B=2. This file already records that GMM at A=8 with the inclination term is an n_eff lottery -- 2 of 8 seeds collapsed to sigma 0.028 and 0.073, over MAX_SIGMA -- which is why no CPU lane runs it. Eight clean GPU seeds do not refute a lottery; the sweep that found it was also eight seeds. Calibration, 8 seeds on ldas-pcdev2 slot 0 (A100, cc 8.0), cupy 12.0.0, recorded in the table with its host twins. Worst |z| across the whole gate moves 2.15 -> 2.30, so Z_TOLERANCE's margin is 2.2x rather than 2.3x; no constant changed. Mutation-checked on the A100: a silent host fallback reddens all six device lanes; a recorded failure that starts passing reddens; a failure that moves site reddens. Co-Authored-By: Claude Opus 5 --- .travis/test-integrate.sh | 9 +- .../integrators/make_e2e_calibration.py | 24 ++ .../Code/test/test_e2e_analytic_pipeline.py | 283 ++++++++++++++++-- 3 files changed, 294 insertions(+), 22 deletions(-) diff --git a/.travis/test-integrate.sh b/.travis/test-integrate.sh index 686786696..277d9f112 100755 --- a/.travis/test-integrate.sh +++ b/.travis/test-integrate.sh @@ -151,10 +151,15 @@ python -m pytest -q "$_ZLSTANDIN_TESTS" # driver's own default --sampler-method adaptive_cartesian. Both were invisible because the # driver catches the exception, prints FAILED ANALYSIS and EXITS 0. It also caught ILE # --sampler-method GMM returning an evidence ~30 nats wrong; #359 fixed that, and GMM is now a -# lane here rather than a recorded defect. Needs no network, no real event and no GPU; about +# lane here rather than a recorded defect. Needs no network and no real event; about # 8-12 s per ILE arm plus ~15 s once for the distance-marginalization lookup table. +# Its section 5 runs the same answers with a GPU VISIBLE and asserts the child reached one. +# Those 7 lanes SKIP here, because the runners have no cupy -- so a green run in CI has NOT +# exercised the device path. Run the file by hand on a CIT GPU node with CUDA_VISIBLE_DEVICES +# pinned to a slot the installed cupy supports; see the gpu_slot fixture. Skipped tests are +# still collected, so the count below is the same everywhere. _E2E_TESTS=MonteCarloMarginalizeCode/Code/test/test_e2e_analytic_pipeline.py -_E2E_EXPECTED=17 +_E2E_EXPECTED=24 _E2E_FOUND=$(python -m pytest -q --collect-only "$_E2E_TESTS" 2>/dev/null | grep -c '::' || true) if [ "$_E2E_FOUND" -ne "$_E2E_EXPECTED" ]; then echo "e2e analytic gate: collected $_E2E_FOUND tests, expected $_E2E_EXPECTED" >&2 diff --git a/MonteCarloMarginalizeCode/Code/test/expensive_before_merging/integrators/make_e2e_calibration.py b/MonteCarloMarginalizeCode/Code/test/expensive_before_merging/integrators/make_e2e_calibration.py index 0e782d9e0..47ea6354e 100644 --- a/MonteCarloMarginalizeCode/Code/test/expensive_before_merging/integrators/make_e2e_calibration.py +++ b/MonteCarloMarginalizeCode/Code/test/expensive_before_merging/integrators/make_e2e_calibration.py @@ -53,6 +53,15 @@ ("distance-marginalized", _AV, 8.0, 2.0, {"_needs_dmarg": True}), ("adaptive_cartesian, --n-max 60000", ["--sampler-method", "adaptive_cartesian"], 8.0, 2.0, dict(n_max=60000)), + # DEVICE lanes. They need --gpu-slot, and are refused rather than skipped without it, for + # the same reason as a --lane typo: a table that quietly stops covering four of its rows + # under a summary line that reads clean is worse than no table. + ("GPU prior-only, AV", _AV, None, 0.0, {"_needs_gpu": True}), + ("GPU prior-only, GMM", _GMM, None, 0.0, {"_needs_gpu": True}), + ("GPU A=8 B=2, AV", _AV, 8.0, 2.0, {"_needs_gpu": True}), + # NOT A=8 B=2: that is the n_eff lottery recorded at the end of the gate's CALIBRATION + # section. Mirrors the CPU "A=0.75 B=3, GMM" row instead. + ("GPU A=0.75 B=3, GMM", _GMM, 0.75, 3.0, {"_needs_gpu": True}), ] @@ -77,6 +86,10 @@ def main(): ap.add_argument("--seeds", type=int, default=8) ap.add_argument("--first-seed", type=int, default=1000) ap.add_argument("--lane", default=None, help="substring; measure only matching lanes") + ap.add_argument("--gpu-slot", default=None, + help="CUDA slot for the GPU lanes, as the CHILD should see it. Probe it " + "first: the CIT slot map moves and cupy 12.0.0 cannot build a kernel " + "for the Blackwell cards. Without it the GPU lanes are refused.") args = ap.parse_args() # Lane selection BEFORE anything expensive or on-disk. It depends on nothing but argv, and @@ -87,6 +100,12 @@ def main(): # The index is carried alongside each lane, and it is the index into LANES rather than into # this filtered list, so --lane does not renumber the output tags. selected = [(i, L) for i, L in enumerate(LANES) if not args.lane or args.lane in L[0]] + if args.gpu_slot is None: + refused = [L[0].strip() for _i, L in selected if L[4].get("_needs_gpu")] + if refused: + raise SystemExit("--gpu-slot is required to measure:\n %s\nProbe a slot this " + "cupy can build a kernel for, or narrow with --lane." + % "\n ".join(refused)) if not selected: # Exiting 0 having measured nothing, under a summary line that reads like a clean # result, is the exact shape this whole branch exists to remove. @@ -109,6 +128,11 @@ def main(): kw = dict(kw) if kw.pop("_needs_dmarg", False): kw.update(_dmarg_extra(event, out)) + if kw.pop("_needs_gpu", False): + # expect_device=True, so a lane that fell back to the host FAILS instead of + # contributing host numbers to a row labelled GPU. + kw.update(cuda=args.gpu_slot, expect_device=True, + extra=tuple(kw.get("extra", ())) + gate._GPU_FLAGS) exact = 0.0 if a is None else gate._exact(a, b) zs, sigs, neffs = [], [], [] for seed in seeds: diff --git a/MonteCarloMarginalizeCode/Code/test/test_e2e_analytic_pipeline.py b/MonteCarloMarginalizeCode/Code/test/test_e2e_analytic_pipeline.py index f59b0df66..97ba64cea 100644 --- a/MonteCarloMarginalizeCode/Code/test/test_e2e_analytic_pipeline.py +++ b/MonteCarloMarginalizeCode/Code/test/test_e2e_analytic_pipeline.py @@ -33,8 +33,17 @@ measuring something else. If a lane will not converge, scope it or leave it out; do not add these. +THE DEVICE. Section 5 runs the prior-only and factor answers again with a GPU VISIBLE, and +asserts the child reached it. RIFT binds its array module at import from whether cupy imports, +not from --gpu, so on a GPU node production is on the device path by default and the rest of +this file -- which pins CUDA_VISIBLE_DEVICES="" -- said nothing about it. Two samplers cannot +run there at all; they are recorded as known-failing lanes rather than skipped, so the hole is +visible. What section 5 does NOT cover is the GPU signal likelihood: --zero-likelihood +replaces likelihood_function outright, so the NoLoop path never runs. + WHAT EACH ARM COSTS. About 8-12 s of one core per ILE arm, plus ~15 s once for the -distance-marginalization lookup table. No network, no real event, no GPU. +distance-marginalization lookup table. No network, no real event. The CPU lanes need no GPU; +the device lanes skip without one, and a skip there is not a pass. """ import json import os @@ -141,20 +150,98 @@ def dmarg_table(event): return out +# --------------------------------------------------------------------------------------- +# the device + +# Printed by the driver's own preamble only AFTER `cupy.array(5)` succeeds, so it cannot appear +# on a host. That makes it the one cheap proof a "GPU lane" was not a CPU lane. +_DEVICE_MARKER = "cupy memory [total, available]" + +# The production shape of a deliberate GPU invocation. Note these are NOT what puts the run on +# a device: xpy_default is bound at IMPORT from whether cupy imports, so a device that is merely +# visible is already in use. That is why the bare-flags lane below exists as well. +_GPU_FLAGS = ("--vectorized", "--gpu", "--force-xpy") + +# Every verdict is ONE line beginning with a known word, and every message is flattened, because +# cupy's ImportError is a multi-line banner: reading "the last line of the probe's output" turned +# it into the skip reason "no usable GPU (If you installed CuPy via whee)". +_GPU_PROBE = r""" +import numpy as np +def _flat(e): + return ("%s: %s" % (type(e).__name__, e)).replace("\n", " ")[:150] +try: + import cupy +except Exception as e: + print("VERDICT NOCUPY %s" % _flat(e)); raise SystemExit(0) +bad = [] +for d in range(cupy.cuda.runtime.getDeviceCount()): + try: + with cupy.cuda.Device(d): + cupy.asnumpy(cupy.cos(cupy.asarray(np.zeros(2), dtype=float))) + except Exception as e: + bad.append("%d:%s" % (d, type(e).__name__)); continue + print("VERDICT SLOT %d" % d); raise SystemExit(0) +print("VERDICT NOSLOT %s" % (",".join(bad) or "no devices at all")) +""" + + +@pytest.fixture(scope="module") +def gpu_slot(): + """A CUDA slot this cupy can actually build a kernel for, as the child should see it. + + TWO failures that look alike, and conflating them is how a device lane quietly never runs. + `import cupy` fails outright where there is no CUDA runtime (ldas-grid, CI runners). It + SUCCEEDS on the CIT GPU head nodes while the visible device is one this cupy cannot compile + for: ldas-pcdev13 slots 0-2 and all of ldas-pcdev11 are Blackwell cc 12.0, and cupy 12.0.0 + answers `nvrtc: error: invalid value for --gpu-architecture`. So the probe RUNS a kernel + instead of trusting the import, and it probes at dispatch, because the slot map moves. + + It probes in a SUBPROCESS so this pytest process never holds a CUDA context while the ILE + children run, and it re-indexes: the probe's index is into the CURRENTLY VISIBLE list, so + when the parent already has CUDA_VISIBLE_DEVICES set, what the child is given is that + list's d-th entry, not d. + + A SKIP HERE IS NOT A PASS.""" + proc = subprocess.run([sys.executable, "-c", _GPU_PROBE], env=_child_env(cuda=None), + stdout=subprocess.PIPE, stderr=subprocess.STDOUT, timeout=600) + verdicts = [l for l in proc.stdout.decode().splitlines() if l.startswith("VERDICT ")] + line = verdicts[-1][len("VERDICT "):] if verdicts else ( + "probe produced no verdict (rc=%d): %s" % (proc.returncode, + proc.stdout.decode()[-300:])) + if not line.startswith("SLOT "): + pytest.skip("no usable GPU for these lanes -- %s. A skip is NOT a pass: pin " + "CUDA_VISIBLE_DEVICES to a slot the installed cupy supports and rerun." + % line) + d = int(line.split()[1]) + visible = os.environ.get("CUDA_VISIBLE_DEVICES") + if visible: + return visible.split(",")[d].strip() + return str(d) + + # --------------------------------------------------------------------------------------- # running one ILE arm -def _child_env(**extra): +def _child_env(cuda="", **extra): # Every RIFT_* variable is stripped, not just the one that bit. RIFT_HYPERPIPELINE_FORMAT # moves lnL/sigma to columns 0/1 and adds a header, under which the tail-column read below # returns a spin component (0.0) as sigma and the z-test passes VACUOUSLY. Others # (RIFT_LOWLATENCY, RIFT_NO_GWSIGNAL, RIFT_GPU_*) change the code path. A gate whose # answer depends on the invoking shell's environment is not a gate. env = {k: v for k, v in os.environ.items() if not k.startswith("RIFT_")} - # CUDA_VISIBLE_DEVICES="" is required, and is not this test's problem: on a host where - # cupy sees a GPU the scalar AV path raises "Unsupported dtype float128" (see the same - # note in test_psi_marginalization.py). - env["CUDA_VISIBLE_DEVICES"] = "" + # CUDA_VISIBLE_DEVICES is pinned so the answer does not depend on the invoking shell. "" + # is the default and what every CPU lane uses. The device lanes pass a real slot; see + # gpu_slot. cuda=None leaves whatever the parent has, which only the device PROBE wants. + # + # The empty default used to carry "required, because on a host where cupy sees a GPU the + # scalar AV path raises Unsupported dtype float128". That is true where it was written + # (test_psi_marginalization.py) and NOT true here, measured: this gate runs + # --zero-likelihood, so make_zero_likelihood_standin builds lnL with xpy.zeros -- float64 + # on the device -- and the scalar likelihood that would have built a RiftFloat never runs. + # Verified on ldas-pcdev2 (A100) 2026-09-17: --sampler-method AV with the device visible + # completes and lands 1.5 sigma from the closed form. Do not re-copy that sentence here. + if cuda is not None: + env["CUDA_VISIBLE_DEVICES"] = cuda env["OMP_NUM_THREADS"] = "1" env["MPLBACKEND"] = "Agg" # THIS tree, not whatever RIFT is installed: the driver is run by absolute path out of the @@ -193,12 +280,15 @@ def _read_result(d, tag): return lnL, sigma, neff -def _run_ile(event, tag, sampler_args, a_coeff=None, b_coeff=0.0, incl_is_cosine=False, - n_max=20000, n_eff=250, seed=1000, extra=()): - """One ILE job as a subprocess. Returns (lnL, sigma_lnL, n_eff).""" +def _invoke_ile(event, tag, sampler_args, a_coeff=None, b_coeff=0.0, incl_is_cosine=False, + n_max=20000, n_eff=250, seed=1000, extra=(), cuda=""): + """Run one ILE arm. Returns (dir, returncode, stdout text, whether the result row exists). + + Split out of _run_ile because the recorded known-failing device lanes need the CHILD'S LOG, + not a pytest failure: what they pin is the shape of a failure that is real today.""" d = event["dir"] / tag d.mkdir(exist_ok=True) - env = _child_env() + env = _child_env(cuda=cuda) cmd = [sys.executable, ILE, "--cache-file", str(event["cache"]), "--channel-name", "H1=FAKE-STRAIN", "--psd-file", "H1=%s" % event["psd"], @@ -222,10 +312,26 @@ def _run_ile(event, tag, sampler_args, a_coeff=None, b_coeff=0.0, incl_is_cosine # The driver CATCHES an exception from analyze_event, prints "FAILED ANALYSIS", skips the # point and EXITS 0 -- so in a DAG a crashed configuration is silent. Absence of the # output row, not the exit code, is what says the run failed. - row = d / ("%s_0_.dat" % tag) - if proc.returncode != 0 or not row.exists(): + return d, proc.returncode, proc.stdout.decode(), (d / ("%s_0_.dat" % tag)).exists() + + +def _run_ile(event, tag, sampler_args, expect_device=False, **kw): + """One ILE job as a subprocess. Returns (lnL, sigma_lnL, n_eff). + + expect_device asserts the child really reached a GPU. Without it a "GPU lane" that + silently fell back to the host is indistinguishable from one that ran, which is the whole + failure mode these lanes exist to remove: the marker is printed only after + `cupy.array(5)` succeeds in the driver's own preamble, so it cannot be true on a host.""" + d, rc, out, have_row = _invoke_ile(event, tag, sampler_args, **kw) + if rc != 0 or not have_row: pytest.fail("ILE (%s) exited %d and wrote no result row; tail:\n%s" - % (tag, proc.returncode, proc.stdout.decode()[-3000:])) + % (tag, rc, out[-3000:])) + if expect_device: + assert _DEVICE_MARKER in out, ( + "%s was supposed to run on a device, but the child never printed %r, so cupy did " + "not initialise there and this lane measured the HOST path under a GPU name. " + "CUDA_VISIBLE_DEVICES reached the child as %r." + % (tag, _DEVICE_MARKER, kw.get("cuda"))) return _read_result(d, tag) @@ -381,6 +487,122 @@ def test_adaptive_cartesian(event): _assert_lnZ(tag, lnL, sigma, neff, _exact(a, b)) +# --------------------------------------------------------------------------------------- +# 5. the same answers, on a real GPU +# +# WHY THIS IS NOT THE SAME TEST TWICE. Every lane above pins CUDA_VISIBLE_DEVICES="", so the +# whole file used to say nothing about the device path -- and RIFT picks that path up from +# whether cupy IMPORTS, not from --gpu, so on a GPU node production takes it by default. These +# lanes run with a device visible and assert the child reached it. +# +# WHAT THEY COVER, precisely: the extrinsic sampler on device, the --zero-likelihood stand-in +# building its base array with xpy_default=cupy, and the supplementary factor evaluating on +# device. They do NOT cover the GPU SIGNAL likelihood: --zero-likelihood replaces +# likelihood_function outright, so the NoLoop path never runs. Measured, not assumed: AV with +# and without _GPU_FLAGS returns bit-identical lnZ, because those flags only choose a likelihood +# that this configuration does not evaluate. + +_GPU_SAMPLERS = ["AV", "GMM"] + +# Coefficients per device lane, and they are NOT the same pair for both samplers on purpose. +# Each mirrors the CPU lane that sampler already runs, so a device row is comparable with a host +# row in the table below, and neither lands on a configuration this file records as unstable. +# +# GMM is A=0.75 B=3, not A=8 B=2. The note at the end of the CALIBRATION section records that +# GMM at A=8 WITH the inclination term is an n_eff LOTTERY: over eight seeds, two collapsed to +# n_eff 73.5 and 14.5 with sigma 0.028 and 0.073, and 0.073 is over MAX_SIGMA. The evidence was +# right on every seed; the error bar was not. That is why no CPU lane runs it, and a device lane +# that ran it would be a flake with a ~1-in-4 seed. Eight clean GPU seeds do not refute a +# lottery -- the CPU sweep that FOUND it was also eight seeds. Both lanes keep a B term, so +# mis-routing inclination is still detectable; that is what the B term is for. +_GPU_LANE_COEFFS = {"AV": (8.0, 2.0), "GMM": (0.75, 3.0)} + + +@pytest.mark.parametrize("sampler", _GPU_SAMPLERS) +def test_gpu_zero_likelihood_alone_gives_ln_Z_zero(event, gpu_slot, sampler): + """The prior-only answer, on device. ln Z = 0 exactly, for a normalized extrinsic prior.""" + tag = "gpu_zero_%s" % sampler + lnL, sigma, neff = _run_ile(event, tag, SAMPLER_ARGS[sampler] + list(_GPU_FLAGS), + a_coeff=None, cuda=gpu_slot, expect_device=True) + _assert_lnZ(tag, lnL, sigma, neff, 0.0) + + +@pytest.mark.parametrize("sampler", _GPU_SAMPLERS) +def test_gpu_analytic_factor_marginal_is_exact(event, gpu_slot, sampler): + """The factor's closed-form marginal, on device: ln Z = ln I0(A) + ln(sinh(B)/B). + + See _GPU_LANE_COEFFS for why the two samplers get different coefficients.""" + a, b = _GPU_LANE_COEFFS[sampler] + tag = "gpu_factor_%s" % sampler + lnL, sigma, neff = _run_ile(event, tag, SAMPLER_ARGS[sampler] + list(_GPU_FLAGS), + a_coeff=a, b_coeff=b, cuda=gpu_slot, expect_device=True) + _assert_lnZ(tag, lnL, sigma, neff, _exact(a, b)) + + +def test_the_gpu_flags_do_not_decide_the_backend(event, gpu_slot): + """A VISIBLE DEVICE IS ALREADY A GPU RUN; --gpu does not opt in and its absence does not opt + out. xpy_default is bound at import from whether cupy imports, so someone who runs this + driver on a GPU node without --gpu is on the device path anyway. + + Pinned because it is the assumption the lanes above rest on, and because it is the opposite + of what the flag names suggest. If this ever fails, the backend became flag-driven, which + is a fix -- update these lanes rather than deleting the test.""" + kw = dict(a_coeff=8.0, b_coeff=2.0, cuda=gpu_slot, expect_device=True) + bare = _run_ile(event, "gpu_backend_bare", SAMPLER_ARGS["AV"], **kw) + flagged = _run_ile(event, "gpu_backend_flagged", + SAMPLER_ARGS["AV"] + list(_GPU_FLAGS), **kw) + assert bare == flagged, ( + "--vectorized --gpu --force-xpy changed the answer (%r vs %r). Under --zero-likelihood " + "they select a signal likelihood that never runs, so they were expected to be inert; " + "something else now depends on them." % (bare, flagged)) + + +# Device lanes that DO NOT WORK today, recorded with the shape of the failure so the hole is +# visible instead of hidden behind CUDA_VISIBLE_DEVICES="". Both are host/device mixes, both +# reproduce with NO supplementary factor, and neither is caused by anything in this file: +# +# portfolio (AV + AC) RIFT/likelihood/vectorized_general_tools.py, histogram(): +# `xpy.maximum(samples, blank_array)` with blank_array built by +# xpy.zeros on device and `samples` arriving as a numpy array from +# mcsamplerGPU.compute_hist. +# adaptive_cartesian RIFT/integrators/mcsampler.py, `fval*joint_p_prior/joint_p_s`: the +# integrand is on device, the two prior arrays are numpy. +# +# Both surface as `TypeError: Unsupported type ` from a cupy ufunc, and +# both are invisible in production because the driver catches the exception, prints FAILED +# ANALYSIS and EXITS 0. Measured on ldas-pcdev2 (A100, cc 8.0), cupy 12.0.0, 2026-09-17. +_GPU_KNOWN_FAILING = { + "portfolio": (_PORTFOLIO, {}, "vectorized_general_tools.py"), + "adaptive_cartesian": (["--sampler-method", "adaptive_cartesian"], dict(n_max=60000), + "integrators/mcsampler.py"), +} + + +@pytest.mark.parametrize("sampler", sorted(_GPU_KNOWN_FAILING)) +def test_the_gpu_samplers_that_do_not_work_still_fail_the_known_way(event, gpu_slot, sampler): + """A RECORDED DEFECT, not an excused one. + + Two samplers cannot run on a device at all. Skipping them would leave the gate reporting + green over a broken configuration, which is what pinning CUDA_VISIBLE_DEVICES="" did for + the whole file. So the failure is pinned by its SITE and its EXCEPTION, and this test goes + red in both directions: if someone fixes one, promote it into _GPU_SAMPLERS and delete the + entry; if the failure moves somewhere else, that is a different defect and wants reading.""" + args, kw, site = _GPU_KNOWN_FAILING[sampler] + tag = "gpu_known_fail_%s" % sampler + _d, _rc, out, have_row = _invoke_ile(event, tag, args + list(_GPU_FLAGS), + a_coeff=8.0, b_coeff=2.0, cuda=gpu_slot, **kw) + assert _DEVICE_MARKER in out, ( + "%s never reached a device, so this says nothing about the GPU path" % tag) + assert not have_row, ( + "%s SUCCEEDED on a device. If that is a fix, move %r into _GPU_SAMPLERS, delete its " + "_GPU_KNOWN_FAILING entry and the paragraph above it, and re-measure the calibration " + "table for the new lane." % (tag, sampler)) + assert "Unsupported type " in out and site in out, ( + "%s still fails on a device, but not in the recorded way: expected a cupy ufunc " + "TypeError from %s. A different failure is a different defect; read the log.\n%s" + % (tag, site, out[-2500:])) + + # --------------------------------------------------------------------------------------- # CALIBRATION # @@ -419,13 +641,34 @@ def test_adaptive_cartesian(event): # distance-marginalized 2.15 0.0326 336 # adaptive_cartesian, --n-max 60000 1.35 0.0305 250 # -# Z_TOLERANCE = 5 is 2.3x the worst |z| seen (2.15, distance-marginalized). That lane sets -# the constant, so it was re-measured at eight seeds after the stand-in -# started passing xpy= to factors that accept it: max |z| 2.15, max sigma -# 0.0326, least n_eff 336, unchanged, and seed 1000 bit-identical. -# MAX_SIGMA = 0.06 is 1.7x the worst sigma seen (0.0353). 5 * MAX_SIGMA is a 0.30-nat band. -# MIN_NEFF = 30 is 5.4x below the worst n_eff seen (163). See its comment for why it is -# this loose. +# THE DEVICE LANES, measured separately because they need a GPU: eight seeds (1000-1007) on +# ldas-pcdev2 slot 0 (A100, cc 8.0), cupy 12.0.0, at 5bb8da02b, with +# +# python make_e2e_calibration.py --seeds 8 --lane "GPU " --gpu-slot 0 +# +# The GMM row is A=0.75 B=3 and not A=8 B=2 deliberately; see _GPU_LANE_COEFFS. Each device row +# sits on top of its host twin, which is the point: the device path is not a different answer. +# +# lane max |z| max sigma min n_eff +# GPU prior-only, AV 1.50 0.0134 1343 +# GPU prior-only, GMM 0.93 0.0134 1352 +# GPU A=8 B=2, AV 2.30 0.0325 257 +# GPU A=0.75 B=3, GMM 1.49 0.0235 389 +# +# Host twins for those four, from the table above: 2.03/0.0134/1349, 1.88/0.0134/1335, +# 1.06/0.0332/247, 1.34/0.0235/381. Same sigma to three digits on three of the four; the |z| +# values differ because they are draws, not constants. +# +# Z_TOLERANCE = 5 is 2.2x the worst |z| seen, which is now 2.30 on the GPU A=8 B=2 AV lane; +# the worst host lane is 2.15 (distance-marginalized), re-measured at eight +# seeds after the stand-in started passing xpy= to factors that accept it +# (max |z| 2.15, max sigma 0.0326, least n_eff 336, unchanged, seed 1000 +# bit-identical). Adding the device lanes moved the margin from 2.3x to +# 2.2x and nothing else. +# MAX_SIGMA = 0.06 is 1.7x the worst sigma seen (0.0353, host). The worst device sigma is +# 0.0325. 5 * MAX_SIGMA is a 0.30-nat band. +# MIN_NEFF = 30 is 5.4x below the worst n_eff seen (163, host; 257 on device). See its +# comment for why it is this loose. # # WHAT THE GATE HAS TO SEPARATE A CORRECT RUN FROM. Each row was run on this fixture, not # argued. The smallest is 6.3 sigma, against a tolerance of 5 and a worst observed draw of 2.15: From f0e0d3502a9608573ce51cb51a020fd451247cff Mon Sep 17 00:00:00 2001 From: Richard O'Shaughnessy Date: Fri, 18 Sep 2026 05:47:32 -0700 Subject: [PATCH 4/6] two samplers could not run with a GPU visible: convert on the side that can move Both died in a cupy ufunc handed a numpy array, both reproduce with no supplementary likelihood factor, and both were invisible in production because the ILE driver catches the exception, prints FAILED ANALYSIS and exits 0. Recorded by site and exception in test_e2e_analytic_pipeline.py section 5 and fixed here. The two are NOT the same defect and do not convert the same way. portfolio (AV + AC). mcsamplerPortfolio pins self.xpy = numpy and aggregates on the host, so an AC member built on cupy was handed host samples through external_rvs, and vectorized_general_tools.histogram's xpy.maximum(samples, xpy.zeros(...)) raised. Fixed in mcsamplerGPU.compute_hist, which puts the samples and weights on the MEMBER'S backend: the histogram has to end up on the device anyway (histogram_edges/histogram_cdf are preallocated there and the draws read them), and nothing here carries precision worth keeping on the host -- the coordinates are float64 and the histogram is a proposal density, not an estimator. Fixed at the sampler's own boundary rather than in the shared helper; compute_hist is the only production caller of histogram(), and the symmetric conversion for member DRAWS is already at the same boundary in the portfolio. adaptive_cartesian. fval*joint_p_prior/joint_p_s with a device integrand and host priors. Fixed the other way, in mcsampler.integrate: everything below that line is host numpy, and joint_p_prior is deliberately RiftFloat, which cupy has no dtype for -- so the host side is the one that cannot move. The integrand is copied back once, right after the call, through a new to_host() that mirrors infer_array_module and keeps this file's numpy-only import list. A host input is returned unchanged, so the numpy path is untouched. Casting the ARGUMENT instead -- numpy.asarray(x) on a cupy array -- is what cupy refuses, and is the rewrite both comments warn against. MEASURED, ldas-pcdev2, CUDA slot 0 (A100 cc 8.0, identified by running a kernel: the slot map on that node has moved since it was last written down), cupy 12.0.0. Before: the two recorded lanes fail at their recorded sites. After: both complete and land on the closed form. Mutation-checked one fix at a time -- with only mcsamplerGPU reverted, portfolio fails again at vectorized_general_tools.py and adaptive_cartesian stays fixed; with only mcsampler reverted, the reverse. Neither fix is doing the other's work. Co-Authored-By: Claude Opus 5 --- .../Code/RIFT/integrators/mcsampler.py | 25 +++++++++++++++++++ .../Code/RIFT/integrators/mcsamplerGPU.py | 18 +++++++++++++ 2 files changed, 43 insertions(+) diff --git a/MonteCarloMarginalizeCode/Code/RIFT/integrators/mcsampler.py b/MonteCarloMarginalizeCode/Code/RIFT/integrators/mcsampler.py index 98272e1fd..726661cb0 100644 --- a/MonteCarloMarginalizeCode/Code/RIFT/integrators/mcsampler.py +++ b/MonteCarloMarginalizeCode/Code/RIFT/integrators/mcsampler.py @@ -562,6 +562,16 @@ def integrate(self, func, *args, **kwargs): else: fval = func(**unpacked) # Chris' original plan: note this insures the function arguments are tied to the parameters, using a dictionary. + # THE INTEGRAND CONVERTS, NOT THE ACCUMULATORS. On a GPU node the ILE likelihood + # returns a cupy array (xpy_default binds at import from whether cupy imports, so + # a merely visible device is enough), while everything below here is host numpy: + # numpy.hstack, statutils update/finalize, math.isnan -- and joint_p_prior is + # deliberately RiftFloat, which cupy has no dtype for at all. So the host side is + # the one that cannot move. Without this, `fval*joint_p_prior/joint_p_s` raised + # "TypeError: Unsupported type " from a cupy ufunc and the + # driver swallowed it, printed FAILED ANALYSIS and exited 0. + fval = to_host(fval) + # # Check if there is any practical contribution to the integral # @@ -1035,6 +1045,21 @@ def infer_array_module(x, xpy=None): return numpy +def to_host(x): + """Return `x` as a host (numpy) array, copying it off a device backend if it is on one. + + numpy.asarray(cupy_array) RAISES -- cupy refuses implicit host conversion -- so the copy + has to go through the device module's own asnumpy. The module is looked up from the + value, the same way infer_array_module does it, so this file keeps its numpy-only import + list. A host input is returned UNCHANGED, not re-wrapped: the numpy path must stay bit + for bit what it was. + """ + mod = infer_array_module(x) + if mod is not numpy and hasattr(mod, "asnumpy"): + return mod.asnumpy(x) + return x + + def clip_angle_limits(lo, hi, kind): """Clip an angular range [lo,hi] (radians) to the physical domain of `kind` ('declination' -> [-pi/2,pi/2], 'inclination' -> [0,pi]). diff --git a/MonteCarloMarginalizeCode/Code/RIFT/integrators/mcsamplerGPU.py b/MonteCarloMarginalizeCode/Code/RIFT/integrators/mcsamplerGPU.py index 6750c57d0..fb1d112a9 100644 --- a/MonteCarloMarginalizeCode/Code/RIFT/integrators/mcsamplerGPU.py +++ b/MonteCarloMarginalizeCode/Code/RIFT/integrators/mcsamplerGPU.py @@ -326,6 +326,24 @@ def setup_hist_single_param(self, x_min, x_max, n_bins, param): def compute_hist(self, x_samples, param,weights=None,floor_level=0): + # PUT THE INPUTS ON THIS SAMPLER'S BACKEND FIRST. As a portfolio member the samples + # arrive from the aggregator's host _rvs (mcsamplerPortfolio pins self.xpy = numpy) + # while self.xpy here is cupy, and the first cupy ufunc downstream -- the + # xpy.maximum(samples, xpy.zeros(...)) clip in vectorized_general_tools.histogram -- + # raised "TypeError: Unsupported type ", killing every + # portfolio run with an AC member on a GPU node behind FAILED ANALYSIS and exit 0. + # + # DEVICE is the side to converge on here. The result has to end up there anyway + # (setup_hist_single_param preallocates histogram_edges/histogram_cdf with self.xpy, + # and cdf_inverse_from_hist is read by device draws), and nothing here carries + # precision worth keeping on the host: the coordinates are float64 and this histogram + # is a *proposal* density, not an estimator (see _bincount_weighted). Contrast + # mcsampler.py, whose accumulators are deliberately RiftFloat and so convert the + # other way. asarray is a no-op when the input is already on this backend, so the + # standalone GPU and pure-numpy paths are untouched. + x_samples = self.xpy.asarray(x_samples) + if weights is not None: + weights = self.xpy.asarray(weights) # Rescale the samples to [0, 1] y_samples = ( (x_samples - self.x_min[param]) / self.x_max_minus_min[param] From 251a55202ca65b122e1f120924d2dbcad1c8f2cf Mon Sep 17 00:00:00 2001 From: Richard O'Shaughnessy Date: Fri, 18 Sep 2026 05:47:40 -0700 Subject: [PATCH 5/6] e2e gate: portfolio and adaptive_cartesian are first-class device lanes now They were recorded in _GPU_KNOWN_FAILING, which asserts the failure by site and exception and goes red when one starts passing. Both now pass, so the dict, its paragraph and the test that carried them are gone, and both samplers join _GPU_SAMPLERS with lane coefficients. Each device lane mirrors the host lane that sampler already runs, so its row sits on top of a host row: portfolio takes A=0.75 B=3 (its own host lane with a B term, and NOT A=8 B=2, which is the recorded GMM n_eff lottery), adaptive_cartesian takes A=8 B=2 with the --n-max its host lane needed, now named once as _AC_N_MAX and reached from the device lane through _GPU_LANE_KW. Every lane keeps a B term, so mis-routing inclination stays detectable. CALIBRATION, eight seeds (1000-1007) on ldas-pcdev2 CUDA slot 0 (A100 cc 8.0, probed with a kernel), cupy 12.0.0: GPU prior-only, portfolio 2.22 0.0104 2335 (host 1.68/0.0104/2316) GPU A=0.75 B=3, portfolio 2.32 0.0228 411 (host 1.54/0.0227/ 392) GPU prior-only, adaptive_cartesian 2.39 0.0091 3262 (no host twin) GPU A=8 B=2, adaptive_cartesian 1.35 0.0305 250 (host 1.35/0.0305/ 250) The last row is its host twin exactly, which is expected and now stated in the file as a tripwire: under --zero-likelihood the only device work is xpy.zeros, and the integrand is copied to the host before anything else touches it, so both arms run the same host arithmetic off the same seed. Drift there means real device arithmetic appeared. The four AV/GMM device rows were re-derived in the same sweep and came back identical to the values measured before the two fixes, so the fixes do not move the lanes that already worked. Worst |z| over all device lanes is now 2.39, leaving Z_TOLERANCE = 5 at 2.1x; MAX_SIGMA and MIN_NEFF are unmoved. Collected count 24 -> 26 in .travis/test-integrate.sh (four lanes added, two recorded failures removed), and the generator grows the four rows so the table stays re-derivable. RESULTS. ldas-pcdev2, slot 0, cupy 12.0.0: 26 passed, nothing skipped. ldas-grid, cupy absent: 17 passed, 9 skipped -- every device lane, with "a skip is NOT a pass" in the reason. test-ci-roster.py PASS; test-core-units.sh PASS (580 passed, 13 skipped) on ldas-grid. Co-Authored-By: Claude Opus 5 --- .travis/test-integrate.sh | 4 +- .../integrators/make_e2e_calibration.py | 14 +- .../Code/test/test_e2e_analytic_pipeline.py | 139 ++++++++---------- 3 files changed, 79 insertions(+), 78 deletions(-) diff --git a/.travis/test-integrate.sh b/.travis/test-integrate.sh index 277d9f112..6b2550e43 100755 --- a/.travis/test-integrate.sh +++ b/.travis/test-integrate.sh @@ -154,12 +154,12 @@ python -m pytest -q "$_ZLSTANDIN_TESTS" # lane here rather than a recorded defect. Needs no network and no real event; about # 8-12 s per ILE arm plus ~15 s once for the distance-marginalization lookup table. # Its section 5 runs the same answers with a GPU VISIBLE and asserts the child reached one. -# Those 7 lanes SKIP here, because the runners have no cupy -- so a green run in CI has NOT +# Those 9 lanes SKIP here, because the runners have no cupy -- so a green run in CI has NOT # exercised the device path. Run the file by hand on a CIT GPU node with CUDA_VISIBLE_DEVICES # pinned to a slot the installed cupy supports; see the gpu_slot fixture. Skipped tests are # still collected, so the count below is the same everywhere. _E2E_TESTS=MonteCarloMarginalizeCode/Code/test/test_e2e_analytic_pipeline.py -_E2E_EXPECTED=24 +_E2E_EXPECTED=26 _E2E_FOUND=$(python -m pytest -q --collect-only "$_E2E_TESTS" 2>/dev/null | grep -c '::' || true) if [ "$_E2E_FOUND" -ne "$_E2E_EXPECTED" ]; then echo "e2e analytic gate: collected $_E2E_FOUND tests, expected $_E2E_EXPECTED" >&2 diff --git a/MonteCarloMarginalizeCode/Code/test/expensive_before_merging/integrators/make_e2e_calibration.py b/MonteCarloMarginalizeCode/Code/test/expensive_before_merging/integrators/make_e2e_calibration.py index 47ea6354e..8ea7967b3 100644 --- a/MonteCarloMarginalizeCode/Code/test/expensive_before_merging/integrators/make_e2e_calibration.py +++ b/MonteCarloMarginalizeCode/Code/test/expensive_before_merging/integrators/make_e2e_calibration.py @@ -29,7 +29,7 @@ import test_e2e_analytic_pipeline as gate # noqa: E402 -_AV, _PORTFOLIO, _GMM = gate._AV, gate._PORTFOLIO, gate._GMM +_AV, _PORTFOLIO, _GMM, _AC = gate._AV, gate._PORTFOLIO, gate._GMM, gate._AC # (label, sampler argv, A, B, extra kwargs for _run_ile). A is None for a prior-only lane. LANES = [ @@ -62,6 +62,16 @@ # NOT A=8 B=2: that is the n_eff lottery recorded at the end of the gate's CALIBRATION # section. Mirrors the CPU "A=0.75 B=3, GMM" row instead. ("GPU A=0.75 B=3, GMM", _GMM, 0.75, 3.0, {"_needs_gpu": True}), + # portfolio and adaptive_cartesian could not run on a device at all until the host/device + # conversions in mcsamplerGPU.compute_hist and mcsampler.integrate. Each mirrors its host + # twin: the portfolio rows the plain-portfolio lanes above, adaptive_cartesian its own + # --n-max, taken from the gate rather than repeated here. + ("GPU prior-only, portfolio", _PORTFOLIO, None, 0.0, {"_needs_gpu": True}), + ("GPU A=0.75 B=3, portfolio", _PORTFOLIO, 0.75, 3.0, {"_needs_gpu": True}), + ("GPU prior-only, adaptive_cartesian", _AC, None, 0.0, + dict(_needs_gpu=True, n_max=gate._AC_N_MAX)), + ("GPU A=8 B=2, adaptive_cartesian", _AC, 8.0, 2.0, + dict(_needs_gpu=True, n_max=gate._AC_N_MAX)), ] @@ -138,7 +148,7 @@ def main(): for seed in seeds: # `idx` from enumerate, NOT LANES.index(...): index is a first-match lookup, so two # identical lane rows would silently share a tag and overwrite each other's ILE - # output directory. Correct for today's 17 distinct rows; wrong the moment one is + # output directory. Correct for today's 25 distinct rows; wrong the moment one is # duplicated, which is exactly the kind of edit this table invites. tag = "cal_%d_%d" % (idx, seed) lnL, sigma, neff = gate._run_ile(event, tag, sampler, a_coeff=a, b_coeff=b, diff --git a/MonteCarloMarginalizeCode/Code/test/test_e2e_analytic_pipeline.py b/MonteCarloMarginalizeCode/Code/test/test_e2e_analytic_pipeline.py index 97ba64cea..6de4d9d79 100644 --- a/MonteCarloMarginalizeCode/Code/test/test_e2e_analytic_pipeline.py +++ b/MonteCarloMarginalizeCode/Code/test/test_e2e_analytic_pipeline.py @@ -36,9 +36,8 @@ THE DEVICE. Section 5 runs the prior-only and factor answers again with a GPU VISIBLE, and asserts the child reached it. RIFT binds its array module at import from whether cupy imports, not from --gpu, so on a GPU node production is on the device path by default and the rest of -this file -- which pins CUDA_VISIBLE_DEVICES="" -- said nothing about it. Two samplers cannot -run there at all; they are recorded as known-failing lanes rather than skipped, so the hole is -visible. What section 5 does NOT cover is the GPU signal likelihood: --zero-likelihood +this file -- which pins CUDA_VISIBLE_DEVICES="" -- said nothing about it. All four samplers +run there. What section 5 does NOT cover is the GPU signal likelihood: --zero-likelihood replaces likelihood_function outright, so the NoLoop path never runs. WHAT EACH ARM COSTS. About 8-12 s of one core per ILE arm, plus ~15 s once for the @@ -82,7 +81,11 @@ _PORTFOLIO = ["--sampler-method", "portfolio", "--sampler-portfolio", "AV", "--sampler-portfolio", "AC"] _GMM = ["--sampler-method", "GMM"] -SAMPLER_ARGS = {"AV": _AV, "portfolio": _PORTFOLIO, "GMM": _GMM} +_AC = ["--sampler-method", "adaptive_cartesian"] +# adaptive_cartesian stops on the sample BUDGET rather than on --n-eff at the 20000 the other +# lanes use; see test_adaptive_cartesian for the eight-seed measurement behind this number. +_AC_N_MAX = 60000 +SAMPLER_ARGS = {"AV": _AV, "portfolio": _PORTFOLIO, "GMM": _GMM, "adaptive_cartesian": _AC} # --------------------------------------------------------------------------------------- @@ -481,9 +484,9 @@ def test_adaptive_cartesian(event): # rather than on --n-eff: over eight seeds it reported n_eff 91 to 220 against a target of # 250, with sigma up to 0.0430 against a 0.06 cap. At 60000 it stops on n_eff instead, at # ntotal 30000, with n_eff ~300 and sigma ~0.03. One-and-a-half times the arm, and the - # lane stops sitting one bad draw away from its own informativeness floor. - lnL, sigma, neff = _run_ile(event, tag, ["--sampler-method", "adaptive_cartesian"], - a_coeff=a, b_coeff=b, n_max=60000) + # lane stops sitting one bad draw away from its own informativeness floor. The device + # lane in section 5 inherits it through _GPU_LANE_KW. + lnL, sigma, neff = _run_ile(event, tag, _AC, a_coeff=a, b_coeff=b, n_max=_AC_N_MAX) _assert_lnZ(tag, lnL, sigma, neff, _exact(a, b)) @@ -502,20 +505,34 @@ def test_adaptive_cartesian(event): # and without _GPU_FLAGS returns bit-identical lnZ, because those flags only choose a likelihood # that this configuration does not evaluate. -_GPU_SAMPLERS = ["AV", "GMM"] +# portfolio and adaptive_cartesian were RECORDED HERE AS BROKEN when this section was written: +# both mixed host and device arrays and died in a cupy ufunc with "TypeError: Unsupported type +# ", invisible in production because the driver catches it, prints FAILED +# ANALYSIS and exits 0. Fixed on opposite sides, because the two sides are not alike -- the +# samples go TO the device in mcsamplerGPU.compute_hist, the integrand comes BACK to the host in +# mcsampler.integrate, whose accumulators are deliberately RiftFloat. Read those two comments +# before moving either conversion. +_GPU_SAMPLERS = ["AV", "GMM", "portfolio", "adaptive_cartesian"] -# Coefficients per device lane, and they are NOT the same pair for both samplers on purpose. +# Coefficients per device lane, and they are NOT the same pair for every sampler on purpose. # Each mirrors the CPU lane that sampler already runs, so a device row is comparable with a host -# row in the table below, and neither lands on a configuration this file records as unstable. +# row in the table below, and none lands on a configuration this file records as unstable. # # GMM is A=0.75 B=3, not A=8 B=2. The note at the end of the CALIBRATION section records that # GMM at A=8 WITH the inclination term is an n_eff LOTTERY: over eight seeds, two collapsed to # n_eff 73.5 and 14.5 with sigma 0.028 and 0.073, and 0.073 is over MAX_SIGMA. The evidence was # right on every seed; the error bar was not. That is why no CPU lane runs it, and a device lane # that ran it would be a flake with a ~1-in-4 seed. Eight clean GPU seeds do not refute a -# lottery -- the CPU sweep that FOUND it was also eight seeds. Both lanes keep a B term, so -# mis-routing inclination is still detectable; that is what the B term is for. -_GPU_LANE_COEFFS = {"AV": (8.0, 2.0), "GMM": (0.75, 3.0)} +# lottery -- the CPU sweep that FOUND it was also eight seeds. portfolio takes the same pair as +# its own A=0.75 B=3 host lane. Every lane keeps a B term, so mis-routing inclination is still +# detectable; that is what the B term is for. +_GPU_LANE_COEFFS = {"AV": (8.0, 2.0), "GMM": (0.75, 3.0), + "portfolio": (0.75, 3.0), "adaptive_cartesian": (8.0, 2.0)} + +# Per-lane _run_ile overrides, so a device lane runs its host twin's configuration and not a +# nearby one. Applied to the prior-only lane too: there the integrand is flat and the run stops +# on --n-eff long before any budget, so the larger cap costs nothing and keeps one definition. +_GPU_LANE_KW = {"adaptive_cartesian": dict(n_max=_AC_N_MAX)} @pytest.mark.parametrize("sampler", _GPU_SAMPLERS) @@ -523,7 +540,8 @@ def test_gpu_zero_likelihood_alone_gives_ln_Z_zero(event, gpu_slot, sampler): """The prior-only answer, on device. ln Z = 0 exactly, for a normalized extrinsic prior.""" tag = "gpu_zero_%s" % sampler lnL, sigma, neff = _run_ile(event, tag, SAMPLER_ARGS[sampler] + list(_GPU_FLAGS), - a_coeff=None, cuda=gpu_slot, expect_device=True) + a_coeff=None, cuda=gpu_slot, expect_device=True, + **_GPU_LANE_KW.get(sampler, {})) _assert_lnZ(tag, lnL, sigma, neff, 0.0) @@ -535,7 +553,8 @@ def test_gpu_analytic_factor_marginal_is_exact(event, gpu_slot, sampler): a, b = _GPU_LANE_COEFFS[sampler] tag = "gpu_factor_%s" % sampler lnL, sigma, neff = _run_ile(event, tag, SAMPLER_ARGS[sampler] + list(_GPU_FLAGS), - a_coeff=a, b_coeff=b, cuda=gpu_slot, expect_device=True) + a_coeff=a, b_coeff=b, cuda=gpu_slot, expect_device=True, + **_GPU_LANE_KW.get(sampler, {})) _assert_lnZ(tag, lnL, sigma, neff, _exact(a, b)) @@ -557,52 +576,6 @@ def test_the_gpu_flags_do_not_decide_the_backend(event, gpu_slot): "something else now depends on them." % (bare, flagged)) -# Device lanes that DO NOT WORK today, recorded with the shape of the failure so the hole is -# visible instead of hidden behind CUDA_VISIBLE_DEVICES="". Both are host/device mixes, both -# reproduce with NO supplementary factor, and neither is caused by anything in this file: -# -# portfolio (AV + AC) RIFT/likelihood/vectorized_general_tools.py, histogram(): -# `xpy.maximum(samples, blank_array)` with blank_array built by -# xpy.zeros on device and `samples` arriving as a numpy array from -# mcsamplerGPU.compute_hist. -# adaptive_cartesian RIFT/integrators/mcsampler.py, `fval*joint_p_prior/joint_p_s`: the -# integrand is on device, the two prior arrays are numpy. -# -# Both surface as `TypeError: Unsupported type ` from a cupy ufunc, and -# both are invisible in production because the driver catches the exception, prints FAILED -# ANALYSIS and EXITS 0. Measured on ldas-pcdev2 (A100, cc 8.0), cupy 12.0.0, 2026-09-17. -_GPU_KNOWN_FAILING = { - "portfolio": (_PORTFOLIO, {}, "vectorized_general_tools.py"), - "adaptive_cartesian": (["--sampler-method", "adaptive_cartesian"], dict(n_max=60000), - "integrators/mcsampler.py"), -} - - -@pytest.mark.parametrize("sampler", sorted(_GPU_KNOWN_FAILING)) -def test_the_gpu_samplers_that_do_not_work_still_fail_the_known_way(event, gpu_slot, sampler): - """A RECORDED DEFECT, not an excused one. - - Two samplers cannot run on a device at all. Skipping them would leave the gate reporting - green over a broken configuration, which is what pinning CUDA_VISIBLE_DEVICES="" did for - the whole file. So the failure is pinned by its SITE and its EXCEPTION, and this test goes - red in both directions: if someone fixes one, promote it into _GPU_SAMPLERS and delete the - entry; if the failure moves somewhere else, that is a different defect and wants reading.""" - args, kw, site = _GPU_KNOWN_FAILING[sampler] - tag = "gpu_known_fail_%s" % sampler - _d, _rc, out, have_row = _invoke_ile(event, tag, args + list(_GPU_FLAGS), - a_coeff=8.0, b_coeff=2.0, cuda=gpu_slot, **kw) - assert _DEVICE_MARKER in out, ( - "%s never reached a device, so this says nothing about the GPU path" % tag) - assert not have_row, ( - "%s SUCCEEDED on a device. If that is a fix, move %r into _GPU_SAMPLERS, delete its " - "_GPU_KNOWN_FAILING entry and the paragraph above it, and re-measure the calibration " - "table for the new lane." % (tag, sampler)) - assert "Unsupported type " in out and site in out, ( - "%s still fails on a device, but not in the recorded way: expected a cupy ufunc " - "TypeError from %s. A different failure is a different defect; read the log.\n%s" - % (tag, site, out[-2500:])) - - # --------------------------------------------------------------------------------------- # CALIBRATION # @@ -642,32 +615,50 @@ def test_the_gpu_samplers_that_do_not_work_still_fail_the_known_way(event, gpu_s # adaptive_cartesian, --n-max 60000 1.35 0.0305 250 # # THE DEVICE LANES, measured separately because they need a GPU: eight seeds (1000-1007) on -# ldas-pcdev2 slot 0 (A100, cc 8.0), cupy 12.0.0, at 5bb8da02b, with +# ldas-pcdev2 CUDA slot 0 (A100, cc 8.0 -- the slot MAP MOVES on that node, so the card was +# identified by running a kernel, not by nvidia-smi order), cupy 12.0.0, with # # python make_e2e_calibration.py --seeds 8 --lane "GPU " --gpu-slot 0 # -# The GMM row is A=0.75 B=3 and not A=8 B=2 deliberately; see _GPU_LANE_COEFFS. Each device row -# sits on top of its host twin, which is the point: the device path is not a different answer. +# The GMM and portfolio rows are A=0.75 B=3 and not A=8 B=2 deliberately; see _GPU_LANE_COEFFS. +# Each device row sits on top of its host twin, which is the point: the device path is not a +# different answer. # # lane max |z| max sigma min n_eff # GPU prior-only, AV 1.50 0.0134 1343 # GPU prior-only, GMM 0.93 0.0134 1352 # GPU A=8 B=2, AV 2.30 0.0325 257 # GPU A=0.75 B=3, GMM 1.49 0.0235 389 +# GPU prior-only, portfolio 2.22 0.0104 2335 +# GPU A=0.75 B=3, portfolio 2.32 0.0228 411 +# GPU prior-only, adaptive_cartesian 2.39 0.0091 3262 +# GPU A=8 B=2, adaptive_cartesian 1.35 0.0305 250 +# +# Host twins, from the table above: 2.03/0.0134/1349, 1.88/0.0134/1335, 1.06/0.0332/247, +# 1.34/0.0235/381, 1.68/0.0104/2316, 1.54/0.0227/392, -- (no host prior-only AC lane), +# 1.35/0.0305/250. Sigma is identical to all four decimals on five of the seven twinned rows; +# the other two differ by 0.0007 (A=8 B=2 AV) and 0.0001 (A=0.75 B=3 portfolio). The |z| +# values differ freely, because they are draws, not constants. +# +# The last row is the same three numbers as its host twin, not merely close, and that is +# expected rather than lucky: under --zero-likelihood the only device work is xpy.zeros, and +# mcsampler.integrate now copies the integrand back to the host before anything else touches +# it, so the two arms run identical host arithmetic off the same seed. If that row ever +# DRIFTS from its host twin, something started doing real arithmetic on the device. # -# Host twins for those four, from the table above: 2.03/0.0134/1349, 1.88/0.0134/1335, -# 1.06/0.0332/247, 1.34/0.0235/381. Same sigma to three digits on three of the four; the |z| -# values differ because they are draws, not constants. +# The first four rows were re-derived here, after the two host/device fixes, and came back +# identical to the values measured before them -- so those fixes do not move the lanes that +# already worked. # -# Z_TOLERANCE = 5 is 2.2x the worst |z| seen, which is now 2.30 on the GPU A=8 B=2 AV lane; -# the worst host lane is 2.15 (distance-marginalized), re-measured at eight -# seeds after the stand-in started passing xpy= to factors that accept it -# (max |z| 2.15, max sigma 0.0326, least n_eff 336, unchanged, seed 1000 -# bit-identical). Adding the device lanes moved the margin from 2.3x to -# 2.2x and nothing else. +# Z_TOLERANCE = 5 is 2.1x the worst |z| seen, which is now 2.39 on the GPU prior-only +# adaptive_cartesian lane; the worst host lane is 2.15 (distance- +# marginalized), re-measured at eight seeds after the stand-in started +# passing xpy= to factors that accept it (max |z| 2.15, max sigma 0.0326, +# least n_eff 336, unchanged, seed 1000 bit-identical). Adding four more +# device lanes moved the margin from 2.2x to 2.1x and nothing else. # MAX_SIGMA = 0.06 is 1.7x the worst sigma seen (0.0353, host). The worst device sigma is # 0.0325. 5 * MAX_SIGMA is a 0.30-nat band. -# MIN_NEFF = 30 is 5.4x below the worst n_eff seen (163, host; 257 on device). See its +# MIN_NEFF = 30 is 5.4x below the worst n_eff seen (163, host; 250 on device). See its # comment for why it is this loose. # # WHAT THE GATE HAS TO SEPARATE A CORRECT RUN FROM. Each row was run on this fixture, not From 6799e53d35ebc5f45c69f2da799b9dcf0d8b55d4 Mon Sep 17 00:00:00 2001 From: Richard O'Shaughnessy Date: Fri, 18 Sep 2026 14:35:21 -0700 Subject: [PATCH 6/6] act on the adversarial review: one wrong claim, one silent pass-through, one stated gap WRONG, AND MINE. The CALIBRATION note said the GPU A=8 B=2 adaptive_cartesian row equals its host twin because "the only device work is xpy.zeros", so any drift would mean device arithmetic had appeared. The reviewer disproved the mechanism by reading the code and then measuring it: for a sampler with return_lnL False the --zero-likelihood stand-in builds xpy.ones(n) * xpy.exp(supp), so cos and exp run on the DEVICE already. cupy 12.0.0 and numpy 1.26.4 disagree bitwise on about a fifth of 200000 float64 draws, and at seed 1000 that lane returns 6.646959838539267 on the host against 6.646957839585277 on the device. The rows agree at the precision the table PRINTS, which is a different and much weaker claim, and the tripwire built on the wrong one would have sent a reader hunting for device arithmetic that was there from the start. Restated with the measurement. to_host PASSED A DEVICE ARRAY THROUGH for a backend it could not convert. infer_array_module recognizes any module exposing asarray/where/clip; to_host only converted when it also exposed asnumpy, and torch is in that gap. So the one function written to remove a silent host/device mix would have reintroduced one, one backend later. Now: asnumpy, else the array's own .get(), else a named TypeError. Not reachable today -- nothing in the ILE path returns a torch tensor -- which is why it had to be closed by reading rather than by a test going red. All four branches checked, including that the numpy and RiftFloat paths still return the SAME OBJECT. A STATED GAP, not a fix. --sampler-method adaptive_cartesian_gpu is the driver's own GPU default and reaches the same compute_hist conversion by a different route, and section 4 has no lane for it. Measured by hand on ldas-pcdev2 slot 0 (A100), cupy 12.0.0: it works and lands on the closed form, z = +1.0 prior-only and -0.40 at A=8 B=2. Recorded in the file with the numbers and labelled a to-do rather than coverage, because a hand measurement is not a lane. Three further findings -- the calibration generator refusing --lane portfolio outright, _invoke_ile's docstring still citing the deleted known-failing lanes, and the host adaptive_cartesian row left un-centralized while its device twin used _AC/_AC_N_MAX -- were already resolved by the rift_O4d merge in the previous commit. The reviewer also ran the base-arm control this branch's claim rests on: at 02d66cc28 with the new test file, the four promoted lanes give "4 failed ... Unsupported type "; at the fix, 10 passed. And confirmed the CPU path is untouched by byte-comparing five CPU lanes' .dat and status JSON across the two trees. Co-Authored-By: Claude Opus 5 --- .../Code/RIFT/integrators/mcsampler.py | 19 +++++++++++-- .../Code/test/test_e2e_analytic_pipeline.py | 27 +++++++++++++++---- 2 files changed, 39 insertions(+), 7 deletions(-) diff --git a/MonteCarloMarginalizeCode/Code/RIFT/integrators/mcsampler.py b/MonteCarloMarginalizeCode/Code/RIFT/integrators/mcsampler.py index 726661cb0..46bbfab5a 100644 --- a/MonteCarloMarginalizeCode/Code/RIFT/integrators/mcsampler.py +++ b/MonteCarloMarginalizeCode/Code/RIFT/integrators/mcsampler.py @@ -1053,11 +1053,26 @@ def to_host(x): value, the same way infer_array_module does it, so this file keeps its numpy-only import list. A host input is returned UNCHANGED, not re-wrapped: the numpy path must stay bit for bit what it was. + + A non-numpy backend it cannot convert RAISES rather than passing the value through. + infer_array_module recognizes any module exposing asarray/where/clip, which is a wider + set than the ones exposing asnumpy (torch is in the gap), so "return it unchanged" would + hand a device array to the host-only arithmetic below the call site -- the silent + host/device mix this function exists to remove, reintroduced one backend later. """ mod = infer_array_module(x) - if mod is not numpy and hasattr(mod, "asnumpy"): + if mod is numpy: + return x + if hasattr(mod, "asnumpy"): return mod.asnumpy(x) - return x + _get = getattr(x, "get", None) # cupy-like array, module without a top-level asnumpy + if callable(_get): + return numpy.asarray(_get()) + raise TypeError( + "mcsampler cannot bring a %s.%s back to the host: the module exposes neither asnumpy " + "nor a .get() on the array, and this sampler's accumulators are host RiftFloat, which " + "no device backend can hold. Use a sampler that runs on that backend." + % (mod.__name__, type(x).__name__)) def clip_angle_limits(lo, hi, kind): diff --git a/MonteCarloMarginalizeCode/Code/test/test_e2e_analytic_pipeline.py b/MonteCarloMarginalizeCode/Code/test/test_e2e_analytic_pipeline.py index a98ec4b7d..7d3ae0b5e 100644 --- a/MonteCarloMarginalizeCode/Code/test/test_e2e_analytic_pipeline.py +++ b/MonteCarloMarginalizeCode/Code/test/test_e2e_analytic_pipeline.py @@ -552,6 +552,17 @@ def test_adaptive_cartesian(event): _GPU_LANE_COEFFS = {"AV": (8.0, 2.0), "GMM": (0.75, 3.0), "portfolio": (0.75, 3.0), "adaptive_cartesian": (8.0, 2.0)} +# WHAT SECTION 4 STILL DOES NOT COVER, stated because an unlisted sampler is an invisible gap +# rather than an absent one. `--sampler-method adaptive_cartesian_gpu` -- mcsamplerGPU +# standalone, and the ILE driver's own GPU default -- reaches the same +# mcsamplerGPU.compute_hist conversion these lanes exercise, by a different route (its +# integrate/integrate_log adaptation blocks rather than a portfolio member's external_rvs), +# and has no lane here. Measured by hand on ldas-pcdev2 slot 0 (A100), cupy 12.0.0, seed +# 1000: prior-only ln Z 0.00893 +- 0.00888 with n_eff 3364 (z = +1.0), and A=8 B=2 +# ln Z 6.64213 +- 0.02819 against 6.65332 (z = -0.40). It works; it is simply not gated. +# A hand measurement is not a lane -- see the note at the top of the CALIBRATION section +# about re-deriving rather than trusting -- so treat this as a to-do, not as coverage. + # Per-lane _run_ile overrides, so a device lane runs its host twin's configuration and not a # nearby one. Applied to the prior-only lane too: there the integrand is flat and the run stops # on --n-eff long before any budget, so the larger cap costs nothing and keeps one definition. @@ -670,11 +681,17 @@ def test_the_gpu_flags_do_not_decide_the_backend(event, gpu_slot): # the seven twinned rows; the other two differ by 0.0007 (A=8 B=2 AV) and 0.0001 (A=0.75 B=3 # portfolio). The |z| values differ freely, because they are draws, not constants. # -# THE LAST ROW IS ITS HOST TWIN EXACTLY, not merely close, and that is expected rather than -# lucky: under --zero-likelihood the only device work is xpy.zeros, and mcsampler.integrate now -# copies the integrand back to the host before anything else touches it, so the two arms run -# identical host arithmetic off the same seed. If that row ever DRIFTS from its host twin, -# something started doing real arithmetic on the device -- read it, do not just re-record it. +# THE LAST ROW PRINTS AS ITS HOST TWIN, and that is agreement at the table's precision, NOT a +# bitwise identity -- do not read it as one. An earlier version of this comment claimed the +# two arms run identical host arithmetic because "the only device work is xpy.zeros". That is +# wrong and was disproved by measurement: for a sampler with return_lnL False the +# --zero-likelihood stand-in builds xpy.ones(n) * xpy.exp(supp), so cos and exp run on the +# DEVICE. cupy 12.0.0 and numpy 1.26.4 disagree bitwise on about a fifth of 200000 float64 +# draws (max relative 2e-15), and at seed 1000 this lane returns ln Z 6.646959838539267 on the +# host against 6.646957839585277 on the device -- a 2e-6 difference, four orders below the +# 0.0305 error bar and well below what these three columns show. So a last-digit move in this +# row is a rounding boundary, not a finding; a move you can see in the SECOND digit is worth +# reading. # # The four AV/GMM rows above were RE-DERIVED in the sweep that added the last four, i.e. after # the two host/device fixes, and came back identical -- so those fixes do not move the lanes