diff --git a/.travis/test-integrate.sh b/.travis/test-integrate.sh index 4a67ec1bd..1dda5740b 100755 --- a/.travis/test-integrate.sh +++ b/.travis/test-integrate.sh @@ -171,14 +171,18 @@ _ZLSTANDIN_OUT=$(python -m pytest -q -rs "$_ZLSTANDIN_TESTS" 2>&1 | tee >(cat >& # and a device computes, so there is nothing left for a skip to legitimately mean. Measured on # ldas-pcdev2 CUDA slot 0 (A100): 0 skips. if [[ "${RIFT_CI_REQUIRE_GPU:-0}" == "1" ]]; then - _ZLSTANDIN_BAD=$(echo "$_ZLSTANDIN_OUT" | grep -cE '^SKIPPED' || true) + _ZLSTANDIN_BAD_LINES=$(echo "$_ZLSTANDIN_OUT" | grep -E '^SKIPPED' || true) else - _ZLSTANDIN_BAD=$(echo "$_ZLSTANDIN_OUT" | grep -E '^SKIPPED' | grep -vciE 'cupy|gpu|cuda' || true) + _ZLSTANDIN_BAD_LINES=$(echo "$_ZLSTANDIN_OUT" | grep -E '^SKIPPED' | grep -viE 'cupy|gpu|cuda' || true) fi -_ZLSTANDIN_BAD=${_ZLSTANDIN_BAD:-0} +# ONE list decides the count AND the listing. They used to be two greps, so on the +# no-flag path the count was reason-filtered and the listing was not: the gate said +# "1 unacceptable" and printed two lines, the first of them an ACCEPTABLE skip. +if [ -z "$_ZLSTANDIN_BAD_LINES" ]; then _ZLSTANDIN_BAD=0; else _ZLSTANDIN_BAD=$(printf '%s +' "$_ZLSTANDIN_BAD_LINES" | wc -l); fi if [ "$_ZLSTANDIN_BAD" -ne 0 ]; then - echo "zero-likelihood stand-in gate: $_ZLSTANDIN_BAD unacceptable SKIPPED line(s) -- pytest -rs groups equal reasons, so this is not a test count (RIFT_CI_REQUIRE_GPU=${RIFT_CI_REQUIRE_GPU:-0}; with it set, ANY skip is unacceptable):" >&2 - echo "$_ZLSTANDIN_OUT" | grep -E '^SKIPPED' >&2 + echo "zero-likelihood stand-in gate: $_ZLSTANDIN_BAD unacceptable SKIPPED line(s) -- pytest -rs groups by (file:line, reason), so ONE line can cover N tests and this is not a test count (RIFT_CI_REQUIRE_GPU=${RIFT_CI_REQUIRE_GPU:-0}; with it set, ANY skip is unacceptable):" >&2 + printf '%s\n' "$_ZLSTANDIN_BAD_LINES" >&2 exit 1 fi @@ -224,14 +228,18 @@ _E2E_OUT=$(python -m pytest -q -rs "$_E2E_TESTS" 2>&1 | tee >(cat >&2)) \ # and a device computes, so there is nothing left for a skip to legitimately mean. Measured on # ldas-pcdev2 CUDA slot 0 (A100): 0 skips. if [[ "${RIFT_CI_REQUIRE_GPU:-0}" == "1" ]]; then - _E2E_BAD=$(echo "$_E2E_OUT" | grep -cE '^SKIPPED' || true) + _E2E_BAD_LINES=$(echo "$_E2E_OUT" | grep -E '^SKIPPED' || true) else - _E2E_BAD=$(echo "$_E2E_OUT" | grep -E '^SKIPPED' | grep -vciE 'cupy|gpu|cuda' || true) + _E2E_BAD_LINES=$(echo "$_E2E_OUT" | grep -E '^SKIPPED' | grep -viE 'cupy|gpu|cuda' || true) fi -_E2E_BAD=${_E2E_BAD:-0} +# ONE list decides the count AND the listing. They used to be two greps, so on the +# no-flag path the count was reason-filtered and the listing was not: the gate said +# "1 unacceptable" and printed two lines, the first of them an ACCEPTABLE skip. +if [ -z "$_E2E_BAD_LINES" ]; then _E2E_BAD=0; else _E2E_BAD=$(printf '%s +' "$_E2E_BAD_LINES" | wc -l); fi if [ "$_E2E_BAD" -ne 0 ]; then - echo "e2e analytic gate: $_E2E_BAD unacceptable SKIPPED line(s) -- pytest -rs groups equal reasons, so this is not a test count (RIFT_CI_REQUIRE_GPU=${RIFT_CI_REQUIRE_GPU:-0}; with it set, ANY skip is unacceptable):" >&2 - echo "$_E2E_OUT" | grep -E '^SKIPPED' >&2 + echo "e2e analytic gate: $_E2E_BAD unacceptable SKIPPED line(s) -- pytest -rs groups by (file:line, reason), so ONE line can cover N tests and this is not a test count (RIFT_CI_REQUIRE_GPU=${RIFT_CI_REQUIRE_GPU:-0}; with it set, ANY skip is unacceptable):" >&2 + printf '%s\n' "$_E2E_BAD_LINES" >&2 exit 1 fi @@ -282,11 +290,25 @@ fi # whose REASON names cupy/GPU, and fail on any other skip whatever the total. _TMARG_OUT=$(python -m pytest -q -rs "${_TMARG_TESTS[@]}" 2>&1) || { echo "$_TMARG_OUT"; exit 1; } echo "$_TMARG_OUT" | tail -20 -_TMARG_BAD=$(echo "$_TMARG_OUT" | grep -E '^SKIPPED' | grep -vciE 'cupy|gpu|cuda' || true) -_TMARG_BAD=${_TMARG_BAD:-0} +# On a runner that PROMISES a device, any skip is bad -- the same branch the stand-in and e2e +# gates above carry, and it was missing from this block and the next. Nothing reachable +# exploited that on 2026-09-19: every GPU-reason skip these files can emit goes through a guard +# that already fails under RIFT_CI_REQUIRE_GPU=1. It is here so that stays true. What the +# reason-match alone accepts, measured by feeding it one line +# `SKIPPED [1] x.py:12: no cupy on this host` with the flag set: BAD=0, gate ACCEPTED. +if [[ "${RIFT_CI_REQUIRE_GPU:-0}" == "1" ]]; then + _TMARG_BAD_LINES=$(echo "$_TMARG_OUT" | grep -E '^SKIPPED' || true) +else + _TMARG_BAD_LINES=$(echo "$_TMARG_OUT" | grep -E '^SKIPPED' | grep -viE 'cupy|gpu|cuda' || true) +fi +# ONE list decides the count AND the listing. They used to be two greps, so on the +# no-flag path the count was reason-filtered and the listing was not: the gate said +# "1 unacceptable" and printed two lines, the first of them an ACCEPTABLE skip. +if [ -z "$_TMARG_BAD_LINES" ]; then _TMARG_BAD=0; else _TMARG_BAD=$(printf '%s +' "$_TMARG_BAD_LINES" | wc -l); fi if [ "$_TMARG_BAD" -ne 0 ]; then - echo "time-marginalization gate: $_TMARG_BAD test(s) skipped for a reason other than an absent GPU:" >&2 - echo "$_TMARG_OUT" | grep -E '^SKIPPED' | grep -viE 'cupy|gpu|cuda' >&2 + echo "time-marginalization gate: $_TMARG_BAD unacceptable SKIPPED line(s) -- pytest -rs groups by (file:line, reason), so ONE line can cover N tests and this is not a test count (RIFT_CI_REQUIRE_GPU=${RIFT_CI_REQUIRE_GPU:-0}; with it set, ANY skip is unacceptable):" >&2 + printf '%s\n' "$_TMARG_BAD_LINES" >&2 exit 1 fi @@ -314,11 +336,20 @@ fi # whose REASON names cupy/GPU, and fail on any other skip whatever the total. _TMARG_PL_OUT=$(python -m pytest -q -rs "$_TMARG_PL_TESTS" 2>&1) || { echo "$_TMARG_PL_OUT"; exit 1; } echo "$_TMARG_PL_OUT" | tail -20 -_TMARG_PL_BAD=$(echo "$_TMARG_PL_OUT" | grep -E '^SKIPPED' | grep -vciE 'cupy|gpu|cuda' || true) -_TMARG_PL_BAD=${_TMARG_PL_BAD:-0} +# As in the time-marginalization block above: under RIFT_CI_REQUIRE_GPU=1 any skip is bad. +if [[ "${RIFT_CI_REQUIRE_GPU:-0}" == "1" ]]; then + _TMARG_PL_BAD_LINES=$(echo "$_TMARG_PL_OUT" | grep -E '^SKIPPED' || true) +else + _TMARG_PL_BAD_LINES=$(echo "$_TMARG_PL_OUT" | grep -E '^SKIPPED' | grep -viE 'cupy|gpu|cuda' || true) +fi +# ONE list decides the count AND the listing. They used to be two greps, so on the +# no-flag path the count was reason-filtered and the listing was not: the gate said +# "1 unacceptable" and printed two lines, the first of them an ACCEPTABLE skip. +if [ -z "$_TMARG_PL_BAD_LINES" ]; then _TMARG_PL_BAD=0; else _TMARG_PL_BAD=$(printf '%s +' "$_TMARG_PL_BAD_LINES" | wc -l); fi if [ "$_TMARG_PL_BAD" -ne 0 ]; then - echo "peak-local gate: $_TMARG_PL_BAD test(s) skipped for a reason other than an absent GPU:" >&2 - echo "$_TMARG_PL_OUT" | grep -E '^SKIPPED' | grep -viE 'cupy|gpu|cuda' >&2 + echo "peak-local gate: $_TMARG_PL_BAD unacceptable SKIPPED line(s) -- pytest -rs groups by (file:line, reason), so ONE line can cover N tests and this is not a test count (RIFT_CI_REQUIRE_GPU=${RIFT_CI_REQUIRE_GPU:-0}; with it set, ANY skip is unacceptable):" >&2 + printf '%s\n' "$_TMARG_PL_BAD_LINES" >&2 exit 1 fi diff --git a/MonteCarloMarginalizeCode/Code/test/gpu_slot_probe.py b/MonteCarloMarginalizeCode/Code/test/gpu_slot_probe.py index 8df26a729..847e92fea 100644 --- a/MonteCarloMarginalizeCode/Code/test/gpu_slot_probe.py +++ b/MonteCarloMarginalizeCode/Code/test/gpu_slot_probe.py @@ -138,12 +138,19 @@ def no_gpu(reason): RIFT_CI_REQUIRE_GPU=1 is the GPU runner saying it HAS a device; there, a skipped device lane is a green report for a lane that never ran. - THIS IS THE ONLY PLACE THAT RULE IS ENFORCED for these two files, so do not delete it as a - duplicate of the shell. .travis/test-integrate.sh does carry an "under RIFT_CI_REQUIRE_GPU - any skip is fatal" branch, but only for the zero-likelihood stand-in and e2e gates; its - peak-local block, and the whole of .travis/test-q-window-stencil.sh, score skips by REASON - alone and would accept these. Checked 2026-09-18: `grep -n RIFT_CI_REQUIRE_GPU - .travis/test-q-window-stencil.sh` prints nothing. + DO NOT DELETE THIS AS A DUPLICATE OF THE SHELL. Re-checked 2026-09-19, because the first + version of this paragraph went stale within a day and said three wrong things: + + * all four skip-scoring blocks in .travis/test-integrate.sh now carry an "under + RIFT_CI_REQUIRE_GPU any skip is fatal" branch. Two did when this was written. + * .travis/test-q-window-stencil.sh has no RIFT_CI_REQUIRE_GPU branch and no reason + matching either. It scores by junit COUNTS -- EXPECTED_PASSED=75, MAX_SKIPS=3 -- so a + new skip there drops `passed` below the floor and fails on the count, not the reason. + * this helper has three callers, not two. + + What the shell still cannot do is survive `pytest ` run by hand on the GPU node, + which is what someone does to reproduce a CI failure, and that run would report green with + the device lane skipped. That is what this function is for. """ if os.environ.get("RIFT_CI_REQUIRE_GPU", "0") == "1": pytest.fail("RIFT_CI_REQUIRE_GPU=1 promised a usable device and there is none: %s. " @@ -176,8 +183,9 @@ def cupy_or_skip(require_rift_backend=False): """The imported cupy module, with a slot that can build a kernel already selected. Skips (or fails under RIFT_CI_REQUIRE_GPU=1) instead of raising, on both of the failures - above. Every skip reason names cupy/GPU/CUDA, which is what the skip-reason guards in - .travis/test-integrate.sh and .travis/test-q-window-stencil.sh accept. + above. Every skip reason names cupy/GPU/CUDA, which is what the reason-matching guards in + .travis/test-integrate.sh accept when RIFT_CI_REQUIRE_GPU is unset. + (.travis/test-q-window-stencil.sh does not match reasons; see no_gpu.) ``require_rift_backend`` additionally demands that RIFT's own import-time probes took the device. OPT-IN, because the two answers differ and a blanket check costs a lane that diff --git a/MonteCarloMarginalizeCode/Code/test/test_time_marginalization_quadrature.py b/MonteCarloMarginalizeCode/Code/test/test_time_marginalization_quadrature.py index bbb964ead..10908dca7 100644 --- a/MonteCarloMarginalizeCode/Code/test/test_time_marginalization_quadrature.py +++ b/MonteCarloMarginalizeCode/Code/test/test_time_marginalization_quadrature.py @@ -40,6 +40,11 @@ from RIFT.likelihood import time_marginalization_quadrature as tmq from RIFT.likelihood import factored_likelihood as fl +_HERE = os.path.dirname(os.path.abspath(__file__)) +if _HERE not in sys.path: + sys.path.insert(0, _HERE) +import gpu_slot_probe # noqa: E402 (sibling helper) + simpson = getattr(integrate, 'simpson', None) or integrate.simps SRATE = 4096.0 @@ -967,22 +972,33 @@ def test_driver_announces_the_quadrature_it_will_actually_use(): # --------------------------------------------------------------- GPU parity def _cupy_or_skip(): - """cupy, or skip -- unless the GPU gate demands a device, in which case FAIL. - - `RIFT_CI_REQUIRE_GPU=1` is how `.travis/test-integrate.sh` says "this job runs - on hardware". A skip under that flag would be a GPU gate reporting green + """cupy on a slot that can build a kernel, or skip -- a FAILURE under + RIFT_CI_REQUIRE_GPU=1, which is how `.travis/test-integrate.sh` says "this job + runs on hardware". A skip under that flag would be a GPU gate reporting green without having touched a GPU, which is the failure mode this whole file is written against. + + This used to import cupy and accept any `getDeviceCount() >= 1`. COUNTING + DEVICES IS NOT RUNNING ONE: measured on ldas-pcdev2 2026-09-19 with + CUDA_VISIBLE_DEVICES=1, a Blackwell cc 12.0 that cupy 12.0.0 cannot compile + for, the count was 1, the guard returned cupy, and + test_bandlimited_runs_on_the_gpu_backend_and_matches_numpy died on + `CompileException: nvrtc: error: invalid value for --gpu-architecture`. At + CUDA_VISIBLE_DEVICES="1,0" it failed the same way while a usable A100 sat in + the list, because it never chose a slot either. gpu_slot_probe runs a kernel + per visible slot in a subprocess and selects one that works. + + NO ``require_rift_backend=True`` here, and the reason is a measurement rather + than the shape of the traceback. RIFT DOES fall back to numpy at both mixed + orderings -- ldas-pcdev2, CUDA_VISIBLE_DEVICES="1,0" and "1,2": + ``factored_likelihood.xpy_default is numpy`` and + ``SphericalHarmonics_gpu.cupy_here is False`` in each. This test passes anyway, + because it takes ``xpy`` as an argument and reaches only + ``tmq.time_marginalize_bandlimited`` and ``optimized_gpu_tools.simps``, never + ``SphericalHarmonics_gpu._coeffs``. A caller that DOES reach the likelihood needs + the flag; see gpu_slot_probe.cupy_or_skip. """ - try: - import cupy - if cupy.cuda.runtime.getDeviceCount() < 1: - raise RuntimeError("cupy imported but reports zero CUDA devices") - return cupy - except Exception as exc: - if os.environ.get('RIFT_CI_REQUIRE_GPU') == '1': - pytest.fail("RIFT_CI_REQUIRE_GPU=1 but cupy/GPU unavailable: %s" % exc) - pytest.skip("cupy/GPU unavailable: %s" % exc) + return gpu_slot_probe.cupy_or_skip() def test_bandlimited_runs_on_the_gpu_backend_and_matches_numpy():