Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 6 additions & 6 deletions .travis/test-integrate.sh
Original file line number Diff line number Diff line change
Expand Up @@ -165,7 +165,7 @@ _ZLSTANDIN_OUT=$(python -m pytest -q -rs "$_ZLSTANDIN_TESTS" 2>&1 | tee >(cat >&
# On a runner that PROMISES a device, any skip is bad. The reason-matching below cannot tell
# "there is no GPU here" from "the GPU probe itself broke": break the probe and every device
# lane skips with a reason naming the GPU, which this would score as fine. Measured -- an
# unimportable module inside the e2e gate's _GPU_PROBE silently removed all 7 device lanes and
# unimportable module inside the e2e gate's _GPU_PROBE silently removed all 9 device lanes and
# scored 0 bad. RIFT_CI_REQUIRE_GPU=1 means the preflight at the top of this file already
# proved cupy works and a device computes, so there is nothing left for a skip to legitimately
# mean. Measured on ldas-pcdev2 slot 0: 0 skips.
Expand Down Expand Up @@ -195,20 +195,20 @@ fi
# lane here rather than a recorded defect. Needs no network and no real event; about
# 8-12 s per ILE arm plus ~15 s once for the distance-marginalization lookup table.
# Its section 4 runs the same answers with a GPU VISIBLE and asserts the child reached one.
# Those 7 lanes skip on a CPU-only runner and RUN under .gitlab-ci.yml's `gpu_integration` job
# Those 9 lanes skip on a CPU-only runner and RUN under .gitlab-ci.yml's `gpu_integration` job
# (RIFT_CI_REQUIRE_GPU=1, CUDA_VISIBLE_DEVICES=0, GPU container). Skipped tests are still
# collected, so the count below is the same on both; the SKIP GUARD after the run is what stops
# a CPU-only pass from reading as device coverage.
_E2E_TESTS=MonteCarloMarginalizeCode/Code/test/test_e2e_analytic_pipeline.py
_E2E_EXPECTED=24
_E2E_EXPECTED=26
_E2E_FOUND=$(python -m pytest -q --collect-only "$_E2E_TESTS" 2>/dev/null | grep -c '::' || true)
if [ "$_E2E_FOUND" -ne "$_E2E_EXPECTED" ]; then
echo "e2e analytic gate: collected $_E2E_FOUND tests, expected $_E2E_EXPECTED" >&2
exit 1
fi
# SKIP guard, as above. This gate skips 7 of 24 on a CPU-only runner, and it has a non-GPU
# SKIP guard, as above. This gate skips 9 of 26 on a CPU-only runner, and it has a non-GPU
# skip path that matters: build_event returns None when lal_path2cache is missing, which skips
# the WHOLE module -- 24 silent skips under a green exit 0. So a skip whose reason does not
# the WHOLE module -- 26 silent skips under a green exit 0. So a skip whose reason does not
# name cupy/GPU/CUDA fails the gate.
# Streamed, not `tee /dev/stderr` -- see the note on the stand-in gate above. This is the gate
# that made streaming worth having: 3-9 minutes, and silent when captured.
Expand All @@ -217,7 +217,7 @@ _E2E_OUT=$(python -m pytest -q -rs "$_E2E_TESTS" 2>&1 | tee >(cat >&2)) \
# On a runner that PROMISES a device, any skip is bad. The reason-matching below cannot tell
# "there is no GPU here" from "the GPU probe itself broke": break the probe and every device
# lane skips with a reason naming the GPU, which this would score as fine. Measured -- an
# unimportable module inside the e2e gate's _GPU_PROBE silently removed all 7 device lanes and
# unimportable module inside the e2e gate's _GPU_PROBE silently removed all 9 device lanes and
# scored 0 bad. RIFT_CI_REQUIRE_GPU=1 means the preflight at the top of this file already
# proved cupy works and a device computes, so there is nothing left for a skip to legitimately
# mean. Measured on ldas-pcdev2 slot 0: 0 skips.
Expand Down
40 changes: 40 additions & 0 deletions MonteCarloMarginalizeCode/Code/RIFT/integrators/mcsampler.py
Original file line number Diff line number Diff line change
Expand Up @@ -562,6 +562,16 @@ def integrate(self, func, *args, **kwargs):
else:
fval = func(**unpacked) # Chris' original plan: note this insures the function arguments are tied to the parameters, using a dictionary.

# THE INTEGRAND CONVERTS, NOT THE ACCUMULATORS. On a GPU node the ILE likelihood
# returns a cupy array (xpy_default binds at import from whether cupy imports, so
# a merely visible device is enough), while everything below here is host numpy:
# numpy.hstack, statutils update/finalize, math.isnan -- and joint_p_prior is
# deliberately RiftFloat, which cupy has no dtype for at all. So the host side is
# the one that cannot move. Without this, `fval*joint_p_prior/joint_p_s` raised
# "TypeError: Unsupported type <class 'numpy.ndarray'>" from a cupy ufunc and the
# driver swallowed it, printed FAILED ANALYSIS and exited 0.
fval = to_host(fval)

#
# Check if there is any practical contribution to the integral
#
Expand Down Expand Up @@ -1035,6 +1045,36 @@ def infer_array_module(x, xpy=None):
return numpy


def to_host(x):
"""Return `x` as a host (numpy) array, copying it off a device backend if it is on one.

numpy.asarray(cupy_array) RAISES -- cupy refuses implicit host conversion -- so the copy
has to go through the device module's own asnumpy. The module is looked up from the
value, the same way infer_array_module does it, so this file keeps its numpy-only import
list. A host input is returned UNCHANGED, not re-wrapped: the numpy path must stay bit
for bit what it was.

A non-numpy backend it cannot convert RAISES rather than passing the value through.
infer_array_module recognizes any module exposing asarray/where/clip, which is a wider
set than the ones exposing asnumpy (torch is in the gap), so "return it unchanged" would
hand a device array to the host-only arithmetic below the call site -- the silent
host/device mix this function exists to remove, reintroduced one backend later.
"""
mod = infer_array_module(x)
if mod is numpy:
return x
if hasattr(mod, "asnumpy"):
return mod.asnumpy(x)
_get = getattr(x, "get", None) # cupy-like array, module without a top-level asnumpy
if callable(_get):
return numpy.asarray(_get())
raise TypeError(
"mcsampler cannot bring a %s.%s back to the host: the module exposes neither asnumpy "
"nor a .get() on the array, and this sampler's accumulators are host RiftFloat, which "
"no device backend can hold. Use a sampler that runs on that backend."
% (mod.__name__, type(x).__name__))


def clip_angle_limits(lo, hi, kind):
"""Clip an angular range [lo,hi] (radians) to the physical domain of `kind`
('declination' -> [-pi/2,pi/2], 'inclination' -> [0,pi]).
Expand Down
18 changes: 18 additions & 0 deletions MonteCarloMarginalizeCode/Code/RIFT/integrators/mcsamplerGPU.py
Original file line number Diff line number Diff line change
Expand Up @@ -326,6 +326,24 @@ def setup_hist_single_param(self, x_min, x_max, n_bins, param):


def compute_hist(self, x_samples, param,weights=None,floor_level=0):
# PUT THE INPUTS ON THIS SAMPLER'S BACKEND FIRST. As a portfolio member the samples
# arrive from the aggregator's host _rvs (mcsamplerPortfolio pins self.xpy = numpy)
# while self.xpy here is cupy, and the first cupy ufunc downstream -- the
# xpy.maximum(samples, xpy.zeros(...)) clip in vectorized_general_tools.histogram --
# raised "TypeError: Unsupported type <class 'numpy.ndarray'>", killing every
# portfolio run with an AC member on a GPU node behind FAILED ANALYSIS and exit 0.
#
# DEVICE is the side to converge on here. The result has to end up there anyway
# (setup_hist_single_param preallocates histogram_edges/histogram_cdf with self.xpy,
# and cdf_inverse_from_hist is read by device draws), and nothing here carries
# precision worth keeping on the host: the coordinates are float64 and this histogram
# is a *proposal* density, not an estimator (see _bincount_weighted). Contrast
# mcsampler.py, whose accumulators are deliberately RiftFloat and so convert the
# other way. asarray is a no-op when the input is already on this backend, so the
# standalone GPU and pure-numpy paths are untouched.
x_samples = self.xpy.asarray(x_samples)
if weights is not None:
weights = self.xpy.asarray(weights)
# Rescale the samples to [0, 1]
y_samples = (
(x_samples - self.x_min[param]) / self.x_max_minus_min[param]
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -29,7 +29,7 @@

import test_e2e_analytic_pipeline as gate # noqa: E402

_AV, _PORTFOLIO, _GMM = gate._AV, gate._PORTFOLIO, gate._GMM
_AV, _PORTFOLIO, _GMM, _AC = gate._AV, gate._PORTFOLIO, gate._GMM, gate._AC

# (label, sampler argv, A, B, extra kwargs for _run_ile). A is None for a prior-only lane.
LANES = [
Expand All @@ -52,16 +52,26 @@
dict(extra=("--time-marginalization",))),
("distance-marginalized", _AV, 8.0, 2.0, {"_needs_dmarg": True}),
("adaptive_cartesian, --n-max 60000",
["--sampler-method", "adaptive_cartesian"], 8.0, 2.0, dict(n_max=60000)),
_AC, 8.0, 2.0, dict(n_max=gate._AC_N_MAX)),
# DEVICE lanes. They need --gpu-slot, and are refused rather than skipped without it, for
# the same reason as a --lane typo: a table that quietly stops covering four of its rows
# the same reason as a --lane typo: a table that quietly stops covering eight of its rows
# under a summary line that reads clean is worse than no table.
("GPU prior-only, AV", _AV, None, 0.0, {"_needs_gpu": True}),
("GPU prior-only, GMM", _GMM, None, 0.0, {"_needs_gpu": True}),
("GPU A=8 B=2, AV", _AV, 8.0, 2.0, {"_needs_gpu": True}),
# NOT A=8 B=2: that is the n_eff lottery recorded at the end of the gate's CALIBRATION
# section. Mirrors the CPU "A=0.75 B=3, GMM" row instead.
("GPU A=0.75 B=3, GMM", _GMM, 0.75, 3.0, {"_needs_gpu": True}),
# portfolio and adaptive_cartesian could not run on a device at all until the host/device
# conversions in mcsamplerGPU.compute_hist and mcsampler.integrate. Each mirrors its host
# twin: portfolio the plain-portfolio lanes above, adaptive_cartesian its own --n-max, taken
# from the gate rather than repeated here.
("GPU prior-only, portfolio", _PORTFOLIO, None, 0.0, {"_needs_gpu": True}),
("GPU A=0.75 B=3, portfolio", _PORTFOLIO, 0.75, 3.0, {"_needs_gpu": True}),
("GPU prior-only, adaptive_cartesian", _AC, None, 0.0,
dict(_needs_gpu=True, n_max=gate._AC_N_MAX)),
("GPU A=8 B=2, adaptive_cartesian", _AC, 8.0, 2.0,
dict(_needs_gpu=True, n_max=gate._AC_N_MAX)),
]


Expand Down
Loading
Loading