Cooperate across 32 lanes per PSF and use two completion-protected staging slots with complete batch timing. Restore coarse OpenMP event production while serializing shared GPU submissions and direct fallback boundaries. Add bounded benchmarks, streaming and renderer regressions, and preserve validation evidence and ownership documentation.
62 lines
3.1 KiB
Python
62 lines
3.1 KiB
Python
#!/usr/bin/env python3
|
|
"""Sequential, bounded CPU/HIP renderer checks for shared-HDR fallback safety."""
|
|
import argparse
|
|
import importlib.util
|
|
import math
|
|
import os
|
|
from pathlib import Path
|
|
import re
|
|
import struct
|
|
import subprocess
|
|
import tempfile
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument("cpu", type=Path, help="Minkowski CPU executable")
|
|
parser.add_argument("hip", type=Path, help="Minkowski HIP executable")
|
|
args = parser.parse_args()
|
|
repo = Path(__file__).resolve().parents[1]
|
|
spec = importlib.util.spec_from_file_location("fits", repo / "scripts/fits_floatdiff.py")
|
|
fits = importlib.util.module_from_spec(spec)
|
|
spec.loader.exec_module(fits)
|
|
env = dict(os.environ, OMP_NUM_THREADS="4")
|
|
with tempfile.TemporaryDirectory(prefix="gr-hip-renderer-test-") as directory:
|
|
for min_y, expected in [("0", (2, 1, 0)), ("1e30", (0, 0, 3))]:
|
|
results = []
|
|
for label, binary in [("cpu", args.cpu), ("hip", args.hip)]:
|
|
output = Path(directory) / f"{label}-{min_y}.png"
|
|
command = [str(binary.resolve()), "--catalog",
|
|
str(repo / "tests/data/hip_psf_mixed_catalog.csv"),
|
|
"--width", "96", "--height", "64", "--fov-deg", "30",
|
|
"--look-ra-deg", "2", "--look-dec-deg", "1",
|
|
"--coarse-cell-pixels", "16", "--refine-max-level", "0",
|
|
"--exposure", "1000", "--psf-fwhm-pixels", "2.7",
|
|
"--psf-moffat-beta", "4.5", "--psf-relative-tail", "1e-8",
|
|
"--max-cache-psf-flux", "1", "--psf-min-y", min_y,
|
|
"--hdr-output", "--output", str(output)]
|
|
print("RUN", command, flush=True)
|
|
run = subprocess.run(command, env=env, text=True, capture_output=True, timeout=30)
|
|
print(run.stdout, run.stderr, flush=True)
|
|
run.check_returncode()
|
|
match = re.search(r"PSF splats: cached (\d+), cached wing-clipped (\d+), "
|
|
r"direct fallbacks (\d+), discarded below min-Y (\d+)",
|
|
run.stdout + run.stderr)
|
|
assert match, "missing splat accounting"
|
|
stats = tuple(map(int, match.groups()))
|
|
assert (stats[0], stats[2], stats[3]) == expected, stats
|
|
results.append((stats, fits.read_fits(output.with_name(output.stem + "_HDR.fits"))))
|
|
assert results[0][0] == results[1][0], "CPU/HIP statistics differ"
|
|
(shape, cpu), (gpu_shape, gpu) = results[0][1], results[1][1]
|
|
assert shape == gpu_shape
|
|
maximum = 0.0
|
|
for (a,), (b,) in zip(struct.iter_unpack(">f", cpu), struct.iter_unpack(">f", gpu)):
|
|
assert math.isfinite(a) and math.isfinite(b)
|
|
assert math.isclose(a, b, rel_tol=2**-23, abs_tol=0), (a, b)
|
|
maximum = max(maximum, abs(a - b))
|
|
print(f"min_y={min_y}: statistics match; HDR max_abs={maximum}; exact={cpu == gpu}", flush=True)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|