HIP: accelerate PSF accumulation and restore parallel producers

Cooperate across 32 lanes per PSF and use two completion-protected staging slots with complete batch timing. Restore coarse OpenMP event production while serializing shared GPU submissions and direct fallback boundaries.

Add bounded benchmarks, streaming and renderer regressions, and preserve validation evidence and ownership documentation.
This commit is contained in:
wyj committed 2026-09-06 21:38:53 -04:00
1 parent 265d7b95d5
commit e1ec480669
45 files changed
+10251 -95

No files matched your search

+96
View File
@@ -0,0 +1,96 @@
#include "hip_psf.h"
#include <hip/hip_runtime.h>
#include <omp.h>
#include <cmath>
#include <cstdio>
#include <cstdlib>
#include <cstring>
#include <vector>
/* Bounded production-cache replay. One process runs CPU reference then HIP;
* no catalog loading or geodesic work. Arguments: events, spread pixels. */
int main(int argc, char **argv) {
if (argc == 2 && !std::strcmp(argv[1], "--help")) {
std::puts("Usage: benchmark_hip_psf [EVENTS [SPREAD_PIXELS]]\n"
" EVENTS Cached events, 1..65536 (default: 4096)\n"
" SPREAD_PIXELS Star-center spread, 1..3840 (default: 16)\n"
"Runs a serial CPU reference, then HIP, at 3840x2160; PSF 2.7/4.5, tail 1e-8.\n"
"Use an external timeout and run only one benchmark process at a time.");
return 0;
}
auto number = [](const char *s, long upper) {
char *end = nullptr;
const long n = std::strtol(s, &end, 10);
return s != end && !*end && n >= 1 && n <= upper ? (int)n : 0;
};
const int count = argc > 1 ? number(argv[1], 65536) : 4096;
const int spread = argc > 2 ? number(argv[2], 3840) : 16;
if (argc > 3 || !count || !spread) {
std::fputs("Invalid bounded benchmark arguments; see --help.\n", stderr);
return 2;
}
std::setvbuf(stdout, nullptr, _IOLBF, 0);
const int width = 3840, height = 2160;
const size_t values = (size_t)width * height * 3;
PsfKernelCache cache = {};
const PointSpreadFunction psf = {2.7, 4.5};
if (psf_kernel_cache_init(&cache, &psf, 1e-8)) return 1;
psf_kernel_cache_report_ready(&cache, stdout);
std::vector<PsfCachedEvent> events(count);
unsigned state = 12345;
auto uniform = [&]() { state = state * 1664525u + 1013904223u;
return (state >> 8) * (1.0 / 16777216.0); };
for (auto &event : events) {
const double x = (width - spread) * 0.5 + uniform() * spread;
const double sy = spread < height ? spread : height;
const double y = (height - sy) * 0.5 + uniform() * sy;
const LinearRgb color = {0.1 + uniform(), 0.1 + uniform(), 0.1 + uniform()};
const int result = psf_prepare_cached_event(&event, x, y, color,
0.1 + uniform(), &psf, &cache, 1e8, 1e-8, 0);
if (result != 0) return 1;
}
std::vector<double> cpu(values, 0), gpu(values, 0);
double start = omp_get_wtime();
for (const auto &event : events)
splat_prepared_cached_event(cpu.data(), width, height, &event, &cache);
const double cpu_seconds = omp_get_wtime() - start;
char message[256] = {};
hipDeviceProp_t prop;
if (hipGetDeviceProperties(&prop, 0) != hipSuccess) return 1;
std::printf("device=%s arch=%s events=%d spread=%d frame=%dx%d event_bytes=%zu CPU_reference=%.6f s\n",
prop.name, prop.gcnArchName, count, spread, width, height, sizeof(PsfCachedEvent), cpu_seconds);
HipPsfSink *sink = nullptr;
start = omp_get_wtime();
if (hip_psf_sink_create(&sink, width, height, &cache, 16384, message, sizeof message)) {
std::fprintf(stderr, "create: %s\n", message); return 1;
}
const double create_seconds = omp_get_wtime() - start;
start = omp_get_wtime();
for (size_t i = 0; i < events.size(); i += 16384) {
size_t n = events.size() - i;
if (n > 16384) n = 16384;
if (hip_psf_sink_submit(sink, events.data() + i, n, message, sizeof message)) {
std::fprintf(stderr, "submit: %s\n", message); return 1;
}
}
if (hip_psf_sink_finish(sink, gpu.data(), message, sizeof message)) {
std::fprintf(stderr, "finish: %s\n", message); return 1;
}
const double wall = omp_get_wtime() - start;
HipPsfTiming timing = {};
hip_psf_sink_get_timing(sink, &timing);
double max_abs = 0, max_rel = 0, sum = 0, diff = 0;
for (size_t i = 0; i < values; ++i) {
if (!std::isfinite(gpu[i])) return 1;
const double d = std::fabs(gpu[i] - cpu[i]);
max_abs = std::fmax(max_abs, d);
if (cpu[i] != 0) max_rel = std::fmax(max_rel, d / std::fabs(cpu[i]));
sum += cpu[i]; diff += gpu[i] - cpu[i];
}
std::printf("HIP create=%.6f replay_wall=%.6f upload=%.6f kernel=%.6f download=%.6f batches=%zu timed=%zu max_abs=%.17g max_rel=%.17g flux_rel=%.17g\n",
create_seconds, wall, timing.upload_seconds, timing.kernel_seconds, timing.download_seconds,
timing.batch_count, timing.timed_batch_count, max_abs, max_rel, diff / sum);
hip_psf_sink_destroy(sink);
psf_kernel_cache_destroy(&cache);
return max_abs < 1e-10 && max_rel < 1e-10 ? 0 : 1;
}
+32 -1
View File
@@ -57,7 +57,38 @@ int main() {
return 1;
}
double max_abs = 0.0, max_rel = 0.0;
/* Exercise both reusable slots, more than the old 64-batch timing limit,
* immediate caller-buffer reuse, edge clipping and an ordered CPU boundary. */
for (int batch = 0; batch < 70; ++batch) {
for (int j = 0; j < 3; ++j) {
events[j] = PsfCachedEvent{};
const int result = psf_prepare_cached_event(&events[j],
-4.75 + (batch * 7 + j * 13) % 75,
-3.125 + (batch * 11 + j * 5) % 58, colors[j],
batch % 7 == 0 ? 1000.0 : 0.01, &psf, &cache, 1e8, 1e-8, 0.0);
if (result != 0 && result != 2) return 1;
splat_prepared_cached_event(cpu_hdr, width, height, &events[j], &cache);
}
if (hip_psf_sink_submit(sink, events, 3, message, sizeof message)) {
std::fprintf(stderr, "streaming submit failed: %s\n", message); return 1;
}
for (auto &event : events) event = PsfCachedEvent{};
if (batch == 35) {
if (hip_psf_sink_finish(sink, gpu_hdr, message, sizeof message)) return 1;
for (double *hdr : {cpu_hdr, gpu_hdr})
splat_moffat_direct(hdr, width, height, 12.25, 20.75, colors[0],
0.2, &psf, 1e-8, 0.0);
if (hip_psf_sink_load_hdr(sink, gpu_hdr, message, sizeof message)) return 1;
}
}
if (hip_psf_sink_finish(sink, gpu_hdr, message, sizeof message) ||
hip_psf_sink_get_timing(sink, &timing) || timing.event_count != 213 ||
timing.batch_count != 71 || timing.timed_batch_count != 71) {
std::fprintf(stderr, "streaming completion/accounting failed: %s\n", message);
return 1;
}
for (size_t i = 0; i < values; ++i) {
if (!std::isfinite(cpu_hdr[i]) || !std::isfinite(gpu_hdr[i])) return 1;
const double absolute = std::fabs(cpu_hdr[i] - gpu_hdr[i]);
max_abs = std::fmax(max_abs, absolute);
if (std::fabs(cpu_hdr[i]) > 1e-30)
@@ -66,5 +97,5 @@ int main() {
std::printf("HIP PSF cache comparison: max_abs=%.17g max_rel=%.17g\n", max_abs, max_rel);
hip_psf_sink_destroy(sink);
std::free(cpu_hdr); std::free(gpu_hdr); psf_kernel_cache_destroy(&cache);
return max_abs <= 1e-12 && max_rel <= 1e-12 ? 0 : 1;
return max_abs <= 1e-10 && max_rel <= 1e-12 ? 0 : 1;
}
+61
View File
@@ -0,0 +1,61 @@
#!/usr/bin/env python3
"""Sequential, bounded CPU/HIP renderer checks for shared-HDR fallback safety."""
import argparse
import importlib.util
import math
import os
from pathlib import Path
import re
import struct
import subprocess
import tempfile
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("cpu", type=Path, help="Minkowski CPU executable")
parser.add_argument("hip", type=Path, help="Minkowski HIP executable")
args = parser.parse_args()
repo = Path(__file__).resolve().parents[1]
spec = importlib.util.spec_from_file_location("fits", repo / "scripts/fits_floatdiff.py")
fits = importlib.util.module_from_spec(spec)
spec.loader.exec_module(fits)
env = dict(os.environ, OMP_NUM_THREADS="4")
with tempfile.TemporaryDirectory(prefix="gr-hip-renderer-test-") as directory:
for min_y, expected in [("0", (2, 1, 0)), ("1e30", (0, 0, 3))]:
results = []
for label, binary in [("cpu", args.cpu), ("hip", args.hip)]:
output = Path(directory) / f"{label}-{min_y}.png"
command = [str(binary.resolve()), "--catalog",
str(repo / "tests/data/hip_psf_mixed_catalog.csv"),
"--width", "96", "--height", "64", "--fov-deg", "30",
"--look-ra-deg", "2", "--look-dec-deg", "1",
"--coarse-cell-pixels", "16", "--refine-max-level", "0",
"--exposure", "1000", "--psf-fwhm-pixels", "2.7",
"--psf-moffat-beta", "4.5", "--psf-relative-tail", "1e-8",
"--max-cache-psf-flux", "1", "--psf-min-y", min_y,
"--hdr-output", "--output", str(output)]
print("RUN", command, flush=True)
run = subprocess.run(command, env=env, text=True, capture_output=True, timeout=30)
print(run.stdout, run.stderr, flush=True)
run.check_returncode()
match = re.search(r"PSF splats: cached (\d+), cached wing-clipped (\d+), "
r"direct fallbacks (\d+), discarded below min-Y (\d+)",
run.stdout + run.stderr)
assert match, "missing splat accounting"
stats = tuple(map(int, match.groups()))
assert (stats[0], stats[2], stats[3]) == expected, stats
results.append((stats, fits.read_fits(output.with_name(output.stem + "_HDR.fits"))))
assert results[0][0] == results[1][0], "CPU/HIP statistics differ"
(shape, cpu), (gpu_shape, gpu) = results[0][1], results[1][1]
assert shape == gpu_shape
maximum = 0.0
for (a,), (b,) in zip(struct.iter_unpack(">f", cpu), struct.iter_unpack(">f", gpu)):
assert math.isfinite(a) and math.isfinite(b)
assert math.isclose(a, b, rel_tol=2**-23, abs_tol=0), (a, b)
maximum = max(maximum, abs(a - b))
print(f"min_y={min_y}: statistics match; HDR max_abs={maximum}; exact={cpu == gpu}", flush=True)
if __name__ == "__main__":
main()