HIP: accelerate PSF accumulation and restore parallel producers
Cooperate across 32 lanes per PSF and use two completion-protected staging slots with complete batch timing. Restore coarse OpenMP event production while serializing shared GPU submissions and direct fallback boundaries. Add bounded benchmarks, streaming and renderer regressions, and preserve validation evidence and ownership documentation.
This commit is contained in:
1 parent
265d7b95d5
commit
e1ec480669
45 files changed
+10251
-95
No files matched your search
@@ -0,0 +1,96 @@
|
||||
#include "hip_psf.h"
|
||||
#include <hip/hip_runtime.h>
|
||||
#include <omp.h>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <vector>
|
||||
|
||||
/* Bounded production-cache replay. One process runs CPU reference then HIP;
|
||||
* no catalog loading or geodesic work. Arguments: events, spread pixels. */
|
||||
int main(int argc, char **argv) {
|
||||
if (argc == 2 && !std::strcmp(argv[1], "--help")) {
|
||||
std::puts("Usage: benchmark_hip_psf [EVENTS [SPREAD_PIXELS]]\n"
|
||||
" EVENTS Cached events, 1..65536 (default: 4096)\n"
|
||||
" SPREAD_PIXELS Star-center spread, 1..3840 (default: 16)\n"
|
||||
"Runs a serial CPU reference, then HIP, at 3840x2160; PSF 2.7/4.5, tail 1e-8.\n"
|
||||
"Use an external timeout and run only one benchmark process at a time.");
|
||||
return 0;
|
||||
}
|
||||
auto number = [](const char *s, long upper) {
|
||||
char *end = nullptr;
|
||||
const long n = std::strtol(s, &end, 10);
|
||||
return s != end && !*end && n >= 1 && n <= upper ? (int)n : 0;
|
||||
};
|
||||
const int count = argc > 1 ? number(argv[1], 65536) : 4096;
|
||||
const int spread = argc > 2 ? number(argv[2], 3840) : 16;
|
||||
if (argc > 3 || !count || !spread) {
|
||||
std::fputs("Invalid bounded benchmark arguments; see --help.\n", stderr);
|
||||
return 2;
|
||||
}
|
||||
std::setvbuf(stdout, nullptr, _IOLBF, 0);
|
||||
const int width = 3840, height = 2160;
|
||||
const size_t values = (size_t)width * height * 3;
|
||||
PsfKernelCache cache = {};
|
||||
const PointSpreadFunction psf = {2.7, 4.5};
|
||||
if (psf_kernel_cache_init(&cache, &psf, 1e-8)) return 1;
|
||||
psf_kernel_cache_report_ready(&cache, stdout);
|
||||
std::vector<PsfCachedEvent> events(count);
|
||||
unsigned state = 12345;
|
||||
auto uniform = [&]() { state = state * 1664525u + 1013904223u;
|
||||
return (state >> 8) * (1.0 / 16777216.0); };
|
||||
for (auto &event : events) {
|
||||
const double x = (width - spread) * 0.5 + uniform() * spread;
|
||||
const double sy = spread < height ? spread : height;
|
||||
const double y = (height - sy) * 0.5 + uniform() * sy;
|
||||
const LinearRgb color = {0.1 + uniform(), 0.1 + uniform(), 0.1 + uniform()};
|
||||
const int result = psf_prepare_cached_event(&event, x, y, color,
|
||||
0.1 + uniform(), &psf, &cache, 1e8, 1e-8, 0);
|
||||
if (result != 0) return 1;
|
||||
}
|
||||
std::vector<double> cpu(values, 0), gpu(values, 0);
|
||||
double start = omp_get_wtime();
|
||||
for (const auto &event : events)
|
||||
splat_prepared_cached_event(cpu.data(), width, height, &event, &cache);
|
||||
const double cpu_seconds = omp_get_wtime() - start;
|
||||
char message[256] = {};
|
||||
hipDeviceProp_t prop;
|
||||
if (hipGetDeviceProperties(&prop, 0) != hipSuccess) return 1;
|
||||
std::printf("device=%s arch=%s events=%d spread=%d frame=%dx%d event_bytes=%zu CPU_reference=%.6f s\n",
|
||||
prop.name, prop.gcnArchName, count, spread, width, height, sizeof(PsfCachedEvent), cpu_seconds);
|
||||
HipPsfSink *sink = nullptr;
|
||||
start = omp_get_wtime();
|
||||
if (hip_psf_sink_create(&sink, width, height, &cache, 16384, message, sizeof message)) {
|
||||
std::fprintf(stderr, "create: %s\n", message); return 1;
|
||||
}
|
||||
const double create_seconds = omp_get_wtime() - start;
|
||||
start = omp_get_wtime();
|
||||
for (size_t i = 0; i < events.size(); i += 16384) {
|
||||
size_t n = events.size() - i;
|
||||
if (n > 16384) n = 16384;
|
||||
if (hip_psf_sink_submit(sink, events.data() + i, n, message, sizeof message)) {
|
||||
std::fprintf(stderr, "submit: %s\n", message); return 1;
|
||||
}
|
||||
}
|
||||
if (hip_psf_sink_finish(sink, gpu.data(), message, sizeof message)) {
|
||||
std::fprintf(stderr, "finish: %s\n", message); return 1;
|
||||
}
|
||||
const double wall = omp_get_wtime() - start;
|
||||
HipPsfTiming timing = {};
|
||||
hip_psf_sink_get_timing(sink, &timing);
|
||||
double max_abs = 0, max_rel = 0, sum = 0, diff = 0;
|
||||
for (size_t i = 0; i < values; ++i) {
|
||||
if (!std::isfinite(gpu[i])) return 1;
|
||||
const double d = std::fabs(gpu[i] - cpu[i]);
|
||||
max_abs = std::fmax(max_abs, d);
|
||||
if (cpu[i] != 0) max_rel = std::fmax(max_rel, d / std::fabs(cpu[i]));
|
||||
sum += cpu[i]; diff += gpu[i] - cpu[i];
|
||||
}
|
||||
std::printf("HIP create=%.6f replay_wall=%.6f upload=%.6f kernel=%.6f download=%.6f batches=%zu timed=%zu max_abs=%.17g max_rel=%.17g flux_rel=%.17g\n",
|
||||
create_seconds, wall, timing.upload_seconds, timing.kernel_seconds, timing.download_seconds,
|
||||
timing.batch_count, timing.timed_batch_count, max_abs, max_rel, diff / sum);
|
||||
hip_psf_sink_destroy(sink);
|
||||
psf_kernel_cache_destroy(&cache);
|
||||
return max_abs < 1e-10 && max_rel < 1e-10 ? 0 : 1;
|
||||
}
|
||||
+32
-1
@@ -57,7 +57,38 @@ int main() {
|
||||
return 1;
|
||||
}
|
||||
double max_abs = 0.0, max_rel = 0.0;
|
||||
/* Exercise both reusable slots, more than the old 64-batch timing limit,
|
||||
* immediate caller-buffer reuse, edge clipping and an ordered CPU boundary. */
|
||||
for (int batch = 0; batch < 70; ++batch) {
|
||||
for (int j = 0; j < 3; ++j) {
|
||||
events[j] = PsfCachedEvent{};
|
||||
const int result = psf_prepare_cached_event(&events[j],
|
||||
-4.75 + (batch * 7 + j * 13) % 75,
|
||||
-3.125 + (batch * 11 + j * 5) % 58, colors[j],
|
||||
batch % 7 == 0 ? 1000.0 : 0.01, &psf, &cache, 1e8, 1e-8, 0.0);
|
||||
if (result != 0 && result != 2) return 1;
|
||||
splat_prepared_cached_event(cpu_hdr, width, height, &events[j], &cache);
|
||||
}
|
||||
if (hip_psf_sink_submit(sink, events, 3, message, sizeof message)) {
|
||||
std::fprintf(stderr, "streaming submit failed: %s\n", message); return 1;
|
||||
}
|
||||
for (auto &event : events) event = PsfCachedEvent{};
|
||||
if (batch == 35) {
|
||||
if (hip_psf_sink_finish(sink, gpu_hdr, message, sizeof message)) return 1;
|
||||
for (double *hdr : {cpu_hdr, gpu_hdr})
|
||||
splat_moffat_direct(hdr, width, height, 12.25, 20.75, colors[0],
|
||||
0.2, &psf, 1e-8, 0.0);
|
||||
if (hip_psf_sink_load_hdr(sink, gpu_hdr, message, sizeof message)) return 1;
|
||||
}
|
||||
}
|
||||
if (hip_psf_sink_finish(sink, gpu_hdr, message, sizeof message) ||
|
||||
hip_psf_sink_get_timing(sink, &timing) || timing.event_count != 213 ||
|
||||
timing.batch_count != 71 || timing.timed_batch_count != 71) {
|
||||
std::fprintf(stderr, "streaming completion/accounting failed: %s\n", message);
|
||||
return 1;
|
||||
}
|
||||
for (size_t i = 0; i < values; ++i) {
|
||||
if (!std::isfinite(cpu_hdr[i]) || !std::isfinite(gpu_hdr[i])) return 1;
|
||||
const double absolute = std::fabs(cpu_hdr[i] - gpu_hdr[i]);
|
||||
max_abs = std::fmax(max_abs, absolute);
|
||||
if (std::fabs(cpu_hdr[i]) > 1e-30)
|
||||
@@ -66,5 +97,5 @@ int main() {
|
||||
std::printf("HIP PSF cache comparison: max_abs=%.17g max_rel=%.17g\n", max_abs, max_rel);
|
||||
hip_psf_sink_destroy(sink);
|
||||
std::free(cpu_hdr); std::free(gpu_hdr); psf_kernel_cache_destroy(&cache);
|
||||
return max_abs <= 1e-12 && max_rel <= 1e-12 ? 0 : 1;
|
||||
return max_abs <= 1e-10 && max_rel <= 1e-12 ? 0 : 1;
|
||||
}
|
||||
@@ -0,0 +1,61 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Sequential, bounded CPU/HIP renderer checks for shared-HDR fallback safety."""
|
||||
import argparse
|
||||
import importlib.util
|
||||
import math
|
||||
import os
|
||||
from pathlib import Path
|
||||
import re
|
||||
import struct
|
||||
import subprocess
|
||||
import tempfile
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("cpu", type=Path, help="Minkowski CPU executable")
|
||||
parser.add_argument("hip", type=Path, help="Minkowski HIP executable")
|
||||
args = parser.parse_args()
|
||||
repo = Path(__file__).resolve().parents[1]
|
||||
spec = importlib.util.spec_from_file_location("fits", repo / "scripts/fits_floatdiff.py")
|
||||
fits = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(fits)
|
||||
env = dict(os.environ, OMP_NUM_THREADS="4")
|
||||
with tempfile.TemporaryDirectory(prefix="gr-hip-renderer-test-") as directory:
|
||||
for min_y, expected in [("0", (2, 1, 0)), ("1e30", (0, 0, 3))]:
|
||||
results = []
|
||||
for label, binary in [("cpu", args.cpu), ("hip", args.hip)]:
|
||||
output = Path(directory) / f"{label}-{min_y}.png"
|
||||
command = [str(binary.resolve()), "--catalog",
|
||||
str(repo / "tests/data/hip_psf_mixed_catalog.csv"),
|
||||
"--width", "96", "--height", "64", "--fov-deg", "30",
|
||||
"--look-ra-deg", "2", "--look-dec-deg", "1",
|
||||
"--coarse-cell-pixels", "16", "--refine-max-level", "0",
|
||||
"--exposure", "1000", "--psf-fwhm-pixels", "2.7",
|
||||
"--psf-moffat-beta", "4.5", "--psf-relative-tail", "1e-8",
|
||||
"--max-cache-psf-flux", "1", "--psf-min-y", min_y,
|
||||
"--hdr-output", "--output", str(output)]
|
||||
print("RUN", command, flush=True)
|
||||
run = subprocess.run(command, env=env, text=True, capture_output=True, timeout=30)
|
||||
print(run.stdout, run.stderr, flush=True)
|
||||
run.check_returncode()
|
||||
match = re.search(r"PSF splats: cached (\d+), cached wing-clipped (\d+), "
|
||||
r"direct fallbacks (\d+), discarded below min-Y (\d+)",
|
||||
run.stdout + run.stderr)
|
||||
assert match, "missing splat accounting"
|
||||
stats = tuple(map(int, match.groups()))
|
||||
assert (stats[0], stats[2], stats[3]) == expected, stats
|
||||
results.append((stats, fits.read_fits(output.with_name(output.stem + "_HDR.fits"))))
|
||||
assert results[0][0] == results[1][0], "CPU/HIP statistics differ"
|
||||
(shape, cpu), (gpu_shape, gpu) = results[0][1], results[1][1]
|
||||
assert shape == gpu_shape
|
||||
maximum = 0.0
|
||||
for (a,), (b,) in zip(struct.iter_unpack(">f", cpu), struct.iter_unpack(">f", gpu)):
|
||||
assert math.isfinite(a) and math.isfinite(b)
|
||||
assert math.isclose(a, b, rel_tol=2**-23, abs_tol=0), (a, b)
|
||||
maximum = max(maximum, abs(a - b))
|
||||
print(f"min_y={min_y}: statistics match; HDR max_abs={maximum}; exact={cpu == gpu}", flush=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in new issue
Block a user