Files
GR-raytracing/tests/test_hip_psf.hip
T
wyj e1ec480669 HIP: accelerate PSF accumulation and restore parallel producers
Cooperate across 32 lanes per PSF and use two completion-protected staging slots with complete batch timing. Restore coarse OpenMP event production while serializing shared GPU submissions and direct fallback boundaries.

Add bounded benchmarks, streaming and renderer regressions, and preserve validation evidence and ownership documentation.
2026-09-06 21:38:53 -04:00

102 lines
4.5 KiB
Plaintext

#include "hip_psf.h"
#include <cmath>
#include <cstdio>
#include <cstdlib>
int main() {
constexpr int width = 64, height = 48;
constexpr size_t values = (size_t)width * height * 3;
const PointSpreadFunction psf = {2.7, 4.5};
PsfKernelCache cache = {};
if (psf_kernel_cache_init(&cache, &psf, 1e-8)) {
std::fputs("could not build PSF cache\n", stderr);
return 1;
}
PsfCachedEvent events[3] = {};
const LinearRgb colors[3] = {{1.0, 0.2, 0.6}, {0.1, 0.8, 0.3}, {0.7, 0.4, 1.0}};
const double positions[3][2] = {{23.25, 19.75}, {24.50, 20.125}, {23.875, 21.25}};
for (size_t i = 0; i < 3; ++i) {
if (psf_prepare_cached_event(&events[i], positions[i][0], positions[i][1],
colors[i], 0.1 + 0.03 * i, &psf, &cache,
1.0, 1e-8, 0.0) != 0) {
std::fputs("test event unexpectedly missed the cache\n", stderr);
psf_kernel_cache_destroy(&cache);
return 1;
}
}
double *cpu_hdr = (double *)std::calloc(values, sizeof *cpu_hdr);
double *gpu_hdr = (double *)std::calloc(values, sizeof *gpu_hdr);
if (!cpu_hdr || !gpu_hdr) {
std::fputs("HDR allocation failed\n", stderr);
std::free(cpu_hdr); std::free(gpu_hdr); psf_kernel_cache_destroy(&cache);
return 1;
}
for (const PsfCachedEvent &event : events)
splat_prepared_cached_event(cpu_hdr, width, height, &event, &cache);
char message[256] = {};
HipPsfSink *sink = nullptr;
if (hip_psf_available(message, sizeof message) ||
hip_psf_sink_create(&sink, width, height, &cache, 3, message, sizeof message) ||
hip_psf_sink_submit(sink, events, 3, message, sizeof message) ||
hip_psf_sink_finish(sink, gpu_hdr, message, sizeof message)) {
std::fprintf(stderr, "HIP PSF test failed: %s\n", message);
hip_psf_sink_destroy(sink);
std::free(cpu_hdr); std::free(gpu_hdr); psf_kernel_cache_destroy(&cache);
return 1;
}
HipPsfTiming timing = {};
if (hip_psf_sink_get_timing(sink, &timing) || timing.event_count != 3 ||
timing.batch_count != 1 || timing.timed_batch_count != 1 ||
timing.upload_seconds < 0.0 || timing.kernel_seconds < 0.0 ||
timing.download_seconds < 0.0) {
std::fputs("HIP PSF timing accounting failed\n", stderr);
hip_psf_sink_destroy(sink);
std::free(cpu_hdr); std::free(gpu_hdr); psf_kernel_cache_destroy(&cache);
return 1;
}
double max_abs = 0.0, max_rel = 0.0;
/* Exercise both reusable slots, more than the old 64-batch timing limit,
* immediate caller-buffer reuse, edge clipping and an ordered CPU boundary. */
for (int batch = 0; batch < 70; ++batch) {
for (int j = 0; j < 3; ++j) {
events[j] = PsfCachedEvent{};
const int result = psf_prepare_cached_event(&events[j],
-4.75 + (batch * 7 + j * 13) % 75,
-3.125 + (batch * 11 + j * 5) % 58, colors[j],
batch % 7 == 0 ? 1000.0 : 0.01, &psf, &cache, 1e8, 1e-8, 0.0);
if (result != 0 && result != 2) return 1;
splat_prepared_cached_event(cpu_hdr, width, height, &events[j], &cache);
}
if (hip_psf_sink_submit(sink, events, 3, message, sizeof message)) {
std::fprintf(stderr, "streaming submit failed: %s\n", message); return 1;
}
for (auto &event : events) event = PsfCachedEvent{};
if (batch == 35) {
if (hip_psf_sink_finish(sink, gpu_hdr, message, sizeof message)) return 1;
for (double *hdr : {cpu_hdr, gpu_hdr})
splat_moffat_direct(hdr, width, height, 12.25, 20.75, colors[0],
0.2, &psf, 1e-8, 0.0);
if (hip_psf_sink_load_hdr(sink, gpu_hdr, message, sizeof message)) return 1;
}
}
if (hip_psf_sink_finish(sink, gpu_hdr, message, sizeof message) ||
hip_psf_sink_get_timing(sink, &timing) || timing.event_count != 213 ||
timing.batch_count != 71 || timing.timed_batch_count != 71) {
std::fprintf(stderr, "streaming completion/accounting failed: %s\n", message);
return 1;
}
for (size_t i = 0; i < values; ++i) {
if (!std::isfinite(cpu_hdr[i]) || !std::isfinite(gpu_hdr[i])) return 1;
const double absolute = std::fabs(cpu_hdr[i] - gpu_hdr[i]);
max_abs = std::fmax(max_abs, absolute);
if (std::fabs(cpu_hdr[i]) > 1e-30)
max_rel = std::fmax(max_rel, absolute / std::fabs(cpu_hdr[i]));
}
std::printf("HIP PSF cache comparison: max_abs=%.17g max_rel=%.17g\n", max_abs, max_rel);
hip_psf_sink_destroy(sink);
std::free(cpu_hdr); std::free(gpu_hdr); psf_kernel_cache_destroy(&cache);
return max_abs <= 1e-10 && max_rel <= 1e-12 ? 0 : 1;
}