Cooperate across 32 lanes per PSF and use two completion-protected staging slots with complete batch timing. Restore coarse OpenMP event production while serializing shared GPU submissions and direct fallback boundaries. Add bounded benchmarks, streaming and renderer regressions, and preserve validation evidence and ownership documentation.
97 lines
4.4 KiB
Plaintext
97 lines
4.4 KiB
Plaintext
#include "hip_psf.h"
|
|
#include <hip/hip_runtime.h>
|
|
#include <omp.h>
|
|
#include <cmath>
|
|
#include <cstdio>
|
|
#include <cstdlib>
|
|
#include <cstring>
|
|
#include <vector>
|
|
|
|
/* Bounded production-cache replay. One process runs CPU reference then HIP;
|
|
* no catalog loading or geodesic work. Arguments: events, spread pixels. */
|
|
int main(int argc, char **argv) {
|
|
if (argc == 2 && !std::strcmp(argv[1], "--help")) {
|
|
std::puts("Usage: benchmark_hip_psf [EVENTS [SPREAD_PIXELS]]\n"
|
|
" EVENTS Cached events, 1..65536 (default: 4096)\n"
|
|
" SPREAD_PIXELS Star-center spread, 1..3840 (default: 16)\n"
|
|
"Runs a serial CPU reference, then HIP, at 3840x2160; PSF 2.7/4.5, tail 1e-8.\n"
|
|
"Use an external timeout and run only one benchmark process at a time.");
|
|
return 0;
|
|
}
|
|
auto number = [](const char *s, long upper) {
|
|
char *end = nullptr;
|
|
const long n = std::strtol(s, &end, 10);
|
|
return s != end && !*end && n >= 1 && n <= upper ? (int)n : 0;
|
|
};
|
|
const int count = argc > 1 ? number(argv[1], 65536) : 4096;
|
|
const int spread = argc > 2 ? number(argv[2], 3840) : 16;
|
|
if (argc > 3 || !count || !spread) {
|
|
std::fputs("Invalid bounded benchmark arguments; see --help.\n", stderr);
|
|
return 2;
|
|
}
|
|
std::setvbuf(stdout, nullptr, _IOLBF, 0);
|
|
const int width = 3840, height = 2160;
|
|
const size_t values = (size_t)width * height * 3;
|
|
PsfKernelCache cache = {};
|
|
const PointSpreadFunction psf = {2.7, 4.5};
|
|
if (psf_kernel_cache_init(&cache, &psf, 1e-8)) return 1;
|
|
psf_kernel_cache_report_ready(&cache, stdout);
|
|
std::vector<PsfCachedEvent> events(count);
|
|
unsigned state = 12345;
|
|
auto uniform = [&]() { state = state * 1664525u + 1013904223u;
|
|
return (state >> 8) * (1.0 / 16777216.0); };
|
|
for (auto &event : events) {
|
|
const double x = (width - spread) * 0.5 + uniform() * spread;
|
|
const double sy = spread < height ? spread : height;
|
|
const double y = (height - sy) * 0.5 + uniform() * sy;
|
|
const LinearRgb color = {0.1 + uniform(), 0.1 + uniform(), 0.1 + uniform()};
|
|
const int result = psf_prepare_cached_event(&event, x, y, color,
|
|
0.1 + uniform(), &psf, &cache, 1e8, 1e-8, 0);
|
|
if (result != 0) return 1;
|
|
}
|
|
std::vector<double> cpu(values, 0), gpu(values, 0);
|
|
double start = omp_get_wtime();
|
|
for (const auto &event : events)
|
|
splat_prepared_cached_event(cpu.data(), width, height, &event, &cache);
|
|
const double cpu_seconds = omp_get_wtime() - start;
|
|
char message[256] = {};
|
|
hipDeviceProp_t prop;
|
|
if (hipGetDeviceProperties(&prop, 0) != hipSuccess) return 1;
|
|
std::printf("device=%s arch=%s events=%d spread=%d frame=%dx%d event_bytes=%zu CPU_reference=%.6f s\n",
|
|
prop.name, prop.gcnArchName, count, spread, width, height, sizeof(PsfCachedEvent), cpu_seconds);
|
|
HipPsfSink *sink = nullptr;
|
|
start = omp_get_wtime();
|
|
if (hip_psf_sink_create(&sink, width, height, &cache, 16384, message, sizeof message)) {
|
|
std::fprintf(stderr, "create: %s\n", message); return 1;
|
|
}
|
|
const double create_seconds = omp_get_wtime() - start;
|
|
start = omp_get_wtime();
|
|
for (size_t i = 0; i < events.size(); i += 16384) {
|
|
size_t n = events.size() - i;
|
|
if (n > 16384) n = 16384;
|
|
if (hip_psf_sink_submit(sink, events.data() + i, n, message, sizeof message)) {
|
|
std::fprintf(stderr, "submit: %s\n", message); return 1;
|
|
}
|
|
}
|
|
if (hip_psf_sink_finish(sink, gpu.data(), message, sizeof message)) {
|
|
std::fprintf(stderr, "finish: %s\n", message); return 1;
|
|
}
|
|
const double wall = omp_get_wtime() - start;
|
|
HipPsfTiming timing = {};
|
|
hip_psf_sink_get_timing(sink, &timing);
|
|
double max_abs = 0, max_rel = 0, sum = 0, diff = 0;
|
|
for (size_t i = 0; i < values; ++i) {
|
|
if (!std::isfinite(gpu[i])) return 1;
|
|
const double d = std::fabs(gpu[i] - cpu[i]);
|
|
max_abs = std::fmax(max_abs, d);
|
|
if (cpu[i] != 0) max_rel = std::fmax(max_rel, d / std::fabs(cpu[i]));
|
|
sum += cpu[i]; diff += gpu[i] - cpu[i];
|
|
}
|
|
std::printf("HIP create=%.6f replay_wall=%.6f upload=%.6f kernel=%.6f download=%.6f batches=%zu timed=%zu max_abs=%.17g max_rel=%.17g flux_rel=%.17g\n",
|
|
create_seconds, wall, timing.upload_seconds, timing.kernel_seconds, timing.download_seconds,
|
|
timing.batch_count, timing.timed_batch_count, max_abs, max_rel, diff / sum);
|
|
hip_psf_sink_destroy(sink);
|
|
psf_kernel_cache_destroy(&cache);
|
|
return max_abs < 1e-10 && max_rel < 1e-10 ? 0 : 1;
|
|
}
|