#include "hip_psf.h" #include #include #include #include #include #include #include /* Bounded production-cache replay. One process runs CPU reference then HIP; * no catalog loading or geodesic work. Arguments: events, spread pixels. */ int main(int argc, char **argv) { if (argc == 2 && !std::strcmp(argv[1], "--help")) { std::puts("Usage: benchmark_hip_psf [EVENTS [SPREAD_PIXELS]]\n" " EVENTS Cached events, 1..65536 (default: 4096)\n" " SPREAD_PIXELS Star-center spread, 1..3840 (default: 16)\n" "Runs a serial CPU reference, then HIP, at 3840x2160; PSF 2.7/4.5, tail 1e-8.\n" "Use an external timeout and run only one benchmark process at a time."); return 0; } auto number = [](const char *s, long upper) { char *end = nullptr; const long n = std::strtol(s, &end, 10); return s != end && !*end && n >= 1 && n <= upper ? (int)n : 0; }; const int count = argc > 1 ? number(argv[1], 65536) : 4096; const int spread = argc > 2 ? number(argv[2], 3840) : 16; if (argc > 3 || !count || !spread) { std::fputs("Invalid bounded benchmark arguments; see --help.\n", stderr); return 2; } std::setvbuf(stdout, nullptr, _IOLBF, 0); const int width = 3840, height = 2160; const size_t values = (size_t)width * height * 3; PsfKernelCache cache = {}; const PointSpreadFunction psf = {2.7, 4.5}; if (psf_kernel_cache_init(&cache, &psf, 1e-8)) return 1; psf_kernel_cache_report_ready(&cache, stdout); std::vector events(count); unsigned state = 12345; auto uniform = [&]() { state = state * 1664525u + 1013904223u; return (state >> 8) * (1.0 / 16777216.0); }; for (auto &event : events) { const double x = (width - spread) * 0.5 + uniform() * spread; const double sy = spread < height ? spread : height; const double y = (height - sy) * 0.5 + uniform() * sy; const LinearRgb color = {0.1 + uniform(), 0.1 + uniform(), 0.1 + uniform()}; const int result = psf_prepare_cached_event(&event, x, y, color, 0.1 + uniform(), &psf, &cache, 1e8, 1e-8, 0); if (result != 0) return 1; } std::vector cpu(values, 0), gpu(values, 0); double start = omp_get_wtime(); for (const auto &event : events) splat_prepared_cached_event(cpu.data(), width, height, &event, &cache); const double cpu_seconds = omp_get_wtime() - start; char message[256] = {}; hipDeviceProp_t prop; if (hipGetDeviceProperties(&prop, 0) != hipSuccess) return 1; std::printf("device=%s arch=%s events=%d spread=%d frame=%dx%d event_bytes=%zu CPU_reference=%.6f s\n", prop.name, prop.gcnArchName, count, spread, width, height, sizeof(PsfCachedEvent), cpu_seconds); HipPsfSink *sink = nullptr; start = omp_get_wtime(); if (hip_psf_sink_create(&sink, width, height, &cache, 16384, message, sizeof message)) { std::fprintf(stderr, "create: %s\n", message); return 1; } const double create_seconds = omp_get_wtime() - start; start = omp_get_wtime(); for (size_t i = 0; i < events.size(); i += 16384) { size_t n = events.size() - i; if (n > 16384) n = 16384; if (hip_psf_sink_submit(sink, events.data() + i, n, message, sizeof message)) { std::fprintf(stderr, "submit: %s\n", message); return 1; } } if (hip_psf_sink_finish(sink, gpu.data(), message, sizeof message)) { std::fprintf(stderr, "finish: %s\n", message); return 1; } const double wall = omp_get_wtime() - start; HipPsfTiming timing = {}; hip_psf_sink_get_timing(sink, &timing); double max_abs = 0, max_rel = 0, sum = 0, diff = 0; for (size_t i = 0; i < values; ++i) { if (!std::isfinite(gpu[i])) return 1; const double d = std::fabs(gpu[i] - cpu[i]); max_abs = std::fmax(max_abs, d); if (cpu[i] != 0) max_rel = std::fmax(max_rel, d / std::fabs(cpu[i])); sum += cpu[i]; diff += gpu[i] - cpu[i]; } std::printf("HIP create=%.6f replay_wall=%.6f upload=%.6f kernel=%.6f download=%.6f batches=%zu timed=%zu max_abs=%.17g max_rel=%.17g flux_rel=%.17g\n", create_seconds, wall, timing.upload_seconds, timing.kernel_seconds, timing.download_seconds, timing.batch_count, timing.timed_batch_count, max_abs, max_rel, diff / sum); hip_psf_sink_destroy(sink); psf_kernel_cache_destroy(&cache); return max_abs < 1e-10 && max_rel < 1e-10 ? 0 : 1; }