HIP: accelerate PSF accumulation and restore parallel producers

Cooperate across 32 lanes per PSF and use two completion-protected staging slots with complete batch timing. Restore coarse OpenMP event production while serializing shared GPU submissions and direct fallback boundaries.

Add bounded benchmarks, streaming and renderer regressions, and preserve validation evidence and ownership documentation.
This commit is contained in:
wyj committed 2026-09-06 21:38:53 -04:00
1 parent 265d7b95d5
commit e1ec480669
45 files changed
+10251 -95

No files matched your search

+133 -22
View File
@@ -133,6 +133,9 @@ typedef struct {
const PsfKernelCache *cache;
#ifdef PSF_BACKEND_HIP
HipPsfSink *hip;
omp_lock_t *hip_lock; /* Shared sink lock; acquired only per chunk/fallback. */
int borrowed_hip;
double submit_seconds;
char hip_message[256];
HipPsfTiming hip_timing;
#endif
@@ -176,7 +179,7 @@ static int psf_event_sink_init(PsfEventSink *sink, double *hdr, int width,
return 0;
}
static int psf_event_sink_flush(PsfEventSink *sink) {
static int psf_event_sink_flush_unlocked(PsfEventSink *sink) {
if (sink->failed)
return -1;
#ifdef PSF_BACKEND_HIP
@@ -197,8 +200,27 @@ static int psf_event_sink_flush(PsfEventSink *sink) {
return 0;
}
static int psf_event_sink_flush(PsfEventSink *sink) {
#ifdef PSF_BACKEND_HIP
const double start = omp_get_wtime();
if (sink->hip_lock) omp_set_lock(sink->hip_lock);
#endif
const int result = psf_event_sink_flush_unlocked(sink);
#ifdef PSF_BACKEND_HIP
if (sink->hip_lock) omp_unset_lock(sink->hip_lock);
sink->submit_seconds += omp_get_wtime() - start;
#endif
return result;
}
/* A borrowed sink calls this only while holding hip_lock across the entire
* download -> CPU fallback -> upload transaction. */
static int psf_event_sink_finish_for_cpu(PsfEventSink *sink) {
#ifdef PSF_BACKEND_HIP
if (psf_event_sink_flush_unlocked(sink))
#else
if (psf_event_sink_flush(sink))
#endif
return -1;
#ifdef PSF_BACKEND_HIP
if (sink->hip != NULL && hip_psf_sink_finish(sink->hip, sink->hdr,
@@ -226,6 +248,13 @@ static int psf_event_sink_resume_gpu(PsfEventSink *sink) {
}
static int psf_event_sink_destroy(PsfEventSink *sink) {
#ifdef PSF_BACKEND_HIP
if (sink->borrowed_hip) {
const int result = psf_event_sink_flush(sink);
free(sink->events);
return result;
}
#endif
int result = psf_event_sink_finish_for_cpu(sink);
#ifdef PSF_BACKEND_HIP
if (sink->hip != NULL && hip_psf_sink_get_timing(sink->hip, &sink->hip_timing)) {
@@ -238,7 +267,7 @@ static int psf_event_sink_destroy(PsfEventSink *sink) {
return result;
}
#if FRAME_PSF_EVENT_SINK
#if FRAME_PSF_EVENT_SINK || defined(PSF_BACKEND_HIP)
static void psf_event_sink_emit(PsfEventSink *sink, const PsfCachedEvent *event) {
if (sink->failed)
return;
@@ -1086,16 +1115,27 @@ static int splat_catalog_tile(const Star *stars, size_t count,
&event, image_x, image_y, color, flux, context->psf, context->psf_cache,
context->max_cache_psf_flux, context->psf_relative_tail, context->psf_min_y);
if (direct_fallback == 1) {
/* A direct fallback forms an explicit ordered CPU/GPU boundary. */
if (psf_event_sink_finish_for_cpu(context->event_sink))
return -1;
splat_moffat_direct(context->hdr, context->width, context->height, image_x,
image_y, color, flux, context->psf,
context->psf_relative_tail, context->psf_min_y);
if (psf_event_sink_resume_gpu(context->event_sink))
return -1;
/* Exclude every other submit and fallback until HDR is back on device. */
#ifdef PSF_BACKEND_HIP
const double submit_start = omp_get_wtime();
if (context->event_sink->hip_lock)
omp_set_lock(context->event_sink->hip_lock);
#endif
int failed = psf_event_sink_finish_for_cpu(context->event_sink);
if (!failed) {
splat_moffat_direct(context->hdr, context->width, context->height, image_x,
image_y, color, flux, context->psf,
context->psf_relative_tail, context->psf_min_y);
failed = psf_event_sink_resume_gpu(context->event_sink);
}
#ifdef PSF_BACKEND_HIP
if (context->event_sink->hip_lock)
omp_unset_lock(context->event_sink->hip_lock);
context->event_sink->submit_seconds += omp_get_wtime() - submit_start;
#endif
if (failed) return -1;
} else if (direct_fallback != 3) {
#if FRAME_PSF_EVENT_SINK
#if FRAME_PSF_EVENT_SINK || defined(PSF_BACKEND_HIP)
psf_event_sink_emit(context->event_sink, &event);
#else
splat_prepared_cached_event(context->hdr, context->width, context->height,
@@ -1261,20 +1301,91 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
mesh->triangle_count);
#ifdef PSF_BACKEND_HIP
/* A single GPU HDR framebuffer owns all cached events. Keep catalog/lens
* work serial for this first direct-atomic integration; the CPU parallel
* private-HDR path remains the PSF_BACKEND=cpu implementation. */
const CatalogSplatStats hip_stats = splat_catalog_triangles(
mesh, catalog, hdr, width, height, exposure, psf, psf_cache,
max_magnification, max_cache_psf_flux, psf_relative_tail, psf_min_y,
0, mesh->triangle_count, NULL);
copy_psf_splat_stats(psf_stats, hip_stats);
if (hip_stats.failed)
return SIZE_MAX;
/* CPU workers own only bounded event chunks. A single device cache/HDR is
* borrowed under a coarse submission lock, never duplicated per worker. */
PsfEventSink owner;
if (psf_event_sink_init(&owner, hdr, width, height, psf_cache)) return SIZE_MAX;
omp_lock_t submit_lock;
omp_init_lock(&submit_lock);
size_t hip_images = 0, hip_direct = 0, hip_clipped = 0, hip_discarded = 0;
int hip_failed = 0, hip_workers = 0;
double hip_generate_seconds = 0.0, hip_submit_seconds = 0.0;
const double hip_start = omp_get_wtime();
#ifdef GR_DEBUG
double hip_max_magnification = 0.0;
size_t hip_clamped = 0;
#pragma omp parallel reduction(+ : hip_images, hip_direct, hip_clipped, hip_discarded, hip_clamped, hip_generate_seconds, hip_submit_seconds) reduction(max : hip_failed, hip_max_magnification)
#else
#pragma omp parallel reduction(+ : hip_images, hip_direct, hip_clipped, hip_discarded, hip_generate_seconds, hip_submit_seconds) reduction(max : hip_failed)
#endif
{
const double worker_start = omp_get_wtime();
#pragma omp single
hip_workers = omp_get_num_threads();
PsfEventSink local = {.hdr = hdr, .width = width, .height = height,
.cache = psf_cache, .hip = owner.hip, .hip_lock = &submit_lock,
.borrowed_hip = 1};
local.events = malloc(PSF_EVENT_SINK_CAPACITY * sizeof *local.events);
if (!local.events) {
snprintf(local.hip_message, sizeof local.hip_message, "worker event allocation failed");
psf_event_sink_mark_failed(&local, "producer initialization");
}
const size_t worker = (size_t)omp_get_thread_num();
const size_t workers = (size_t)omp_get_num_threads();
size_t completed = 0, next_report = 8;
if (progress && progress->worker_callback)
progress->worker_callback(progress->context, worker, workers, 0, 0);
#pragma omp for schedule(dynamic, 1) nowait
for (size_t t = 0; t < mesh->triangle_count; ++t) {
if (local.failed) continue;
const CatalogSplatStats stats = splat_catalog_triangles(
mesh, catalog, hdr, width, height, exposure, psf, psf_cache,
max_magnification, max_cache_psf_flux, psf_relative_tail, psf_min_y,
t, t + 1, &local);
hip_images += stats.images;
hip_direct += stats.direct_fallbacks;
hip_clipped += stats.cached_wing_clipped;
hip_discarded += stats.discarded_below_min_y;
hip_failed |= stats.failed;
#ifdef GR_DEBUG
hip_max_magnification = fmax(hip_max_magnification, stats.max_raw_magnification);
hip_clamped += stats.magnification_clamped_triangles;
#endif
++completed;
if (progress && progress->worker_callback && completed == next_report) {
progress->worker_callback(progress->context, worker, workers, completed, 0);
if (next_report <= SIZE_MAX / 2) next_report *= 2;
}
}
if (psf_event_sink_destroy(&local)) hip_failed = 1;
hip_submit_seconds += local.submit_seconds;
hip_generate_seconds += omp_get_wtime() - worker_start - local.submit_seconds;
if (progress && progress->worker_callback)
progress->worker_callback(progress->context, worker, workers, completed, 1);
}
omp_destroy_lock(&submit_lock);
if (psf_event_sink_destroy(&owner)) hip_failed = 1;
fprintf(stderr, "HIP producers: %d workers; summed generation %.3f s, submission/fallback %.3f s; wall %.3f s\n",
hip_workers, hip_generate_seconds, hip_submit_seconds, omp_get_wtime() - hip_start);
copy_psf_splat_stats(psf_stats, (CatalogSplatStats){
.images = hip_images, .direct_fallbacks = hip_direct,
.cached_wing_clipped = hip_clipped, .discarded_below_min_y = hip_discarded,
.gpu_event_count = owner.hip_timing.event_count,
.gpu_batch_count = owner.hip_timing.batch_count,
.gpu_timed_batch_count = owner.hip_timing.timed_batch_count,
.gpu_upload_seconds = owner.hip_timing.upload_seconds,
.gpu_kernel_seconds = owner.hip_timing.kernel_seconds,
.gpu_download_seconds = owner.hip_timing.download_seconds,
#ifdef GR_DEBUG
.max_raw_magnification = hip_max_magnification,
.magnification_clamped_triangles = hip_clamped,
#endif
});
if (hip_failed) return SIZE_MAX;
if (progress != NULL && progress->callback != NULL)
progress->callback(progress->context, FRAME_SPLAT_PROGRESS_END,
mesh->triangle_count, mesh->triangle_count);
return hip_stats.images;
return hip_images;
#endif
const size_t pixel_count = (size_t)width * height * 3;