From 5fd8e3819dc6b68a426551757f9c816059f442e7 Mon Sep 17 00:00:00 2001 From: Yingjie Wang Date: Fri, 28 Aug 2026 17:20:00 -0400 Subject: [PATCH] Spacetime: free analytic render worker count --- psf_lookup_optimization_plan.md | 5 ++++- src/frame.c | 21 ++++++++++++--------- src/frame.h | 1 + src/main.c | 7 +++++-- src/spacetime.h | 5 +++++ src/spacetime_common.c | 5 +++++ tests/test_frame.c | 10 ++++++---- 7 files changed, 38 insertions(+), 16 deletions(-) diff --git a/psf_lookup_optimization_plan.md b/psf_lookup_optimization_plan.md index ff6e43d..9198fdd 100644 --- a/psf_lookup_optimization_plan.md +++ b/psf_lookup_optimization_plan.md @@ -84,7 +84,10 @@ chosen from the validation experiments, not inferred from a performance goal. For every image event, compute color and final flux exactly as in the existing path. Then multiply the scalar cached PSF weights by that color and flux and add them to the worker's existing private HDR buffer. Preserve the current -private-HDR ownership, memory cap, reduction ordering, and serial fallback; +private-HDR ownership, reduction ordering, and serial fallback. Analytic +backends use all OpenMP render workers; a future numerical backend must opt in +to the private-HDR memory cap when budgeting its metric slabs and evaluator +workspaces; this plan does not introduce atomics, locks, or a different parallel partitioning. diff --git a/src/frame.c b/src/frame.c index 01bdfff..df4c515 100644 --- a/src/frame.c +++ b/src/frame.c @@ -8,9 +8,9 @@ #include #include -/* Private HDR buffers make splatting independent across workers. This is an - * allocation cap, not a rendering parameter: callers transparently fall back - * to the serial implementation when the frame is too large for two buffers. */ +/* Numerical metric backends may reserve substantial memory for slabs and + * thread-local evaluators, so they retain this private-HDR allocation budget. + * Analytic backends deliberately use all OpenMP render workers instead. */ #define FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES ((size_t)512 * 1024 * 1024) static const double pi = 3.14159265358979323846; @@ -290,6 +290,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh, int height, double exposure, const PointSpreadFunction *psf, const PsfKernelCache *psf_cache, + int limit_workers_by_memory, int catalog_load_workers, CatalogPrefetchStats *prefetch_stats, PsfSplatStats *psf_stats, @@ -315,8 +316,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh, mesh->triangle_count); const size_t pixel_count = (size_t)width * height * 3; - if (pixel_count > SIZE_MAX / sizeof(double) || - pixel_count * sizeof(double) > FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES / 2) + if (pixel_count > SIZE_MAX / sizeof(double)) { size_t fallbacks = 0; const size_t images = splat_catalog_triangles(mesh, catalog, hdr, width, height, @@ -324,12 +324,15 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh, mesh->triangle_count, &fallbacks); if (psf_stats != NULL) *psf_stats = (PsfSplatStats){images - fallbacks, fallbacks}; return images; - } + } const size_t buffer_bytes = pixel_count * sizeof(double); - size_t worker_count = FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES / buffer_bytes; const int max_threads = omp_get_max_threads(); - if (worker_count > (size_t)max_threads) - worker_count = (size_t)max_threads; + size_t worker_count = max_threads > 0 ? (size_t)max_threads : 0; + if (limit_workers_by_memory) { + worker_count = FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES / buffer_bytes; + if (worker_count > (size_t)max_threads) + worker_count = (size_t)max_threads; + } if (worker_count < 2 || worker_count > INT_MAX) { size_t fallbacks = 0; diff --git a/src/frame.h b/src/frame.h index c6257d9..d9ce27e 100644 --- a/src/frame.h +++ b/src/frame.h @@ -56,6 +56,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh, int height, double exposure, const PointSpreadFunction *psf, const PsfKernelCache *psf_cache, + int limit_workers_by_memory, int catalog_load_workers, CatalogPrefetchStats *prefetch_stats, PsfSplatStats *psf_stats, diff --git a/src/main.c b/src/main.c index 27b3ece..1df5a91 100644 --- a/src/main.c +++ b/src/main.c @@ -305,7 +305,8 @@ static int render_observer_frame(const Settings *s, StarCatalog *catalog, .all_sky_catalog = catalog->kind == STAR_CATALOG_ALL_SKY}; size_t images = frame_splat_catalog( &mesh, catalog, hdr, s->width, s->height, s->exposure, &s->psf, - &s->psf_cache, s->catalog_load_workers, &prefetch, &psf_stats, + &s->psf_cache, spacetime_limits_render_workers_by_memory(spacetime), + s->catalog_load_workers, &prefetch, &psf_stats, &(FrameSplatProgress){report_splat_progress, &progress}); if (s->draw_mesh) frame_draw_mesh(&mesh, hdr, s->width, s->height, 0.5, 0.5); @@ -434,7 +435,9 @@ static int render_movie(const Settings *s, StarCatalog *catalog, .all_sky_catalog = catalog->kind == STAR_CATALOG_ALL_SKY}; const size_t images = frame_splat_catalog( &movie.frames[i].mesh, catalog, hdr, s->width, s->height, s->exposure, - &s->psf, &s->psf_cache, s->catalog_load_workers, &prefetch, &psf_stats, + &s->psf, &s->psf_cache, + spacetime_limits_render_workers_by_memory(spacetime), + s->catalog_load_workers, &prefetch, &psf_stats, &(FrameSplatProgress){report_splat_progress, &progress}); if (s->draw_mesh) frame_draw_mesh(&movie.frames[i].mesh, hdr, s->width, s->height, 0.5, 0.5); diff --git a/src/spacetime.h b/src/spacetime.h index 30e77a0..7fccebd 100644 --- a/src/spacetime.h +++ b/src/spacetime.h @@ -38,6 +38,10 @@ typedef struct { MetricData *metric); SpacetimeRayStatus (*classify_slab)(const MetricSlab *slab, double t, const double x[3]); + /* Analytic backends have negligible per-ray metric state. A numerical + * backend must opt in once its metric slabs and evaluator workspaces need + * to reserve memory alongside the private HDR render buffers. */ + int limit_render_workers_by_memory; void (*destroy)(SpacetimeSource *source); } SpacetimeOps; @@ -65,5 +69,6 @@ int spacetime_slab_eval(const MetricSlab *slab, double t, const double x[3], MetricData *metric); SpacetimeRayStatus spacetime_slab_classify(const MetricSlab *slab, double t, const double x[3]); +int spacetime_limits_render_workers_by_memory(const SpacetimeSource *source); #endif diff --git a/src/spacetime_common.c b/src/spacetime_common.c index ba625c9..fc606fd 100644 --- a/src/spacetime_common.c +++ b/src/spacetime_common.c @@ -63,3 +63,8 @@ SpacetimeRayStatus spacetime_slab_classify(const MetricSlab *slab, double t, return slab->source->ops->classify_slab(slab, t, x); return spacetime_classify(slab->source, t, x); } + +int spacetime_limits_render_workers_by_memory(const SpacetimeSource *source) { + return source != NULL && source->ops != NULL && + source->ops->limit_render_workers_by_memory; +} diff --git a/tests/test_frame.c b/tests/test_frame.c index b1d8fd7..e63c4e9 100644 --- a/tests/test_frame.c +++ b/tests/test_frame.c @@ -27,7 +27,7 @@ int main(void) { goto done; const size_t images = frame_splat_catalog(&mesh, &catalog, hdr, width, height, test_exposure, - &psf, NULL, 1, NULL, NULL, NULL); + &psf, NULL, 0, 1, NULL, NULL, NULL); if (images != 1 || hdr[3 * (50 * width + 50)] <= 0.0) { fputs("flat-space inverse lens-map regression failed\n", stderr); goto done; @@ -44,10 +44,12 @@ int main(void) { omp_set_dynamic(0); omp_set_num_threads(1); const size_t serial_images = frame_splat_catalog( - &mesh, &catalog, serial_hdr, width, height, test_exposure, &psf, NULL, 1, NULL, NULL, NULL); + &mesh, &catalog, serial_hdr, width, height, test_exposure, &psf, NULL, 0, + 1, NULL, NULL, NULL); omp_set_num_threads(4); const size_t parallel_images = frame_splat_catalog( - &mesh, &catalog, parallel_hdr, width, height, test_exposure, &psf, NULL, 1, NULL, NULL, NULL); + &mesh, &catalog, parallel_hdr, width, height, test_exposure, &psf, NULL, 0, + 1, NULL, NULL, NULL); omp_set_num_threads(original_threads); for (int value = 0; value < width * height * 3; ++value) if (fabs(serial_hdr[value] - parallel_hdr[value]) > @@ -152,7 +154,7 @@ int main(void) { if (frame_lens_mesh_build_coarse(&fine_mesh, width, height, 1, 0.1) || frame_lens_mesh_trace(&fine_mesh, &spacetime, &observer, &trace) || frame_splat_catalog(&fine_mesh, &fine_catalog, hdr, width, height, - test_exposure, &psf, NULL, 1, NULL, NULL, NULL) != 1) { + test_exposure, &psf, NULL, 0, 1, NULL, NULL, NULL) != 1) { fputs("fine source-triangle containment regression failed\n", stderr); frame_lens_mesh_destroy(&fine_mesh); goto done;