Spacetime: free analytic render worker count

This commit is contained in:
wyj committed 2026-08-28 17:20:00 -04:00
1 parent 9b23f3717e
commit 5fd8e3819d
7 files changed
+35 -13

No files matched your search

+4 -1
View File
@@ -84,7 +84,10 @@ chosen from the validation experiments, not inferred from a performance goal.
For every image event, compute color and final flux exactly as in the existing For every image event, compute color and final flux exactly as in the existing
path. Then multiply the scalar cached PSF weights by that color and flux and path. Then multiply the scalar cached PSF weights by that color and flux and
add them to the worker's existing private HDR buffer. Preserve the current add them to the worker's existing private HDR buffer. Preserve the current
private-HDR ownership, memory cap, reduction ordering, and serial fallback; private-HDR ownership, reduction ordering, and serial fallback. Analytic
backends use all OpenMP render workers; a future numerical backend must opt in
to the private-HDR memory cap when budgeting its metric slabs and evaluator
workspaces;
this plan does not introduce atomics, locks, or a different parallel this plan does not introduce atomics, locks, or a different parallel
partitioning. partitioning.
+9 -6
View File
@@ -8,9 +8,9 @@
#include <stdint.h> #include <stdint.h>
#include <stdlib.h> #include <stdlib.h>
/* Private HDR buffers make splatting independent across workers. This is an /* Numerical metric backends may reserve substantial memory for slabs and
* allocation cap, not a rendering parameter: callers transparently fall back * thread-local evaluators, so they retain this private-HDR allocation budget.
* to the serial implementation when the frame is too large for two buffers. */ * Analytic backends deliberately use all OpenMP render workers instead. */
#define FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES ((size_t)512 * 1024 * 1024) #define FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES ((size_t)512 * 1024 * 1024)
static const double pi = 3.14159265358979323846; static const double pi = 3.14159265358979323846;
@@ -290,6 +290,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
int height, double exposure, int height, double exposure,
const PointSpreadFunction *psf, const PointSpreadFunction *psf,
const PsfKernelCache *psf_cache, const PsfKernelCache *psf_cache,
int limit_workers_by_memory,
int catalog_load_workers, int catalog_load_workers,
CatalogPrefetchStats *prefetch_stats, CatalogPrefetchStats *prefetch_stats,
PsfSplatStats *psf_stats, PsfSplatStats *psf_stats,
@@ -315,8 +316,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
mesh->triangle_count); mesh->triangle_count);
const size_t pixel_count = (size_t)width * height * 3; const size_t pixel_count = (size_t)width * height * 3;
if (pixel_count > SIZE_MAX / sizeof(double) || if (pixel_count > SIZE_MAX / sizeof(double))
pixel_count * sizeof(double) > FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES / 2)
{ {
size_t fallbacks = 0; size_t fallbacks = 0;
const size_t images = splat_catalog_triangles(mesh, catalog, hdr, width, height, const size_t images = splat_catalog_triangles(mesh, catalog, hdr, width, height,
@@ -326,10 +326,13 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
return images; return images;
} }
const size_t buffer_bytes = pixel_count * sizeof(double); const size_t buffer_bytes = pixel_count * sizeof(double);
size_t worker_count = FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES / buffer_bytes;
const int max_threads = omp_get_max_threads(); const int max_threads = omp_get_max_threads();
size_t worker_count = max_threads > 0 ? (size_t)max_threads : 0;
if (limit_workers_by_memory) {
worker_count = FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES / buffer_bytes;
if (worker_count > (size_t)max_threads) if (worker_count > (size_t)max_threads)
worker_count = (size_t)max_threads; worker_count = (size_t)max_threads;
}
if (worker_count < 2 || worker_count > INT_MAX) if (worker_count < 2 || worker_count > INT_MAX)
{ {
size_t fallbacks = 0; size_t fallbacks = 0;
+1
View File
@@ -56,6 +56,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
int height, double exposure, int height, double exposure,
const PointSpreadFunction *psf, const PointSpreadFunction *psf,
const PsfKernelCache *psf_cache, const PsfKernelCache *psf_cache,
int limit_workers_by_memory,
int catalog_load_workers, int catalog_load_workers,
CatalogPrefetchStats *prefetch_stats, CatalogPrefetchStats *prefetch_stats,
PsfSplatStats *psf_stats, PsfSplatStats *psf_stats,
+5 -2
View File
@@ -305,7 +305,8 @@ static int render_observer_frame(const Settings *s, StarCatalog *catalog,
.all_sky_catalog = catalog->kind == STAR_CATALOG_ALL_SKY}; .all_sky_catalog = catalog->kind == STAR_CATALOG_ALL_SKY};
size_t images = frame_splat_catalog( size_t images = frame_splat_catalog(
&mesh, catalog, hdr, s->width, s->height, s->exposure, &s->psf, &mesh, catalog, hdr, s->width, s->height, s->exposure, &s->psf,
&s->psf_cache, s->catalog_load_workers, &prefetch, &psf_stats, &s->psf_cache, spacetime_limits_render_workers_by_memory(spacetime),
s->catalog_load_workers, &prefetch, &psf_stats,
&(FrameSplatProgress){report_splat_progress, &progress}); &(FrameSplatProgress){report_splat_progress, &progress});
if (s->draw_mesh) if (s->draw_mesh)
frame_draw_mesh(&mesh, hdr, s->width, s->height, 0.5, 0.5); frame_draw_mesh(&mesh, hdr, s->width, s->height, 0.5, 0.5);
@@ -434,7 +435,9 @@ static int render_movie(const Settings *s, StarCatalog *catalog,
.all_sky_catalog = catalog->kind == STAR_CATALOG_ALL_SKY}; .all_sky_catalog = catalog->kind == STAR_CATALOG_ALL_SKY};
const size_t images = frame_splat_catalog( const size_t images = frame_splat_catalog(
&movie.frames[i].mesh, catalog, hdr, s->width, s->height, s->exposure, &movie.frames[i].mesh, catalog, hdr, s->width, s->height, s->exposure,
&s->psf, &s->psf_cache, s->catalog_load_workers, &prefetch, &psf_stats, &s->psf, &s->psf_cache,
spacetime_limits_render_workers_by_memory(spacetime),
s->catalog_load_workers, &prefetch, &psf_stats,
&(FrameSplatProgress){report_splat_progress, &progress}); &(FrameSplatProgress){report_splat_progress, &progress});
if (s->draw_mesh) if (s->draw_mesh)
frame_draw_mesh(&movie.frames[i].mesh, hdr, s->width, s->height, 0.5, 0.5); frame_draw_mesh(&movie.frames[i].mesh, hdr, s->width, s->height, 0.5, 0.5);
+5
View File
@@ -38,6 +38,10 @@ typedef struct {
MetricData *metric); MetricData *metric);
SpacetimeRayStatus (*classify_slab)(const MetricSlab *slab, double t, SpacetimeRayStatus (*classify_slab)(const MetricSlab *slab, double t,
const double x[3]); const double x[3]);
/* Analytic backends have negligible per-ray metric state. A numerical
* backend must opt in once its metric slabs and evaluator workspaces need
* to reserve memory alongside the private HDR render buffers. */
int limit_render_workers_by_memory;
void (*destroy)(SpacetimeSource *source); void (*destroy)(SpacetimeSource *source);
} SpacetimeOps; } SpacetimeOps;
@@ -65,5 +69,6 @@ int spacetime_slab_eval(const MetricSlab *slab, double t, const double x[3],
MetricData *metric); MetricData *metric);
SpacetimeRayStatus spacetime_slab_classify(const MetricSlab *slab, double t, SpacetimeRayStatus spacetime_slab_classify(const MetricSlab *slab, double t,
const double x[3]); const double x[3]);
int spacetime_limits_render_workers_by_memory(const SpacetimeSource *source);
#endif #endif
+5
View File
@@ -63,3 +63,8 @@ SpacetimeRayStatus spacetime_slab_classify(const MetricSlab *slab, double t,
return slab->source->ops->classify_slab(slab, t, x); return slab->source->ops->classify_slab(slab, t, x);
return spacetime_classify(slab->source, t, x); return spacetime_classify(slab->source, t, x);
} }
int spacetime_limits_render_workers_by_memory(const SpacetimeSource *source) {
return source != NULL && source->ops != NULL &&
source->ops->limit_render_workers_by_memory;
}
+6 -4
View File
@@ -27,7 +27,7 @@ int main(void) {
goto done; goto done;
const size_t images = const size_t images =
frame_splat_catalog(&mesh, &catalog, hdr, width, height, test_exposure, frame_splat_catalog(&mesh, &catalog, hdr, width, height, test_exposure,
&psf, NULL, 1, NULL, NULL, NULL); &psf, NULL, 0, 1, NULL, NULL, NULL);
if (images != 1 || hdr[3 * (50 * width + 50)] <= 0.0) { if (images != 1 || hdr[3 * (50 * width + 50)] <= 0.0) {
fputs("flat-space inverse lens-map regression failed\n", stderr); fputs("flat-space inverse lens-map regression failed\n", stderr);
goto done; goto done;
@@ -44,10 +44,12 @@ int main(void) {
omp_set_dynamic(0); omp_set_dynamic(0);
omp_set_num_threads(1); omp_set_num_threads(1);
const size_t serial_images = frame_splat_catalog( const size_t serial_images = frame_splat_catalog(
&mesh, &catalog, serial_hdr, width, height, test_exposure, &psf, NULL, 1, NULL, NULL, NULL); &mesh, &catalog, serial_hdr, width, height, test_exposure, &psf, NULL, 0,
1, NULL, NULL, NULL);
omp_set_num_threads(4); omp_set_num_threads(4);
const size_t parallel_images = frame_splat_catalog( const size_t parallel_images = frame_splat_catalog(
&mesh, &catalog, parallel_hdr, width, height, test_exposure, &psf, NULL, 1, NULL, NULL, NULL); &mesh, &catalog, parallel_hdr, width, height, test_exposure, &psf, NULL, 0,
1, NULL, NULL, NULL);
omp_set_num_threads(original_threads); omp_set_num_threads(original_threads);
for (int value = 0; value < width * height * 3; ++value) for (int value = 0; value < width * height * 3; ++value)
if (fabs(serial_hdr[value] - parallel_hdr[value]) > if (fabs(serial_hdr[value] - parallel_hdr[value]) >
@@ -152,7 +154,7 @@ int main(void) {
if (frame_lens_mesh_build_coarse(&fine_mesh, width, height, 1, 0.1) || if (frame_lens_mesh_build_coarse(&fine_mesh, width, height, 1, 0.1) ||
frame_lens_mesh_trace(&fine_mesh, &spacetime, &observer, &trace) || frame_lens_mesh_trace(&fine_mesh, &spacetime, &observer, &trace) ||
frame_splat_catalog(&fine_mesh, &fine_catalog, hdr, width, height, frame_splat_catalog(&fine_mesh, &fine_catalog, hdr, width, height,
test_exposure, &psf, NULL, 1, NULL, NULL, NULL) != 1) { test_exposure, &psf, NULL, 0, 1, NULL, NULL, NULL) != 1) {
fputs("fine source-triangle containment regression failed\n", stderr); fputs("fine source-triangle containment regression failed\n", stderr);
frame_lens_mesh_destroy(&fine_mesh); frame_lens_mesh_destroy(&fine_mesh);
goto done; goto done;