Spacetime: free analytic render worker count

This commit is contained in:
wyj committed 2026-08-28 17:20:00 -04:00
1 parent 9b23f3717e
commit 5fd8e3819d
7 files changed
+38 -16

No files matched your search

+4 -1
View File
@@ -84,7 +84,10 @@ chosen from the validation experiments, not inferred from a performance goal.
For every image event, compute color and final flux exactly as in the existing
path. Then multiply the scalar cached PSF weights by that color and flux and
add them to the worker's existing private HDR buffer. Preserve the current
private-HDR ownership, memory cap, reduction ordering, and serial fallback;
private-HDR ownership, reduction ordering, and serial fallback. Analytic
backends use all OpenMP render workers; a future numerical backend must opt in
to the private-HDR memory cap when budgeting its metric slabs and evaluator
workspaces;
this plan does not introduce atomics, locks, or a different parallel
partitioning.
+12 -9
View File
@@ -8,9 +8,9 @@
#include <stdint.h>
#include <stdlib.h>
/* Private HDR buffers make splatting independent across workers. This is an
* allocation cap, not a rendering parameter: callers transparently fall back
* to the serial implementation when the frame is too large for two buffers. */
/* Numerical metric backends may reserve substantial memory for slabs and
* thread-local evaluators, so they retain this private-HDR allocation budget.
* Analytic backends deliberately use all OpenMP render workers instead. */
#define FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES ((size_t)512 * 1024 * 1024)
static const double pi = 3.14159265358979323846;
@@ -290,6 +290,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
int height, double exposure,
const PointSpreadFunction *psf,
const PsfKernelCache *psf_cache,
int limit_workers_by_memory,
int catalog_load_workers,
CatalogPrefetchStats *prefetch_stats,
PsfSplatStats *psf_stats,
@@ -315,8 +316,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
mesh->triangle_count);
const size_t pixel_count = (size_t)width * height * 3;
if (pixel_count > SIZE_MAX / sizeof(double) ||
pixel_count * sizeof(double) > FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES / 2)
if (pixel_count > SIZE_MAX / sizeof(double))
{
size_t fallbacks = 0;
const size_t images = splat_catalog_triangles(mesh, catalog, hdr, width, height,
@@ -324,12 +324,15 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
mesh->triangle_count, &fallbacks);
if (psf_stats != NULL) *psf_stats = (PsfSplatStats){images - fallbacks, fallbacks};
return images;
}
}
const size_t buffer_bytes = pixel_count * sizeof(double);
size_t worker_count = FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES / buffer_bytes;
const int max_threads = omp_get_max_threads();
if (worker_count > (size_t)max_threads)
worker_count = (size_t)max_threads;
size_t worker_count = max_threads > 0 ? (size_t)max_threads : 0;
if (limit_workers_by_memory) {
worker_count = FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES / buffer_bytes;
if (worker_count > (size_t)max_threads)
worker_count = (size_t)max_threads;
}
if (worker_count < 2 || worker_count > INT_MAX)
{
size_t fallbacks = 0;
+1
View File
@@ -56,6 +56,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
int height, double exposure,
const PointSpreadFunction *psf,
const PsfKernelCache *psf_cache,
int limit_workers_by_memory,
int catalog_load_workers,
CatalogPrefetchStats *prefetch_stats,
PsfSplatStats *psf_stats,
+5 -2
View File
@@ -305,7 +305,8 @@ static int render_observer_frame(const Settings *s, StarCatalog *catalog,
.all_sky_catalog = catalog->kind == STAR_CATALOG_ALL_SKY};
size_t images = frame_splat_catalog(
&mesh, catalog, hdr, s->width, s->height, s->exposure, &s->psf,
&s->psf_cache, s->catalog_load_workers, &prefetch, &psf_stats,
&s->psf_cache, spacetime_limits_render_workers_by_memory(spacetime),
s->catalog_load_workers, &prefetch, &psf_stats,
&(FrameSplatProgress){report_splat_progress, &progress});
if (s->draw_mesh)
frame_draw_mesh(&mesh, hdr, s->width, s->height, 0.5, 0.5);
@@ -434,7 +435,9 @@ static int render_movie(const Settings *s, StarCatalog *catalog,
.all_sky_catalog = catalog->kind == STAR_CATALOG_ALL_SKY};
const size_t images = frame_splat_catalog(
&movie.frames[i].mesh, catalog, hdr, s->width, s->height, s->exposure,
&s->psf, &s->psf_cache, s->catalog_load_workers, &prefetch, &psf_stats,
&s->psf, &s->psf_cache,
spacetime_limits_render_workers_by_memory(spacetime),
s->catalog_load_workers, &prefetch, &psf_stats,
&(FrameSplatProgress){report_splat_progress, &progress});
if (s->draw_mesh)
frame_draw_mesh(&movie.frames[i].mesh, hdr, s->width, s->height, 0.5, 0.5);
+5
View File
@@ -38,6 +38,10 @@ typedef struct {
MetricData *metric);
SpacetimeRayStatus (*classify_slab)(const MetricSlab *slab, double t,
const double x[3]);
/* Analytic backends have negligible per-ray metric state. A numerical
* backend must opt in once its metric slabs and evaluator workspaces need
* to reserve memory alongside the private HDR render buffers. */
int limit_render_workers_by_memory;
void (*destroy)(SpacetimeSource *source);
} SpacetimeOps;
@@ -65,5 +69,6 @@ int spacetime_slab_eval(const MetricSlab *slab, double t, const double x[3],
MetricData *metric);
SpacetimeRayStatus spacetime_slab_classify(const MetricSlab *slab, double t,
const double x[3]);
int spacetime_limits_render_workers_by_memory(const SpacetimeSource *source);
#endif
+5
View File
@@ -63,3 +63,8 @@ SpacetimeRayStatus spacetime_slab_classify(const MetricSlab *slab, double t,
return slab->source->ops->classify_slab(slab, t, x);
return spacetime_classify(slab->source, t, x);
}
int spacetime_limits_render_workers_by_memory(const SpacetimeSource *source) {
return source != NULL && source->ops != NULL &&
source->ops->limit_render_workers_by_memory;
}
+6 -4
View File
@@ -27,7 +27,7 @@ int main(void) {
goto done;
const size_t images =
frame_splat_catalog(&mesh, &catalog, hdr, width, height, test_exposure,
&psf, NULL, 1, NULL, NULL, NULL);
&psf, NULL, 0, 1, NULL, NULL, NULL);
if (images != 1 || hdr[3 * (50 * width + 50)] <= 0.0) {
fputs("flat-space inverse lens-map regression failed\n", stderr);
goto done;
@@ -44,10 +44,12 @@ int main(void) {
omp_set_dynamic(0);
omp_set_num_threads(1);
const size_t serial_images = frame_splat_catalog(
&mesh, &catalog, serial_hdr, width, height, test_exposure, &psf, NULL, 1, NULL, NULL, NULL);
&mesh, &catalog, serial_hdr, width, height, test_exposure, &psf, NULL, 0,
1, NULL, NULL, NULL);
omp_set_num_threads(4);
const size_t parallel_images = frame_splat_catalog(
&mesh, &catalog, parallel_hdr, width, height, test_exposure, &psf, NULL, 1, NULL, NULL, NULL);
&mesh, &catalog, parallel_hdr, width, height, test_exposure, &psf, NULL, 0,
1, NULL, NULL, NULL);
omp_set_num_threads(original_threads);
for (int value = 0; value < width * height * 3; ++value)
if (fabs(serial_hdr[value] - parallel_hdr[value]) >
@@ -152,7 +154,7 @@ int main(void) {
if (frame_lens_mesh_build_coarse(&fine_mesh, width, height, 1, 0.1) ||
frame_lens_mesh_trace(&fine_mesh, &spacetime, &observer, &trace) ||
frame_splat_catalog(&fine_mesh, &fine_catalog, hdr, width, height,
test_exposure, &psf, NULL, 1, NULL, NULL, NULL) != 1) {
test_exposure, &psf, NULL, 0, 1, NULL, NULL, NULL) != 1) {
fputs("fine source-triangle containment regression failed\n", stderr);
frame_lens_mesh_destroy(&fine_mesh);
goto done;