Spacetime: free analytic render worker count
This commit is contained in:
1 parent
9b23f3717e
commit
5fd8e3819d
7 files changed
+38
-16
No files matched your search
@@ -84,7 +84,10 @@ chosen from the validation experiments, not inferred from a performance goal.
|
|||||||
For every image event, compute color and final flux exactly as in the existing
|
For every image event, compute color and final flux exactly as in the existing
|
||||||
path. Then multiply the scalar cached PSF weights by that color and flux and
|
path. Then multiply the scalar cached PSF weights by that color and flux and
|
||||||
add them to the worker's existing private HDR buffer. Preserve the current
|
add them to the worker's existing private HDR buffer. Preserve the current
|
||||||
private-HDR ownership, memory cap, reduction ordering, and serial fallback;
|
private-HDR ownership, reduction ordering, and serial fallback. Analytic
|
||||||
|
backends use all OpenMP render workers; a future numerical backend must opt in
|
||||||
|
to the private-HDR memory cap when budgeting its metric slabs and evaluator
|
||||||
|
workspaces;
|
||||||
this plan does not introduce atomics, locks, or a different parallel
|
this plan does not introduce atomics, locks, or a different parallel
|
||||||
partitioning.
|
partitioning.
|
||||||
|
|
||||||
|
|||||||
+12
-9
@@ -8,9 +8,9 @@
|
|||||||
#include <stdint.h>
|
#include <stdint.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
|
|
||||||
/* Private HDR buffers make splatting independent across workers. This is an
|
/* Numerical metric backends may reserve substantial memory for slabs and
|
||||||
* allocation cap, not a rendering parameter: callers transparently fall back
|
* thread-local evaluators, so they retain this private-HDR allocation budget.
|
||||||
* to the serial implementation when the frame is too large for two buffers. */
|
* Analytic backends deliberately use all OpenMP render workers instead. */
|
||||||
#define FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES ((size_t)512 * 1024 * 1024)
|
#define FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES ((size_t)512 * 1024 * 1024)
|
||||||
|
|
||||||
static const double pi = 3.14159265358979323846;
|
static const double pi = 3.14159265358979323846;
|
||||||
@@ -290,6 +290,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
|
|||||||
int height, double exposure,
|
int height, double exposure,
|
||||||
const PointSpreadFunction *psf,
|
const PointSpreadFunction *psf,
|
||||||
const PsfKernelCache *psf_cache,
|
const PsfKernelCache *psf_cache,
|
||||||
|
int limit_workers_by_memory,
|
||||||
int catalog_load_workers,
|
int catalog_load_workers,
|
||||||
CatalogPrefetchStats *prefetch_stats,
|
CatalogPrefetchStats *prefetch_stats,
|
||||||
PsfSplatStats *psf_stats,
|
PsfSplatStats *psf_stats,
|
||||||
@@ -315,8 +316,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
|
|||||||
mesh->triangle_count);
|
mesh->triangle_count);
|
||||||
|
|
||||||
const size_t pixel_count = (size_t)width * height * 3;
|
const size_t pixel_count = (size_t)width * height * 3;
|
||||||
if (pixel_count > SIZE_MAX / sizeof(double) ||
|
if (pixel_count > SIZE_MAX / sizeof(double))
|
||||||
pixel_count * sizeof(double) > FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES / 2)
|
|
||||||
{
|
{
|
||||||
size_t fallbacks = 0;
|
size_t fallbacks = 0;
|
||||||
const size_t images = splat_catalog_triangles(mesh, catalog, hdr, width, height,
|
const size_t images = splat_catalog_triangles(mesh, catalog, hdr, width, height,
|
||||||
@@ -324,12 +324,15 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
|
|||||||
mesh->triangle_count, &fallbacks);
|
mesh->triangle_count, &fallbacks);
|
||||||
if (psf_stats != NULL) *psf_stats = (PsfSplatStats){images - fallbacks, fallbacks};
|
if (psf_stats != NULL) *psf_stats = (PsfSplatStats){images - fallbacks, fallbacks};
|
||||||
return images;
|
return images;
|
||||||
}
|
}
|
||||||
const size_t buffer_bytes = pixel_count * sizeof(double);
|
const size_t buffer_bytes = pixel_count * sizeof(double);
|
||||||
size_t worker_count = FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES / buffer_bytes;
|
|
||||||
const int max_threads = omp_get_max_threads();
|
const int max_threads = omp_get_max_threads();
|
||||||
if (worker_count > (size_t)max_threads)
|
size_t worker_count = max_threads > 0 ? (size_t)max_threads : 0;
|
||||||
worker_count = (size_t)max_threads;
|
if (limit_workers_by_memory) {
|
||||||
|
worker_count = FRAME_SPLAT_MAX_PRIVATE_HDR_BYTES / buffer_bytes;
|
||||||
|
if (worker_count > (size_t)max_threads)
|
||||||
|
worker_count = (size_t)max_threads;
|
||||||
|
}
|
||||||
if (worker_count < 2 || worker_count > INT_MAX)
|
if (worker_count < 2 || worker_count > INT_MAX)
|
||||||
{
|
{
|
||||||
size_t fallbacks = 0;
|
size_t fallbacks = 0;
|
||||||
|
|||||||
@@ -56,6 +56,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
|
|||||||
int height, double exposure,
|
int height, double exposure,
|
||||||
const PointSpreadFunction *psf,
|
const PointSpreadFunction *psf,
|
||||||
const PsfKernelCache *psf_cache,
|
const PsfKernelCache *psf_cache,
|
||||||
|
int limit_workers_by_memory,
|
||||||
int catalog_load_workers,
|
int catalog_load_workers,
|
||||||
CatalogPrefetchStats *prefetch_stats,
|
CatalogPrefetchStats *prefetch_stats,
|
||||||
PsfSplatStats *psf_stats,
|
PsfSplatStats *psf_stats,
|
||||||
|
|||||||
+5
-2
@@ -305,7 +305,8 @@ static int render_observer_frame(const Settings *s, StarCatalog *catalog,
|
|||||||
.all_sky_catalog = catalog->kind == STAR_CATALOG_ALL_SKY};
|
.all_sky_catalog = catalog->kind == STAR_CATALOG_ALL_SKY};
|
||||||
size_t images = frame_splat_catalog(
|
size_t images = frame_splat_catalog(
|
||||||
&mesh, catalog, hdr, s->width, s->height, s->exposure, &s->psf,
|
&mesh, catalog, hdr, s->width, s->height, s->exposure, &s->psf,
|
||||||
&s->psf_cache, s->catalog_load_workers, &prefetch, &psf_stats,
|
&s->psf_cache, spacetime_limits_render_workers_by_memory(spacetime),
|
||||||
|
s->catalog_load_workers, &prefetch, &psf_stats,
|
||||||
&(FrameSplatProgress){report_splat_progress, &progress});
|
&(FrameSplatProgress){report_splat_progress, &progress});
|
||||||
if (s->draw_mesh)
|
if (s->draw_mesh)
|
||||||
frame_draw_mesh(&mesh, hdr, s->width, s->height, 0.5, 0.5);
|
frame_draw_mesh(&mesh, hdr, s->width, s->height, 0.5, 0.5);
|
||||||
@@ -434,7 +435,9 @@ static int render_movie(const Settings *s, StarCatalog *catalog,
|
|||||||
.all_sky_catalog = catalog->kind == STAR_CATALOG_ALL_SKY};
|
.all_sky_catalog = catalog->kind == STAR_CATALOG_ALL_SKY};
|
||||||
const size_t images = frame_splat_catalog(
|
const size_t images = frame_splat_catalog(
|
||||||
&movie.frames[i].mesh, catalog, hdr, s->width, s->height, s->exposure,
|
&movie.frames[i].mesh, catalog, hdr, s->width, s->height, s->exposure,
|
||||||
&s->psf, &s->psf_cache, s->catalog_load_workers, &prefetch, &psf_stats,
|
&s->psf, &s->psf_cache,
|
||||||
|
spacetime_limits_render_workers_by_memory(spacetime),
|
||||||
|
s->catalog_load_workers, &prefetch, &psf_stats,
|
||||||
&(FrameSplatProgress){report_splat_progress, &progress});
|
&(FrameSplatProgress){report_splat_progress, &progress});
|
||||||
if (s->draw_mesh)
|
if (s->draw_mesh)
|
||||||
frame_draw_mesh(&movie.frames[i].mesh, hdr, s->width, s->height, 0.5, 0.5);
|
frame_draw_mesh(&movie.frames[i].mesh, hdr, s->width, s->height, 0.5, 0.5);
|
||||||
|
|||||||
@@ -38,6 +38,10 @@ typedef struct {
|
|||||||
MetricData *metric);
|
MetricData *metric);
|
||||||
SpacetimeRayStatus (*classify_slab)(const MetricSlab *slab, double t,
|
SpacetimeRayStatus (*classify_slab)(const MetricSlab *slab, double t,
|
||||||
const double x[3]);
|
const double x[3]);
|
||||||
|
/* Analytic backends have negligible per-ray metric state. A numerical
|
||||||
|
* backend must opt in once its metric slabs and evaluator workspaces need
|
||||||
|
* to reserve memory alongside the private HDR render buffers. */
|
||||||
|
int limit_render_workers_by_memory;
|
||||||
void (*destroy)(SpacetimeSource *source);
|
void (*destroy)(SpacetimeSource *source);
|
||||||
} SpacetimeOps;
|
} SpacetimeOps;
|
||||||
|
|
||||||
@@ -65,5 +69,6 @@ int spacetime_slab_eval(const MetricSlab *slab, double t, const double x[3],
|
|||||||
MetricData *metric);
|
MetricData *metric);
|
||||||
SpacetimeRayStatus spacetime_slab_classify(const MetricSlab *slab, double t,
|
SpacetimeRayStatus spacetime_slab_classify(const MetricSlab *slab, double t,
|
||||||
const double x[3]);
|
const double x[3]);
|
||||||
|
int spacetime_limits_render_workers_by_memory(const SpacetimeSource *source);
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
@@ -63,3 +63,8 @@ SpacetimeRayStatus spacetime_slab_classify(const MetricSlab *slab, double t,
|
|||||||
return slab->source->ops->classify_slab(slab, t, x);
|
return slab->source->ops->classify_slab(slab, t, x);
|
||||||
return spacetime_classify(slab->source, t, x);
|
return spacetime_classify(slab->source, t, x);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
int spacetime_limits_render_workers_by_memory(const SpacetimeSource *source) {
|
||||||
|
return source != NULL && source->ops != NULL &&
|
||||||
|
source->ops->limit_render_workers_by_memory;
|
||||||
|
}
|
||||||
+6
-4
@@ -27,7 +27,7 @@ int main(void) {
|
|||||||
goto done;
|
goto done;
|
||||||
const size_t images =
|
const size_t images =
|
||||||
frame_splat_catalog(&mesh, &catalog, hdr, width, height, test_exposure,
|
frame_splat_catalog(&mesh, &catalog, hdr, width, height, test_exposure,
|
||||||
&psf, NULL, 1, NULL, NULL, NULL);
|
&psf, NULL, 0, 1, NULL, NULL, NULL);
|
||||||
if (images != 1 || hdr[3 * (50 * width + 50)] <= 0.0) {
|
if (images != 1 || hdr[3 * (50 * width + 50)] <= 0.0) {
|
||||||
fputs("flat-space inverse lens-map regression failed\n", stderr);
|
fputs("flat-space inverse lens-map regression failed\n", stderr);
|
||||||
goto done;
|
goto done;
|
||||||
@@ -44,10 +44,12 @@ int main(void) {
|
|||||||
omp_set_dynamic(0);
|
omp_set_dynamic(0);
|
||||||
omp_set_num_threads(1);
|
omp_set_num_threads(1);
|
||||||
const size_t serial_images = frame_splat_catalog(
|
const size_t serial_images = frame_splat_catalog(
|
||||||
&mesh, &catalog, serial_hdr, width, height, test_exposure, &psf, NULL, 1, NULL, NULL, NULL);
|
&mesh, &catalog, serial_hdr, width, height, test_exposure, &psf, NULL, 0,
|
||||||
|
1, NULL, NULL, NULL);
|
||||||
omp_set_num_threads(4);
|
omp_set_num_threads(4);
|
||||||
const size_t parallel_images = frame_splat_catalog(
|
const size_t parallel_images = frame_splat_catalog(
|
||||||
&mesh, &catalog, parallel_hdr, width, height, test_exposure, &psf, NULL, 1, NULL, NULL, NULL);
|
&mesh, &catalog, parallel_hdr, width, height, test_exposure, &psf, NULL, 0,
|
||||||
|
1, NULL, NULL, NULL);
|
||||||
omp_set_num_threads(original_threads);
|
omp_set_num_threads(original_threads);
|
||||||
for (int value = 0; value < width * height * 3; ++value)
|
for (int value = 0; value < width * height * 3; ++value)
|
||||||
if (fabs(serial_hdr[value] - parallel_hdr[value]) >
|
if (fabs(serial_hdr[value] - parallel_hdr[value]) >
|
||||||
@@ -152,7 +154,7 @@ int main(void) {
|
|||||||
if (frame_lens_mesh_build_coarse(&fine_mesh, width, height, 1, 0.1) ||
|
if (frame_lens_mesh_build_coarse(&fine_mesh, width, height, 1, 0.1) ||
|
||||||
frame_lens_mesh_trace(&fine_mesh, &spacetime, &observer, &trace) ||
|
frame_lens_mesh_trace(&fine_mesh, &spacetime, &observer, &trace) ||
|
||||||
frame_splat_catalog(&fine_mesh, &fine_catalog, hdr, width, height,
|
frame_splat_catalog(&fine_mesh, &fine_catalog, hdr, width, height,
|
||||||
test_exposure, &psf, NULL, 1, NULL, NULL, NULL) != 1) {
|
test_exposure, &psf, NULL, 0, 1, NULL, NULL, NULL) != 1) {
|
||||||
fputs("fine source-triangle containment regression failed\n", stderr);
|
fputs("fine source-triangle containment regression failed\n", stderr);
|
||||||
frame_lens_mesh_destroy(&fine_mesh);
|
frame_lens_mesh_destroy(&fine_mesh);
|
||||||
goto done;
|
goto done;
|
||||||
|
|||||||
Reference in new issue
Block a user