Feat: Add --fast-mode supersampled point-source accumulation

Add an optional CPU preview path that deposits each point-source image as a
supersampled delta and resolves the whole frame with one global Moffat
convolution plus an N x N box average, instead of splatting a per-event PSF.

- optics: FastPsfAccumulator builds the pixel-area-integral kernel of the
  target Moffat at the supersampled scale (width N*alpha, same beta).  The
  1/N^2 box average then reproduces the final pixel-area integral, so the
  requested FWHM and beta are preserved without renormalisation.  Deposits are
  per-cell atomic adds; resolve accumulates into the caller's HDR buffer.
- frame: fast branch in frame_splat_catalog with one shared supersampled
  buffer and a single resolve per frame; the accumulator is reused across
  movie frames and built from the map dimensions on lens-map import.
- main: --fast-mode, --fast-supersample N (1..8, default 2) and
  --fast-deposit nearest|bilinear (default nearest).  CPU-only and rejected in
  the HIP/dummy backends; --psf-min-y still applies per event while
  --max-cache-psf-flux does not.
- The deposition scheme was chosen by scripts/fast_mode_deposit_error.py:
  nearest keeps the PSF shape exactly with <= 0.5/N px position quantization;
  bilinear keeps the exact centroid but broadens FWHM and beta.  Recorded in
  benchmarks/fast_mode_deposit_2026-09-18.md.
- tests/test_frame.c covers fast nearest vs the direct evaluator at the snapped
  centre, flux conservation, bilinear centroid, min-Y discard, frame plumbing,
  and HDR accumulation onto a non-zero background.
- benchmarks/fast_mode_cpu_2026-09-18.md records a ~10x speedup on the 2MASS
  galactic-centre field with small tone-mapped differences.
This commit is contained in:
wyj committed 2026-09-24 01:36:25 -04:00
1 parent 7ddc58515a
commit 3deebfb2fa
13 files changed
+1249 -36

No files matched your search

+80 -11
View File
@@ -1110,6 +1110,7 @@ typedef struct {
double psf_relative_tail;
double psf_min_y;
PsfEventSink *event_sink;
FastPsfAccumulator *fast;
size_t images;
size_t direct_fallbacks;
size_t cached_wing_clipped;
@@ -1187,6 +1188,14 @@ static int splat_catalog_tile(const Star *stars, size_t count,
weights[2] * context->vertex[2]->log_frequency_ratio;
const LinearRgb color = blackbody_to_linear_rgb(star->temperature_K * exp(log_g));
const double flux = context->exposure * star->amplitude * context->magnification;
if (context->fast != NULL) {
const int fast_status = fast_psf_accumulator_deposit(
context->fast, image_x, image_y, color, flux);
context->cached_wing_clipped += fast_status == 2;
context->discarded_below_min_y += fast_status == 3;
++context->images;
continue;
}
PsfCachedEvent event;
const int direct_fallback = psf_prepare_cached_event(
&event, image_x, image_y, color, flux, context->psf, context->psf_cache,
@@ -1272,7 +1281,8 @@ static CatalogSplatStats splat_catalog_triangles(
const PsfKernelCache *psf_cache, double max_magnification,
double max_cache_psf_flux, double psf_relative_tail,
double psf_min_y,
size_t first_triangle, size_t last_triangle, PsfEventSink *event_sink) {
size_t first_triangle, size_t last_triangle, PsfEventSink *event_sink,
FastPsfAccumulator *fast) {
CatalogSplatStats stats = {0};
PsfEventSink owned_sink;
const int owns_sink = event_sink == NULL;
@@ -1321,7 +1331,8 @@ static CatalogSplatStats splat_catalog_triangles(
.max_cache_psf_flux = max_cache_psf_flux,
.psf_relative_tail = psf_relative_tail,
.psf_min_y = psf_min_y,
.event_sink = event_sink};
.event_sink = event_sink,
.fast = fast};
const int visit_result = catalog_visit_source_triangle(
catalog, direction, 0, splat_catalog_tile, &context);
if (visit_result == 0)
@@ -1383,7 +1394,8 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
int catalog_load_workers,
CatalogPrefetchStats *prefetch_stats,
PsfSplatStats *psf_stats,
const FrameSplatProgress *progress) {
const FrameSplatProgress *progress,
FastPsfAccumulator *fast) {
if (mesh == NULL || catalog == NULL || hdr == NULL || exposure <= 0.0 ||
psf == NULL || width <= 0 || height <= 0 || catalog_load_workers <= 0 ||
isnan(max_magnification) || max_magnification <= 0.0 ||
@@ -1408,6 +1420,63 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
progress->callback(progress->context, FRAME_SPLAT_PROGRESS_BEGIN, 0,
mesh->triangle_count);
/* Fast mode replaces the per-event PSF splat with cheap delta deposits into
* one shared supersampled buffer. A single global convolution plus an N x N
* box average then resolves the frame. Deposits are per-cell atomic adds;
* the immutable kernel and input buffers stay shared read-only. */
if (fast != NULL) {
size_t images = 0, direct_fallbacks = 0, cached_wing_clipped = 0,
discarded_below_min_y = 0;
int failed = 0;
int worker_count = omp_get_max_threads();
if (worker_count < 1)
worker_count = 1;
fast_psf_accumulator_clear(fast);
#ifdef GR_DEBUG
double max_raw_magnification = 0.0;
size_t magnification_clamped_triangles = 0;
#pragma omp parallel num_threads(worker_count) reduction(+ : images, direct_fallbacks, cached_wing_clipped, discarded_below_min_y, magnification_clamped_triangles) reduction(max : max_raw_magnification, failed)
#else
#pragma omp parallel num_threads(worker_count) reduction(+ : images, direct_fallbacks, cached_wing_clipped, discarded_below_min_y) reduction(max : failed)
#endif
{
#pragma omp for schedule(dynamic, 1)
for (size_t t = 0; t < mesh->triangle_count; ++t) {
const CatalogSplatStats stats = splat_catalog_triangles(
mesh, catalog, hdr, width, height, exposure, psf, NULL,
max_magnification, max_cache_psf_flux, psf_relative_tail, psf_min_y,
t, t + 1, NULL, fast);
images += stats.images;
direct_fallbacks += stats.direct_fallbacks;
cached_wing_clipped += stats.cached_wing_clipped;
discarded_below_min_y += stats.discarded_below_min_y;
failed |= stats.failed;
#ifdef GR_DEBUG
max_raw_magnification =
fmax(max_raw_magnification, stats.max_raw_magnification);
magnification_clamped_triangles += stats.magnification_clamped_triangles;
#endif
}
}
if (!failed && fast_psf_accumulator_resolve(fast, hdr, worker_count))
failed = 1;
copy_psf_splat_stats(psf_stats, (CatalogSplatStats){
.images = images,
.direct_fallbacks = direct_fallbacks,
.cached_wing_clipped = cached_wing_clipped,
.discarded_below_min_y = discarded_below_min_y,
#ifdef GR_DEBUG
.max_raw_magnification = max_raw_magnification,
.magnification_clamped_triangles =
magnification_clamped_triangles,
#endif
});
if (progress != NULL && progress->callback != NULL)
progress->callback(progress->context, FRAME_SPLAT_PROGRESS_END,
mesh->triangle_count, mesh->triangle_count);
return failed ? SIZE_MAX : images;
}
#ifdef PSF_BACKEND_DUMMY
PsfEventSink owner;
if (psf_event_sink_init(&owner, hdr, width, height, psf_cache)) return SIZE_MAX;
@@ -1442,7 +1511,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
const CatalogSplatStats stats = splat_catalog_triangles(
mesh, catalog, hdr, width, height, exposure, psf, psf_cache,
max_magnification, max_cache_psf_flux, psf_relative_tail, psf_min_y,
t, t + 1, &local);
t, t + 1, &local, NULL);
dummy_images += stats.images;
dummy_direct += stats.direct_fallbacks;
dummy_clipped += stats.cached_wing_clipped;
@@ -1531,7 +1600,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
const CatalogSplatStats stats = splat_catalog_triangles(
mesh, catalog, hdr, width, height, exposure, psf, psf_cache,
max_magnification, max_cache_psf_flux, psf_relative_tail, psf_min_y,
t, t + 1, &local);
t, t + 1, &local, NULL);
hip_images += stats.images;
hip_direct += stats.direct_fallbacks;
hip_clipped += stats.cached_wing_clipped;
@@ -1594,7 +1663,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
const CatalogSplatStats stats = splat_catalog_triangles(
mesh, catalog, hdr, width, height, exposure, psf, psf_cache,
max_magnification, max_cache_psf_flux, psf_relative_tail, psf_min_y,
0, mesh->triangle_count, NULL);
0, mesh->triangle_count, NULL, NULL);
copy_psf_splat_stats(psf_stats, stats);
return stats.images;
}
@@ -1611,7 +1680,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
const CatalogSplatStats stats = splat_catalog_triangles(
mesh, catalog, hdr, width, height, exposure, psf, psf_cache,
max_magnification, max_cache_psf_flux, psf_relative_tail, psf_min_y,
0, mesh->triangle_count, NULL);
0, mesh->triangle_count, NULL, NULL);
copy_psf_splat_stats(psf_stats, stats);
return stats.images;
}
@@ -1622,7 +1691,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
const CatalogSplatStats stats = splat_catalog_triangles(
mesh, catalog, hdr, width, height, exposure, psf, psf_cache,
max_magnification, max_cache_psf_flux, psf_relative_tail, psf_min_y,
0, mesh->triangle_count, NULL);
0, mesh->triangle_count, NULL, NULL);
copy_psf_splat_stats(psf_stats, stats);
return stats.images;
}
@@ -1639,7 +1708,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
const CatalogSplatStats stats = splat_catalog_triangles(
mesh, catalog, hdr, width, height, exposure, psf, psf_cache,
max_magnification, max_cache_psf_flux, psf_relative_tail, psf_min_y,
0, mesh->triangle_count, NULL);
0, mesh->triangle_count, NULL, NULL);
copy_psf_splat_stats(psf_stats, stats);
return stats.images;
}
@@ -1673,7 +1742,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
mesh, catalog, private_hdr[worker], width, height, exposure, psf,
psf_cache, max_magnification, max_cache_psf_flux, psf_relative_tail,
psf_min_y,
triangle, triangle + 1, &event_sink);
triangle, triangle + 1, &event_sink, NULL);
images += stats.images;
direct_fallbacks += stats.direct_fallbacks;
cached_wing_clipped += stats.cached_wing_clipped;
@@ -1710,7 +1779,7 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
mesh, catalog, private_hdr[worker], width, height, exposure, psf,
psf_cache, max_magnification, max_cache_psf_flux, psf_relative_tail,
psf_min_y,
triangle, triangle + 1, &event_sink);
triangle, triangle + 1, &event_sink, NULL);
images += stats.images;
direct_fallbacks += stats.direct_fallbacks;
cached_wing_clipped += stats.cached_wing_clipped;