- Gather the union of every movie frame's all-sky tiles once and read them in bounded batches, replacing the per-frame prefetch scan and log. - Fuse the fast supersampled-buffer clear into the FFTW pack pass so a resolved frame starts clean with no serial memset. - Split parallel HDR->RGB8 tone mapping from PNG encoding. - Add a bounded single-producer/single-writer movie output queue used by both observer movies and multi-frame imported lens maps, with writer timing and error propagation. - Add --png-compression-level and --movie-output-workers shared|reserve-one. - Add --fast-fftw-plan estimate|measure|wisdom|wisdom-update with strict wisdom identity sidecars. - Add staged movie timing, regression tests, and docs.
95 lines
4.4 KiB
C
95 lines
4.4 KiB
C
#ifndef FAST_PSF_FFTW_H
|
|
#define FAST_PSF_FFTW_H
|
|
|
|
#include <stddef.h>
|
|
#include <stdio.h>
|
|
|
|
/* Private FFTW-backed global convolution for fast-mode PSF accumulation.
|
|
*
|
|
* This module owns every FFTW allocation and plan. It replaces only the
|
|
* spatial convolution inside the explicit fast-mode preview path; the impulse
|
|
* buffer stays owned by the caller, and the HDR framebuffer is caller-owned and
|
|
* updated with +=. It is compiled only for the CPU PSF backend (FAST_PSF_FFTW)
|
|
* so the HIP and dummy backends do not link FFTW. */
|
|
|
|
typedef struct FastPsfFftwState FastPsfFftwState;
|
|
|
|
typedef struct {
|
|
double zero_pack_seconds;
|
|
double forward_seconds;
|
|
double multiply_seconds;
|
|
double inverse_seconds;
|
|
double crop_downsample_seconds;
|
|
double total_seconds;
|
|
} FastPsfFftwFrameTiming;
|
|
|
|
/* Builds one reusable convolution state from the immutable kernel `weights`
|
|
* (side 2*kernel_radius+1, row-major) and the cached circular `row_span`
|
|
* (length 2*kernel_radius+1, the number of retained dx at each dy). The state
|
|
* computes a zero-padded linear convolution over Wss x Hss and downsamples by
|
|
* `supersample`. `setup_seconds` receives plan creation time and
|
|
* `kernel_seconds` receives the one-time kernel transform time. Returns NULL
|
|
* on any allocation or plan failure; nothing is leaked. */
|
|
FastPsfFftwState *fast_psf_fftw_create(int ss_width, int ss_height,
|
|
int final_width, int final_height,
|
|
int supersample, int kernel_radius,
|
|
const float *weights,
|
|
const int *row_span, int fft_workers,
|
|
double *setup_seconds,
|
|
double *kernel_seconds);
|
|
|
|
/* Convolves `interleaved_ss_rgb` (3 doubles per supersampled pixel, row-major)
|
|
* into `interleaved_hdr_rgb` using +=. The input buffer is not modified.
|
|
* Returns 0 on success and -1 on invalid state. */
|
|
int fast_psf_fftw_resolve(FastPsfFftwState *state,
|
|
const double *interleaved_ss_rgb,
|
|
double *interleaved_hdr_rgb,
|
|
FastPsfFftwFrameTiming *timing);
|
|
|
|
/* Same convolution, but consumes `interleaved_ss_rgb`: each source triplet is
|
|
* copied into the planar scratch and immediately zeroed by the same worker, so
|
|
* the caller gets a clean accumulator back without a separate serial memset.
|
|
* On failure the source is still zeroed so a later frame cannot inherit stale
|
|
* deposits. Returns 0 on success and -1 on invalid state. */
|
|
int fast_psf_fftw_resolve_and_clear(FastPsfFftwState *state,
|
|
double *interleaved_ss_rgb,
|
|
double *interleaved_hdr_rgb,
|
|
FastPsfFftwFrameTiming *timing);
|
|
|
|
/* Safe for NULL and partially initialized states; releases plans before the
|
|
* buffers they reference. Does not touch process-global FFTW state except on
|
|
* the final live state. */
|
|
void fast_psf_fftw_destroy(FastPsfFftwState *state);
|
|
|
|
/* Prints one diagnostic line when `state` is non-NULL. */
|
|
void fast_psf_fftw_report(const FastPsfFftwState *state, FILE *stream);
|
|
|
|
/* First value >= min_extent whose prime factors are limited to {2,3,5,7}.
|
|
* Returns 0 and stores the result in *out on success; returns -1 when the
|
|
* result would exceed INT_MAX or on overflow. */
|
|
int fast_psf_fftw_next_smooth_size(size_t min_extent, size_t *out);
|
|
|
|
/* Selects FFTW_MEASURE (measure != 0) instead of FFTW_ESTIMATE for plans
|
|
* created after this call. Intended for the benchmark's plan-mode comparison;
|
|
* production uses the default FFTW_ESTIMATE. */
|
|
void fast_psf_fftw_set_plan_mode(int measure);
|
|
|
|
/* Plan strategy for fast-mode convolution plans.
|
|
* ESTIMATE - FFTW_ESTIMATE (current default)
|
|
* MEASURE - FFTW_MEASURE
|
|
* WISDOM - import FILE and require FFTW_WISDOM_ONLY; a miss fails
|
|
* WISDOM_UPDATE - FFTW_MEASURE, then atomically export wisdom to FILE */
|
|
typedef enum {
|
|
FAST_PSF_FFTW_PLAN_ESTIMATE = 0,
|
|
FAST_PSF_FFTW_PLAN_MEASURE = 1,
|
|
FAST_PSF_FFTW_PLAN_WISDOM = 2,
|
|
FAST_PSF_FFTW_PLAN_WISDOM_UPDATE = 3
|
|
} FastPsfFftwPlanMode;
|
|
|
|
/* Must be called before fast_psf_fftw_create(). `wisdom_path` may be NULL for
|
|
* the estimate/measure modes. Returns 0 on success and -1 on an immediately
|
|
* detectable configuration error (missing path for a wisdom mode). */
|
|
int fast_psf_fftw_configure(FastPsfFftwPlanMode mode, const char *wisdom_path);
|
|
|
|
#endif
|