Feat: Use FFTW linear convolution for CPU fast-mode PSF resolve

Replace the nested spatial global convolution in fast_psf_accumulator_resolve with a reusable double-precision FFTW linear convolution on the CPU PSF backend.

- Add the private src/fast_psf_fftw.{c,h} module: zero-padded R2C/C2R plans, cached kernel spectrum and planar scratch, exact 1/(Pwidth*Pheight) and 1/N^2 normalization, (R,R) crop, and additive HDR output.
- Keep the previous nested loops as fast_psf_accumulator_resolve_spatial_reference for tests/benchmarks only; it is not a runtime fallback.
- Cache the circular row spans on FastPsfAccumulator and report one-time plan, kernel transform, scratch, and per-frame stage timings.
- Require fftw3_omp for CPU builds; HIP and dummy builds do not link FFTW.
- Namespace test/helper binaries by spacetime and build tag, and reject make test / psf-capture for non-CPU backends.
- Add tests/test_fast_psf_fftw.c (FFTW versus spatial), tests/benchmark_fast_psf_fftw.c, an FFTW CLI smoke check, and the 2026-09-25 benchmark record.
This commit is contained in:
wyj committed 2026-09-25 23:35:45 -04:00
1 parent 3deebfb2fa
commit 229f50cd86
14 files changed
+2372 -55

No files matched your search

+223
View File
@@ -0,0 +1,223 @@
/* Resolver-only benchmark for the fast-mode FFTW global convolution.
*
* Fills one deterministic impulse buffer, then times the spatial reference and
* the FFTW production resolver on that identical buffer. First-use plan/setup
* is reported separately from steady-state execution. No catalog or geodesic
* work is involved. */
#ifndef FAST_PSF_FFTW
#error "benchmark_fast_psf_fftw requires -DFAST_PSF_FFTW (PSF_BACKEND=cpu)"
#endif
#include "fast_psf_fftw.h"
#include "optics.h"
#include <math.h>
#include <omp.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/resource.h>
typedef struct {
int width, height, supersample, repeats;
int run_spatial;
int measure;
FastPsfDeposit deposit;
} Options;
static int parse_int(const char *text, int *out)
{
char *end = NULL;
const long value = strtol(text, &end, 10);
if (end == text || *end != '\0' || value <= 0 || value > 1000000)
return -1;
*out = (int)value;
return 0;
}
static int parse_options(int argc, char **argv, Options *options)
{
*options = (Options){.width = 320,
.height = 240,
.supersample = 2,
.repeats = 5,
.run_spatial = 0,
.measure = 0,
.deposit = FAST_PSF_DEPOSIT_NEAREST};
for (int i = 1; i < argc; ++i) {
if (!strcmp(argv[i], "--spatial")) {
options->run_spatial = 1;
} else if (!strcmp(argv[i], "--measure")) {
options->measure = 1;
} else if (!strcmp(argv[i], "--bilinear")) {
options->deposit = FAST_PSF_DEPOSIT_BILINEAR;
} else if (!strcmp(argv[i], "--width") && i + 1 < argc) {
if (parse_int(argv[++i], &options->width))
return -1;
} else if (!strcmp(argv[i], "--height") && i + 1 < argc) {
if (parse_int(argv[++i], &options->height))
return -1;
} else if (!strcmp(argv[i], "--supersample") && i + 1 < argc) {
if (parse_int(argv[++i], &options->supersample))
return -1;
} else if (!strcmp(argv[i], "--repeats") && i + 1 < argc) {
if (parse_int(argv[++i], &options->repeats))
return -1;
} else {
fprintf(stderr, "unknown argument '%s'\n", argv[i]);
return -1;
}
}
return 0;
}
static uint64_t hash_bytes(const double *values, size_t count)
{
const unsigned char *bytes = (const unsigned char *)values;
const size_t nbytes = count * sizeof *values;
uint64_t hash = 1469598103934665603ULL;
for (size_t i = 0; i < nbytes; ++i) {
hash ^= bytes[i];
hash *= 1099511628211ULL;
}
return hash;
}
static double wall_seconds(void)
{
return omp_get_wtime();
}
static int compare_double(const void *a, const void *b)
{
const double x = *(const double *)a;
const double y = *(const double *)b;
return x < y ? -1 : x > y ? 1 : 0;
}
int main(int argc, char **argv)
{
Options options;
if (parse_options(argc, argv, &options)) {
fprintf(stderr,
"usage: %s [--width W] [--height H] [--supersample N] "
"[--repeats R] [--bilinear] [--spatial] [--measure]\n",
argv[0]);
return 2;
}
const PointSpreadFunction psf = {.fwhm_pixels = 2.7, .moffat_beta = 4.5};
const double relative_tail = 1e-8;
if (options.repeats > 128)
options.repeats = 128;
printf("impulse: %dx%d supersample=%d deposit=%s repeats=%d spatial=%d\n",
options.width, options.height, options.supersample,
options.deposit == FAST_PSF_DEPOSIT_NEAREST ? "nearest" : "bilinear",
options.repeats, options.run_spatial);
printf("plan_mode=%s\n", options.measure ? "measure" : "estimate");
fast_psf_fftw_set_plan_mode(options.measure);
FastPsfAccumulator acc = {0};
const double init_start = wall_seconds();
if (fast_psf_accumulator_init(&acc, options.width, options.height,
options.supersample, options.deposit, &psf,
relative_tail, 0.0, 1)) {
fputs("accumulator initialization failed\n", stderr);
return 1;
}
const double init_seconds = wall_seconds() - init_start;
printf("setup: init=%.6f s fftw_plan=%.6f s kernel_fft=%.6f s\n",
init_seconds, acc.fftw_setup_seconds, acc.fftw_kernel_seconds);
fast_psf_fftw_report(acc.fftw, stdout);
const size_t ss_count = (size_t)acc.supersampled_width *
acc.supersampled_height;
uint64_t state = 0x9e3779b97f4a7c15ULL;
for (size_t i = 0; i < ss_count; ++i) {
state = state * 6364136223846793005ULL + 1442695040888963407ULL;
if ((state >> 58) != 0)
continue;
state = state * 6364136223846793005ULL + 1442695040888963407ULL;
acc.buffer[3 * i] = (double)(state >> 40) / (double)(1ULL << 24);
state = state * 6364136223846793005ULL + 1442695040888963407ULL;
acc.buffer[3 * i + 1] = (double)(state >> 40) / (double)(1ULL << 24);
state = state * 6364136223846793005ULL + 1442695040888963407ULL;
acc.buffer[3 * i + 2] = (double)(state >> 40) / (double)(1ULL << 24);
}
const size_t hdr_count = (size_t)options.width * options.height * 3;
const size_t buffer_count = ss_count * 3;
const uint64_t impulse_hash = hash_bytes(acc.buffer, buffer_count);
printf("impulse_hash=%016llx\n", (unsigned long long)impulse_hash);
double *fftw_hdr = calloc(hdr_count, sizeof *fftw_hdr);
double *spatial_hdr = calloc(hdr_count, sizeof *spatial_hdr);
if (fftw_hdr == NULL || spatial_hdr == NULL) {
fputs("HDR allocation failed\n", stderr);
free(fftw_hdr);
free(spatial_hdr);
fast_psf_accumulator_destroy(&acc);
return 1;
}
/* Warm-up: first execution includes any lazy per-frame allocation. */
if (fast_psf_accumulator_resolve(&acc, fftw_hdr, omp_get_max_threads())) {
fputs("FFTW warm-up resolve failed\n", stderr);
return 1;
}
double *samples = calloc((size_t)options.repeats, sizeof *samples);
if (samples == NULL) {
fputs("sample allocation failed\n", stderr);
return 1;
}
for (int i = 0; i < options.repeats; ++i) {
memset(fftw_hdr, 0, hdr_count * sizeof *fftw_hdr);
const double start = wall_seconds();
if (fast_psf_accumulator_resolve(&acc, fftw_hdr, omp_get_max_threads())) {
fputs("FFTW resolve failed\n", stderr);
return 1;
}
samples[i] = wall_seconds() - start;
}
const uint64_t fftw_hash = hash_bytes(fftw_hdr, hdr_count);
double sorted[128];
memcpy(sorted, samples, (size_t)options.repeats * sizeof *sorted);
qsort(sorted, (size_t)options.repeats, sizeof *sorted, compare_double);
printf("fftw: min=%.6f s median=%.6f s max=%.6f s\n", sorted[0],
sorted[options.repeats / 2], sorted[options.repeats - 1]);
printf("fftw samples:");
for (int i = 0; i < options.repeats; ++i)
printf(" %.6f", samples[i]);
printf("\nfftw_hdr_hash=%016llx\n", (unsigned long long)fftw_hash);
if (options.run_spatial) {
memset(spatial_hdr, 0, hdr_count * sizeof *spatial_hdr);
const double start = wall_seconds();
if (fast_psf_accumulator_resolve_spatial_reference(
&acc, spatial_hdr, omp_get_max_threads())) {
fputs("spatial resolve failed\n", stderr);
return 1;
}
const double spatial_seconds = wall_seconds() - start;
printf("spatial: %.6f s\n", spatial_seconds);
printf("spatial_hdr_hash=%016llx\n",
(unsigned long long)hash_bytes(spatial_hdr, hdr_count));
double max_abs = 0.0, peak = 0.0;
for (size_t i = 0; i < hdr_count; ++i) {
peak = fmax(peak, fabs(spatial_hdr[i]));
max_abs = fmax(max_abs, fabs(fftw_hdr[i] - spatial_hdr[i]));
}
printf("difference: max_abs=%.3g peak=%.3g speedup=%.3fx\n", max_abs, peak,
spatial_seconds / sorted[options.repeats / 2]);
}
struct rusage usage;
if (getrusage(RUSAGE_SELF, &usage) == 0)
printf("rss_max_kb=%ld\n", usage.ru_maxrss);
free(samples);
free(fftw_hdr);
free(spatial_hdr);
fast_psf_accumulator_destroy(&acc);
return 0;
}
+11 -1
View File
@@ -9,6 +9,7 @@ import tempfile
import zlib
BUILD = Path(sys.argv[1] if len(sys.argv) > 1 else 'build/Release').resolve()
TESTDIR = Path(sys.argv[2]).resolve() if len(sys.argv) > 2 else BUILD
ENV = dict(os.environ, OMP_NUM_THREADS='4')
@@ -67,6 +68,15 @@ with tempfile.TemporaryDirectory(prefix='gr-camera-cli-') as directory:
run(binary, *common, '--output', path, *options)
return image_payload(path)
# CPU fast-mode CLI smoke test: the FFTW resolve must run and report its
# one-time setup line.
if backend == 'minkowski':
fast_path = tmp / 'minkowski_fast.png'
fast = run(binary, *common, '--fast-mode', '--fast-supersample', 2,
'--output', fast_path)
assert 'Fast FFTW:' in fast.stderr, fast.stderr
assert image_payload(fast_path)
# Equivalent independently specified and inferred camera geometry.
inferred = render('position', '--observer-position', -30, 0, 0)
explicit = render('explicit', '--observer-position', -30, 0, 0,
@@ -117,7 +127,7 @@ with tempfile.TemporaryDirectory(prefix='gr-camera-cli-') as directory:
assert not missing_catalog.exists(), result.stderr
assert 'PSF cache ready' not in result.stderr
track = tmp / f'{backend}.csv'
run(BUILD / f'test_observer_{backend}', track)
run(TESTDIR / f'test_observer_{backend}', track)
single_map, movie_map = tmp / 'single.grlens', tmp / 'movie.grlens'
single = render('moving', '--observer-position', 3, -4, 5,
'--observer-velocity', 0.2, -0.1, 0.3,
+378
View File
@@ -0,0 +1,378 @@
/* Dedicated FFTW-versus-spatial fast-mode convolution regression.
*
* Both resolvers run on the identical impulse buffer and must agree to double
* rounding. The spatial resolver is the production reference, not a fallback.
* Compile only for the CPU PSF backend. */
#ifndef FAST_PSF_FFTW
#error "test_fast_psf_fftw requires -DFAST_PSF_FFTW (PSF_BACKEND=cpu)"
#endif
#include "optics.h"
#include "fast_psf_fftw.h"
#include <limits.h>
#include <math.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#ifndef INT_MAX
#define INT_MAX 2147483647
#endif
static int g_failures = 0;
static void fail(const char *what)
{
fprintf(stderr, "FAIL: %s\n", what);
++g_failures;
}
static PointSpreadFunction default_psf(void)
{
PointSpreadFunction psf = {.fwhm_pixels = 2.7, .moffat_beta = 4.5};
return psf;
}
static uint64_t hash_bytes(const double *values, size_t count)
{
const unsigned char *bytes = (const unsigned char *)values;
const size_t nbytes = count * sizeof *values;
uint64_t hash = 1469598103934665603ULL;
for (size_t i = 0; i < nbytes; ++i) {
hash ^= bytes[i];
hash *= 1099511628211ULL;
}
return hash;
}
typedef enum {
SCENE_CENTER_WHITE,
SCENE_DISTINCT_RGB,
SCENE_EDGES,
SCENE_BOUNDARY_PHASES,
SCENE_MULTIPLE,
SCENE_DENSE_CELL,
SCENE_RANDOM,
SCENE_EMPTY,
} Scene;
static void fill_scene(FastPsfAccumulator *acc, Scene scene, int width,
int height)
{
const LinearRgb white = {1.0, 1.0, 1.0};
switch (scene) {
case SCENE_CENTER_WHITE:
fast_psf_accumulator_deposit(acc, 0.5 * width, 0.5 * height, white, 1.0);
break;
case SCENE_DISTINCT_RGB:
fast_psf_accumulator_deposit(acc, 0.5 * width, 0.5 * height,
(LinearRgb){1.0, 0.25, 0.05}, 0.75);
break;
case SCENE_EDGES:
fast_psf_accumulator_deposit(acc, 0.3, 0.3, white, 1.0);
fast_psf_accumulator_deposit(acc, width - 0.7, 0.4, white, 1.0);
fast_psf_accumulator_deposit(acc, 0.5, height - 0.6, white, 1.0);
fast_psf_accumulator_deposit(acc, width - 0.4, height - 0.3, white, 1.0);
break;
case SCENE_BOUNDARY_PHASES:
fast_psf_accumulator_deposit(acc, 10.499, 12.499, white, 1.0);
fast_psf_accumulator_deposit(acc, 10.501, 12.501, white, 1.0);
break;
case SCENE_MULTIPLE:
fast_psf_accumulator_deposit(acc, 5.5, 5.5, (LinearRgb){1.0, 0.0, 0.0}, 0.5);
fast_psf_accumulator_deposit(acc, 12.25, 7.75, (LinearRgb){0.0, 1.0, 0.0}, 0.25);
fast_psf_accumulator_deposit(acc, 8.1, 14.9, (LinearRgb){0.0, 0.0, 1.0}, 1.5);
break;
case SCENE_DENSE_CELL:
for (int i = 0; i < 32; ++i)
fast_psf_accumulator_deposit(acc, 9.0 + 0.01 * i, 9.0 + 0.013 * i,
(LinearRgb){1.0, 0.5, 0.25}, 0.05);
break;
case SCENE_RANDOM: {
uint64_t state = 0x9e3779b97f4a7c15ULL;
const size_t count = (size_t)acc->supersampled_width *
acc->supersampled_height * 3;
for (size_t i = 0; i < count; ++i) {
state = state * 6364136223846793005ULL + 1442695040888963407ULL;
acc->buffer[i] += (double)(state >> 40) / (double)(1ULL << 24) - 0.5;
}
break;
}
case SCENE_EMPTY:
break;
}
}
static int run_compare(const char *name, int width, int height,
int supersample, FastPsfDeposit deposit, Scene scene,
int prefill)
{
FastPsfAccumulator acc = {0};
const PointSpreadFunction psf = default_psf();
const double relative_tail = 1e-8;
if (fast_psf_accumulator_init(&acc, width, height, supersample, deposit, &psf,
relative_tail, 0.0, 1)) {
fprintf(stderr, "FAIL: %s: accumulator init failed\n", name);
++g_failures;
return -1;
}
if (!acc.fftw_enabled) {
fprintf(stderr, "FAIL: %s: FFTW path not enabled\n", name);
++g_failures;
fast_psf_accumulator_destroy(&acc);
return -1;
}
const size_t hdr_count = (size_t)width * height * 3;
const size_t buffer_count = (size_t)acc.supersampled_width *
acc.supersampled_height * 3;
double *fftw_hdr = malloc(hdr_count * sizeof *fftw_hdr);
double *spatial_hdr = malloc(hdr_count * sizeof *spatial_hdr);
if (fftw_hdr == NULL || spatial_hdr == NULL) {
fprintf(stderr, "FAIL: %s: HDR allocation failed\n", name);
++g_failures;
free(fftw_hdr);
free(spatial_hdr);
fast_psf_accumulator_destroy(&acc);
return -1;
}
for (size_t i = 0; i < hdr_count; ++i)
fftw_hdr[i] = spatial_hdr[i] = prefill ? 0.25 : 0.0;
fill_scene(&acc, scene, width, height);
const uint64_t buffer_before = hash_bytes(acc.buffer, buffer_count);
if (fast_psf_accumulator_resolve(&acc, fftw_hdr, 4)) {
fprintf(stderr, "FAIL: %s: FFTW resolve failed\n", name);
++g_failures;
goto cleanup;
}
if (fast_psf_accumulator_resolve_spatial_reference(&acc, spatial_hdr, 4)) {
fprintf(stderr, "FAIL: %s: spatial reference resolve failed\n", name);
++g_failures;
goto cleanup;
}
if (hash_bytes(acc.buffer, buffer_count) != buffer_before) {
fprintf(stderr, "FAIL: %s: resolve mutated the impulse buffer\n", name);
++g_failures;
goto cleanup;
}
double peak = 0.0, max_abs = 0.0, max_rel = 0.0, sum_sq = 0.0;
double flux_fftw[3] = {0.0, 0.0, 0.0};
double flux_spatial[3] = {0.0, 0.0, 0.0};
size_t nan_count = 0;
for (size_t i = 0; i < hdr_count; ++i) {
const double reference = spatial_hdr[i];
const double error = fabs(fftw_hdr[i] - reference);
peak = fmax(peak, fabs(reference));
max_abs = fmax(max_abs, error);
sum_sq += error * error;
if (!isfinite(fftw_hdr[i]) || !isfinite(reference))
++nan_count;
flux_fftw[i % 3] += fftw_hdr[i];
flux_spatial[i % 3] += reference;
}
/* Mixed absolute/relative acceptance: the absolute floor covers the tiny
* Moffat far-wing samples where the FFT and the direct sum disagree only by
* roundoff, while the relative term checks significant samples. A crop,
* wrap, channel, or normalization bug produces O(1) errors well above both. */
const double rms = sqrt(sum_sq / (double)hdr_count);
const double abs_tol = 1e-10 * fmax(1.0, peak);
const double rel_tol = 1e-9;
const double rel_floor = 1e-6 * fmax(peak, 1.0);
double max_violation = 0.0;
for (size_t i = 0; i < hdr_count; ++i) {
const double reference = spatial_hdr[i];
const double error = fabs(fftw_hdr[i] - reference);
max_violation =
fmax(max_violation, error - (abs_tol + rel_tol * fabs(reference)));
if (fabs(reference) > rel_floor)
max_rel = fmax(max_rel, error / fabs(reference));
}
int worst = -1;
double worst_error = 0.0;
for (size_t i = 0; i < hdr_count; ++i) {
const double error = fabs(fftw_hdr[i] - spatial_hdr[i]);
if (error > worst_error) {
worst_error = error;
worst = (int)(i / 3);
}
}
double flux_rel = 0.0;
for (int c = 0; c < 3; ++c) {
const double denom = fmax(fabs(flux_spatial[c]), 1e-30);
flux_rel = fmax(flux_rel, fabs(flux_fftw[c] - flux_spatial[c]) / denom);
}
if (nan_count != 0 || max_violation > 0.0 || max_rel > rel_tol ||
flux_rel > 1e-10) {
fprintf(stderr,
"FAIL: %s: peak=%.6g max_abs=%.3g (tol %.3g) max_rel=%.3g "
"rms=%.3g flux_rel=%.3g nan=%zu worst_pixel=%d\n",
name, peak, max_abs, abs_tol, max_rel, rms, flux_rel, nan_count,
worst);
++g_failures;
} else {
printf("ok %-28s peak=%.4g max_abs=%.3g max_rel=%.3g rms=%.3g\n", name,
peak, max_abs, max_rel, rms);
}
cleanup:
free(fftw_hdr);
free(spatial_hdr);
fast_psf_accumulator_destroy(&acc);
return g_failures == 0 ? 0 : -1;
}
static void test_next_smooth_size(void)
{
struct {
size_t input;
long expected; /* -1 = must fail */
} cases[] = {
{1, 1}, {2, 2}, {3, 3}, {4, 4}, {5, 5}, {6, 6},
{7, 7}, {8, 8}, {9, 9}, {11, 12}, {13, 14}, {15, 15},
{121, 125}, {127, 128}, {200, 200}, {241, 243},
{(size_t)INT_MAX + 1, -1},
{0, -1},
};
for (size_t i = 0; i < sizeof cases / sizeof cases[0]; ++i) {
size_t out = 0;
const int rc = fast_psf_fftw_next_smooth_size(cases[i].input, &out);
if (cases[i].expected < 0) {
if (rc == 0)
fprintf(stderr, "FAIL: next_smooth_size(%zu) unexpectedly returned %zu\n",
cases[i].input, out), ++g_failures;
} else if (rc != 0 || out != (size_t)cases[i].expected) {
fprintf(stderr, "FAIL: next_smooth_size(%zu) = rc %d, %zu (want %ld)\n",
cases[i].input, rc, out, cases[i].expected);
++g_failures;
}
}
size_t out = 0;
if (fast_psf_fftw_next_smooth_size((size_t)1 << 30, &out) != 0 ||
out > (size_t)INT_MAX)
fail("next_smooth_size near INT_MAX did not return a valid FFT size");
}
/* A single impulse must not wrap to the opposite edge, and an empty buffer must
* leave the framebuffer untouched. */
static void test_no_wraparound_and_empty(void)
{
PointSpreadFunction psf = default_psf();
/* A compact kernel keeps the image much wider than the support so a genuine
* circular-wrap bug would show up as an opposite-edge ghost. */
psf.fwhm_pixels = 0.02;
const double relative_tail = 1e-8;
const int width = 12, height = 12, supersample = 2;
FastPsfAccumulator acc = {0};
if (fast_psf_accumulator_init(&acc, width, height, supersample,
FAST_PSF_DEPOSIT_NEAREST, &psf, relative_tail,
0.0, 1)) {
fail("wraparound accumulator init");
return;
}
const size_t count = (size_t)width * height * 3;
double *hdr = calloc(count, sizeof *hdr);
if (hdr == NULL) {
fail("wraparound allocation");
fast_psf_accumulator_destroy(&acc);
return;
}
fast_psf_accumulator_deposit(&acc, 0.6, 0.6, (LinearRgb){1.0, 1.0, 1.0}, 1.0);
if (fast_psf_accumulator_resolve(&acc, hdr, 2)) {
fail("wraparound resolve");
} else {
const int far_x = width - 1, far_y = height - 1;
const double ghost = hdr[3 * (far_y * width + far_x)];
if (fabs(ghost) > 1e-15)
fprintf(stderr, "FAIL: opposite-edge ghost value %.3g\n", ghost),
++g_failures;
if (hdr[0] <= 0.0)
fail("impulse peak missing near the deposited corner");
}
/* Empty buffer: the same accumulator, after clearing, must not change HDR. */
fast_psf_accumulator_clear(&acc);
double background[3] = {0.25, 0.5, 0.75};
for (size_t i = 0; i < count; ++i)
hdr[i] = background[i % 3];
if (fast_psf_accumulator_resolve(&acc, hdr, 2))
fail("empty resolve");
for (size_t i = 0; i < count; ++i) {
if (hdr[i] != background[i % 3]) {
fail("empty input changed the HDR framebuffer");
break;
}
}
free(hdr);
fast_psf_accumulator_destroy(&acc);
}
/* Channel separation: a pure-red impulse must leave green and blue at zero. */
static void test_channel_isolation(void)
{
const PointSpreadFunction psf = default_psf();
const int width = 20, height = 18;
FastPsfAccumulator acc = {0};
if (fast_psf_accumulator_init(&acc, width, height, 2,
FAST_PSF_DEPOSIT_NEAREST, &psf, 1e-8, 0.0, 1)) {
fail("channel isolation init");
return;
}
const size_t count = (size_t)width * height * 3;
double *hdr = calloc(count, sizeof *hdr);
fast_psf_accumulator_deposit(&acc, 10.0, 9.0, (LinearRgb){1.0, 0.0, 0.0}, 1.0);
if (fast_psf_accumulator_resolve(&acc, hdr, 2))
fail("channel isolation resolve");
for (size_t i = 0; i < count; i += 3) {
if (hdr[i + 1] != 0.0 || hdr[i + 2] != 0.0) {
fail("green/blue channel leaked into a pure-red impulse");
break;
}
}
free(hdr);
fast_psf_accumulator_destroy(&acc);
}
int main(void)
{
test_next_smooth_size();
test_no_wraparound_and_empty();
test_channel_isolation();
const int sizes[][2] = {{17, 13}, {16, 16}, {23, 31}, {33, 17}};
const int supersamples[] = {1, 2, 3, 4};
const FastPsfDeposit deposits[] = {FAST_PSF_DEPOSIT_NEAREST,
FAST_PSF_DEPOSIT_BILINEAR};
for (size_t s = 0; s < sizeof sizes / sizeof sizes[0]; ++s) {
for (size_t n = 0; n < sizeof supersamples / sizeof supersamples[0]; ++n) {
for (size_t d = 0; d < sizeof deposits / sizeof deposits[0]; ++d) {
for (Scene scene = SCENE_CENTER_WHITE; scene <= SCENE_EMPTY;
++scene) {
char name[128];
snprintf(name, sizeof name, "%dx%d N%d %s scene%d", sizes[s][0],
sizes[s][1], supersamples[n],
deposits[d] == FAST_PSF_DEPOSIT_NEAREST ? "near" : "bilin",
(int)scene);
if (run_compare(name, sizes[s][0], sizes[s][1], supersamples[n],
deposits[d], scene, 0) != 0) {
/* Keep going: collect all failures before summarizing. */
}
}
}
}
}
run_compare("prefilled background", 24, 20, 2, FAST_PSF_DEPOSIT_NEAREST,
SCENE_MULTIPLE, 1);
/* One axis (height) stays at the exact FFT size while the other grows. */
run_compare("single-axis padding", 32, 8, 2, FAST_PSF_DEPOSIT_NEAREST,
SCENE_CENTER_WHITE, 0);
if (g_failures != 0) {
fprintf(stderr, "%d fast-PSF FFTW regression failure(s)\n", g_failures);
return 1;
}
puts("fast PSF FFTW regressions passed");
return 0;
}