Feat: Use FFTW linear convolution for CPU fast-mode PSF resolve
Replace the nested spatial global convolution in fast_psf_accumulator_resolve with a reusable double-precision FFTW linear convolution on the CPU PSF backend.
- Add the private src/fast_psf_fftw.{c,h} module: zero-padded R2C/C2R plans, cached kernel spectrum and planar scratch, exact 1/(Pwidth*Pheight) and 1/N^2 normalization, (R,R) crop, and additive HDR output.
- Keep the previous nested loops as fast_psf_accumulator_resolve_spatial_reference for tests/benchmarks only; it is not a runtime fallback.
- Cache the circular row spans on FastPsfAccumulator and report one-time plan, kernel transform, scratch, and per-frame stage timings.
- Require fftw3_omp for CPU builds; HIP and dummy builds do not link FFTW.
- Namespace test/helper binaries by spacetime and build tag, and reject make test / psf-capture for non-CPU backends.
- Add tests/test_fast_psf_fftw.c (FFTW versus spatial), tests/benchmark_fast_psf_fftw.c, an FFTW CLI smoke check, and the 2026-09-25 benchmark record.
This commit is contained in:
1 parent
3deebfb2fa
commit
229f50cd86
14 files changed
+2372
-55
No files matched your search
@@ -0,0 +1,223 @@
|
||||
/* Resolver-only benchmark for the fast-mode FFTW global convolution.
|
||||
*
|
||||
* Fills one deterministic impulse buffer, then times the spatial reference and
|
||||
* the FFTW production resolver on that identical buffer. First-use plan/setup
|
||||
* is reported separately from steady-state execution. No catalog or geodesic
|
||||
* work is involved. */
|
||||
#ifndef FAST_PSF_FFTW
|
||||
#error "benchmark_fast_psf_fftw requires -DFAST_PSF_FFTW (PSF_BACKEND=cpu)"
|
||||
#endif
|
||||
|
||||
#include "fast_psf_fftw.h"
|
||||
#include "optics.h"
|
||||
|
||||
#include <math.h>
|
||||
#include <omp.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/resource.h>
|
||||
|
||||
typedef struct {
|
||||
int width, height, supersample, repeats;
|
||||
int run_spatial;
|
||||
int measure;
|
||||
FastPsfDeposit deposit;
|
||||
} Options;
|
||||
|
||||
static int parse_int(const char *text, int *out)
|
||||
{
|
||||
char *end = NULL;
|
||||
const long value = strtol(text, &end, 10);
|
||||
if (end == text || *end != '\0' || value <= 0 || value > 1000000)
|
||||
return -1;
|
||||
*out = (int)value;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int parse_options(int argc, char **argv, Options *options)
|
||||
{
|
||||
*options = (Options){.width = 320,
|
||||
.height = 240,
|
||||
.supersample = 2,
|
||||
.repeats = 5,
|
||||
.run_spatial = 0,
|
||||
.measure = 0,
|
||||
.deposit = FAST_PSF_DEPOSIT_NEAREST};
|
||||
for (int i = 1; i < argc; ++i) {
|
||||
if (!strcmp(argv[i], "--spatial")) {
|
||||
options->run_spatial = 1;
|
||||
} else if (!strcmp(argv[i], "--measure")) {
|
||||
options->measure = 1;
|
||||
} else if (!strcmp(argv[i], "--bilinear")) {
|
||||
options->deposit = FAST_PSF_DEPOSIT_BILINEAR;
|
||||
} else if (!strcmp(argv[i], "--width") && i + 1 < argc) {
|
||||
if (parse_int(argv[++i], &options->width))
|
||||
return -1;
|
||||
} else if (!strcmp(argv[i], "--height") && i + 1 < argc) {
|
||||
if (parse_int(argv[++i], &options->height))
|
||||
return -1;
|
||||
} else if (!strcmp(argv[i], "--supersample") && i + 1 < argc) {
|
||||
if (parse_int(argv[++i], &options->supersample))
|
||||
return -1;
|
||||
} else if (!strcmp(argv[i], "--repeats") && i + 1 < argc) {
|
||||
if (parse_int(argv[++i], &options->repeats))
|
||||
return -1;
|
||||
} else {
|
||||
fprintf(stderr, "unknown argument '%s'\n", argv[i]);
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static uint64_t hash_bytes(const double *values, size_t count)
|
||||
{
|
||||
const unsigned char *bytes = (const unsigned char *)values;
|
||||
const size_t nbytes = count * sizeof *values;
|
||||
uint64_t hash = 1469598103934665603ULL;
|
||||
for (size_t i = 0; i < nbytes; ++i) {
|
||||
hash ^= bytes[i];
|
||||
hash *= 1099511628211ULL;
|
||||
}
|
||||
return hash;
|
||||
}
|
||||
|
||||
static double wall_seconds(void)
|
||||
{
|
||||
return omp_get_wtime();
|
||||
}
|
||||
|
||||
static int compare_double(const void *a, const void *b)
|
||||
{
|
||||
const double x = *(const double *)a;
|
||||
const double y = *(const double *)b;
|
||||
return x < y ? -1 : x > y ? 1 : 0;
|
||||
}
|
||||
|
||||
int main(int argc, char **argv)
|
||||
{
|
||||
Options options;
|
||||
if (parse_options(argc, argv, &options)) {
|
||||
fprintf(stderr,
|
||||
"usage: %s [--width W] [--height H] [--supersample N] "
|
||||
"[--repeats R] [--bilinear] [--spatial] [--measure]\n",
|
||||
argv[0]);
|
||||
return 2;
|
||||
}
|
||||
const PointSpreadFunction psf = {.fwhm_pixels = 2.7, .moffat_beta = 4.5};
|
||||
const double relative_tail = 1e-8;
|
||||
if (options.repeats > 128)
|
||||
options.repeats = 128;
|
||||
|
||||
printf("impulse: %dx%d supersample=%d deposit=%s repeats=%d spatial=%d\n",
|
||||
options.width, options.height, options.supersample,
|
||||
options.deposit == FAST_PSF_DEPOSIT_NEAREST ? "nearest" : "bilinear",
|
||||
options.repeats, options.run_spatial);
|
||||
printf("plan_mode=%s\n", options.measure ? "measure" : "estimate");
|
||||
fast_psf_fftw_set_plan_mode(options.measure);
|
||||
|
||||
FastPsfAccumulator acc = {0};
|
||||
const double init_start = wall_seconds();
|
||||
if (fast_psf_accumulator_init(&acc, options.width, options.height,
|
||||
options.supersample, options.deposit, &psf,
|
||||
relative_tail, 0.0, 1)) {
|
||||
fputs("accumulator initialization failed\n", stderr);
|
||||
return 1;
|
||||
}
|
||||
const double init_seconds = wall_seconds() - init_start;
|
||||
printf("setup: init=%.6f s fftw_plan=%.6f s kernel_fft=%.6f s\n",
|
||||
init_seconds, acc.fftw_setup_seconds, acc.fftw_kernel_seconds);
|
||||
fast_psf_fftw_report(acc.fftw, stdout);
|
||||
|
||||
const size_t ss_count = (size_t)acc.supersampled_width *
|
||||
acc.supersampled_height;
|
||||
uint64_t state = 0x9e3779b97f4a7c15ULL;
|
||||
for (size_t i = 0; i < ss_count; ++i) {
|
||||
state = state * 6364136223846793005ULL + 1442695040888963407ULL;
|
||||
if ((state >> 58) != 0)
|
||||
continue;
|
||||
state = state * 6364136223846793005ULL + 1442695040888963407ULL;
|
||||
acc.buffer[3 * i] = (double)(state >> 40) / (double)(1ULL << 24);
|
||||
state = state * 6364136223846793005ULL + 1442695040888963407ULL;
|
||||
acc.buffer[3 * i + 1] = (double)(state >> 40) / (double)(1ULL << 24);
|
||||
state = state * 6364136223846793005ULL + 1442695040888963407ULL;
|
||||
acc.buffer[3 * i + 2] = (double)(state >> 40) / (double)(1ULL << 24);
|
||||
}
|
||||
const size_t hdr_count = (size_t)options.width * options.height * 3;
|
||||
const size_t buffer_count = ss_count * 3;
|
||||
const uint64_t impulse_hash = hash_bytes(acc.buffer, buffer_count);
|
||||
printf("impulse_hash=%016llx\n", (unsigned long long)impulse_hash);
|
||||
|
||||
double *fftw_hdr = calloc(hdr_count, sizeof *fftw_hdr);
|
||||
double *spatial_hdr = calloc(hdr_count, sizeof *spatial_hdr);
|
||||
if (fftw_hdr == NULL || spatial_hdr == NULL) {
|
||||
fputs("HDR allocation failed\n", stderr);
|
||||
free(fftw_hdr);
|
||||
free(spatial_hdr);
|
||||
fast_psf_accumulator_destroy(&acc);
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* Warm-up: first execution includes any lazy per-frame allocation. */
|
||||
if (fast_psf_accumulator_resolve(&acc, fftw_hdr, omp_get_max_threads())) {
|
||||
fputs("FFTW warm-up resolve failed\n", stderr);
|
||||
return 1;
|
||||
}
|
||||
double *samples = calloc((size_t)options.repeats, sizeof *samples);
|
||||
if (samples == NULL) {
|
||||
fputs("sample allocation failed\n", stderr);
|
||||
return 1;
|
||||
}
|
||||
for (int i = 0; i < options.repeats; ++i) {
|
||||
memset(fftw_hdr, 0, hdr_count * sizeof *fftw_hdr);
|
||||
const double start = wall_seconds();
|
||||
if (fast_psf_accumulator_resolve(&acc, fftw_hdr, omp_get_max_threads())) {
|
||||
fputs("FFTW resolve failed\n", stderr);
|
||||
return 1;
|
||||
}
|
||||
samples[i] = wall_seconds() - start;
|
||||
}
|
||||
const uint64_t fftw_hash = hash_bytes(fftw_hdr, hdr_count);
|
||||
double sorted[128];
|
||||
memcpy(sorted, samples, (size_t)options.repeats * sizeof *sorted);
|
||||
qsort(sorted, (size_t)options.repeats, sizeof *sorted, compare_double);
|
||||
printf("fftw: min=%.6f s median=%.6f s max=%.6f s\n", sorted[0],
|
||||
sorted[options.repeats / 2], sorted[options.repeats - 1]);
|
||||
printf("fftw samples:");
|
||||
for (int i = 0; i < options.repeats; ++i)
|
||||
printf(" %.6f", samples[i]);
|
||||
printf("\nfftw_hdr_hash=%016llx\n", (unsigned long long)fftw_hash);
|
||||
|
||||
if (options.run_spatial) {
|
||||
memset(spatial_hdr, 0, hdr_count * sizeof *spatial_hdr);
|
||||
const double start = wall_seconds();
|
||||
if (fast_psf_accumulator_resolve_spatial_reference(
|
||||
&acc, spatial_hdr, omp_get_max_threads())) {
|
||||
fputs("spatial resolve failed\n", stderr);
|
||||
return 1;
|
||||
}
|
||||
const double spatial_seconds = wall_seconds() - start;
|
||||
printf("spatial: %.6f s\n", spatial_seconds);
|
||||
printf("spatial_hdr_hash=%016llx\n",
|
||||
(unsigned long long)hash_bytes(spatial_hdr, hdr_count));
|
||||
double max_abs = 0.0, peak = 0.0;
|
||||
for (size_t i = 0; i < hdr_count; ++i) {
|
||||
peak = fmax(peak, fabs(spatial_hdr[i]));
|
||||
max_abs = fmax(max_abs, fabs(fftw_hdr[i] - spatial_hdr[i]));
|
||||
}
|
||||
printf("difference: max_abs=%.3g peak=%.3g speedup=%.3fx\n", max_abs, peak,
|
||||
spatial_seconds / sorted[options.repeats / 2]);
|
||||
}
|
||||
|
||||
struct rusage usage;
|
||||
if (getrusage(RUSAGE_SELF, &usage) == 0)
|
||||
printf("rss_max_kb=%ld\n", usage.ru_maxrss);
|
||||
|
||||
free(samples);
|
||||
free(fftw_hdr);
|
||||
free(spatial_hdr);
|
||||
fast_psf_accumulator_destroy(&acc);
|
||||
return 0;
|
||||
}
|
||||
@@ -9,6 +9,7 @@ import tempfile
|
||||
import zlib
|
||||
|
||||
BUILD = Path(sys.argv[1] if len(sys.argv) > 1 else 'build/Release').resolve()
|
||||
TESTDIR = Path(sys.argv[2]).resolve() if len(sys.argv) > 2 else BUILD
|
||||
ENV = dict(os.environ, OMP_NUM_THREADS='4')
|
||||
|
||||
|
||||
@@ -67,6 +68,15 @@ with tempfile.TemporaryDirectory(prefix='gr-camera-cli-') as directory:
|
||||
run(binary, *common, '--output', path, *options)
|
||||
return image_payload(path)
|
||||
|
||||
# CPU fast-mode CLI smoke test: the FFTW resolve must run and report its
|
||||
# one-time setup line.
|
||||
if backend == 'minkowski':
|
||||
fast_path = tmp / 'minkowski_fast.png'
|
||||
fast = run(binary, *common, '--fast-mode', '--fast-supersample', 2,
|
||||
'--output', fast_path)
|
||||
assert 'Fast FFTW:' in fast.stderr, fast.stderr
|
||||
assert image_payload(fast_path)
|
||||
|
||||
# Equivalent independently specified and inferred camera geometry.
|
||||
inferred = render('position', '--observer-position', -30, 0, 0)
|
||||
explicit = render('explicit', '--observer-position', -30, 0, 0,
|
||||
@@ -117,7 +127,7 @@ with tempfile.TemporaryDirectory(prefix='gr-camera-cli-') as directory:
|
||||
assert not missing_catalog.exists(), result.stderr
|
||||
assert 'PSF cache ready' not in result.stderr
|
||||
track = tmp / f'{backend}.csv'
|
||||
run(BUILD / f'test_observer_{backend}', track)
|
||||
run(TESTDIR / f'test_observer_{backend}', track)
|
||||
single_map, movie_map = tmp / 'single.grlens', tmp / 'movie.grlens'
|
||||
single = render('moving', '--observer-position', 3, -4, 5,
|
||||
'--observer-velocity', 0.2, -0.1, 0.3,
|
||||
|
||||
@@ -0,0 +1,378 @@
|
||||
/* Dedicated FFTW-versus-spatial fast-mode convolution regression.
|
||||
*
|
||||
* Both resolvers run on the identical impulse buffer and must agree to double
|
||||
* rounding. The spatial resolver is the production reference, not a fallback.
|
||||
* Compile only for the CPU PSF backend. */
|
||||
#ifndef FAST_PSF_FFTW
|
||||
#error "test_fast_psf_fftw requires -DFAST_PSF_FFTW (PSF_BACKEND=cpu)"
|
||||
#endif
|
||||
|
||||
#include "optics.h"
|
||||
#include "fast_psf_fftw.h"
|
||||
|
||||
#include <limits.h>
|
||||
#include <math.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
#ifndef INT_MAX
|
||||
#define INT_MAX 2147483647
|
||||
#endif
|
||||
|
||||
static int g_failures = 0;
|
||||
|
||||
static void fail(const char *what)
|
||||
{
|
||||
fprintf(stderr, "FAIL: %s\n", what);
|
||||
++g_failures;
|
||||
}
|
||||
|
||||
static PointSpreadFunction default_psf(void)
|
||||
{
|
||||
PointSpreadFunction psf = {.fwhm_pixels = 2.7, .moffat_beta = 4.5};
|
||||
return psf;
|
||||
}
|
||||
|
||||
static uint64_t hash_bytes(const double *values, size_t count)
|
||||
{
|
||||
const unsigned char *bytes = (const unsigned char *)values;
|
||||
const size_t nbytes = count * sizeof *values;
|
||||
uint64_t hash = 1469598103934665603ULL;
|
||||
for (size_t i = 0; i < nbytes; ++i) {
|
||||
hash ^= bytes[i];
|
||||
hash *= 1099511628211ULL;
|
||||
}
|
||||
return hash;
|
||||
}
|
||||
|
||||
typedef enum {
|
||||
SCENE_CENTER_WHITE,
|
||||
SCENE_DISTINCT_RGB,
|
||||
SCENE_EDGES,
|
||||
SCENE_BOUNDARY_PHASES,
|
||||
SCENE_MULTIPLE,
|
||||
SCENE_DENSE_CELL,
|
||||
SCENE_RANDOM,
|
||||
SCENE_EMPTY,
|
||||
} Scene;
|
||||
|
||||
static void fill_scene(FastPsfAccumulator *acc, Scene scene, int width,
|
||||
int height)
|
||||
{
|
||||
const LinearRgb white = {1.0, 1.0, 1.0};
|
||||
switch (scene) {
|
||||
case SCENE_CENTER_WHITE:
|
||||
fast_psf_accumulator_deposit(acc, 0.5 * width, 0.5 * height, white, 1.0);
|
||||
break;
|
||||
case SCENE_DISTINCT_RGB:
|
||||
fast_psf_accumulator_deposit(acc, 0.5 * width, 0.5 * height,
|
||||
(LinearRgb){1.0, 0.25, 0.05}, 0.75);
|
||||
break;
|
||||
case SCENE_EDGES:
|
||||
fast_psf_accumulator_deposit(acc, 0.3, 0.3, white, 1.0);
|
||||
fast_psf_accumulator_deposit(acc, width - 0.7, 0.4, white, 1.0);
|
||||
fast_psf_accumulator_deposit(acc, 0.5, height - 0.6, white, 1.0);
|
||||
fast_psf_accumulator_deposit(acc, width - 0.4, height - 0.3, white, 1.0);
|
||||
break;
|
||||
case SCENE_BOUNDARY_PHASES:
|
||||
fast_psf_accumulator_deposit(acc, 10.499, 12.499, white, 1.0);
|
||||
fast_psf_accumulator_deposit(acc, 10.501, 12.501, white, 1.0);
|
||||
break;
|
||||
case SCENE_MULTIPLE:
|
||||
fast_psf_accumulator_deposit(acc, 5.5, 5.5, (LinearRgb){1.0, 0.0, 0.0}, 0.5);
|
||||
fast_psf_accumulator_deposit(acc, 12.25, 7.75, (LinearRgb){0.0, 1.0, 0.0}, 0.25);
|
||||
fast_psf_accumulator_deposit(acc, 8.1, 14.9, (LinearRgb){0.0, 0.0, 1.0}, 1.5);
|
||||
break;
|
||||
case SCENE_DENSE_CELL:
|
||||
for (int i = 0; i < 32; ++i)
|
||||
fast_psf_accumulator_deposit(acc, 9.0 + 0.01 * i, 9.0 + 0.013 * i,
|
||||
(LinearRgb){1.0, 0.5, 0.25}, 0.05);
|
||||
break;
|
||||
case SCENE_RANDOM: {
|
||||
uint64_t state = 0x9e3779b97f4a7c15ULL;
|
||||
const size_t count = (size_t)acc->supersampled_width *
|
||||
acc->supersampled_height * 3;
|
||||
for (size_t i = 0; i < count; ++i) {
|
||||
state = state * 6364136223846793005ULL + 1442695040888963407ULL;
|
||||
acc->buffer[i] += (double)(state >> 40) / (double)(1ULL << 24) - 0.5;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case SCENE_EMPTY:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
static int run_compare(const char *name, int width, int height,
|
||||
int supersample, FastPsfDeposit deposit, Scene scene,
|
||||
int prefill)
|
||||
{
|
||||
FastPsfAccumulator acc = {0};
|
||||
const PointSpreadFunction psf = default_psf();
|
||||
const double relative_tail = 1e-8;
|
||||
if (fast_psf_accumulator_init(&acc, width, height, supersample, deposit, &psf,
|
||||
relative_tail, 0.0, 1)) {
|
||||
fprintf(stderr, "FAIL: %s: accumulator init failed\n", name);
|
||||
++g_failures;
|
||||
return -1;
|
||||
}
|
||||
if (!acc.fftw_enabled) {
|
||||
fprintf(stderr, "FAIL: %s: FFTW path not enabled\n", name);
|
||||
++g_failures;
|
||||
fast_psf_accumulator_destroy(&acc);
|
||||
return -1;
|
||||
}
|
||||
const size_t hdr_count = (size_t)width * height * 3;
|
||||
const size_t buffer_count = (size_t)acc.supersampled_width *
|
||||
acc.supersampled_height * 3;
|
||||
double *fftw_hdr = malloc(hdr_count * sizeof *fftw_hdr);
|
||||
double *spatial_hdr = malloc(hdr_count * sizeof *spatial_hdr);
|
||||
if (fftw_hdr == NULL || spatial_hdr == NULL) {
|
||||
fprintf(stderr, "FAIL: %s: HDR allocation failed\n", name);
|
||||
++g_failures;
|
||||
free(fftw_hdr);
|
||||
free(spatial_hdr);
|
||||
fast_psf_accumulator_destroy(&acc);
|
||||
return -1;
|
||||
}
|
||||
for (size_t i = 0; i < hdr_count; ++i)
|
||||
fftw_hdr[i] = spatial_hdr[i] = prefill ? 0.25 : 0.0;
|
||||
fill_scene(&acc, scene, width, height);
|
||||
const uint64_t buffer_before = hash_bytes(acc.buffer, buffer_count);
|
||||
|
||||
if (fast_psf_accumulator_resolve(&acc, fftw_hdr, 4)) {
|
||||
fprintf(stderr, "FAIL: %s: FFTW resolve failed\n", name);
|
||||
++g_failures;
|
||||
goto cleanup;
|
||||
}
|
||||
if (fast_psf_accumulator_resolve_spatial_reference(&acc, spatial_hdr, 4)) {
|
||||
fprintf(stderr, "FAIL: %s: spatial reference resolve failed\n", name);
|
||||
++g_failures;
|
||||
goto cleanup;
|
||||
}
|
||||
if (hash_bytes(acc.buffer, buffer_count) != buffer_before) {
|
||||
fprintf(stderr, "FAIL: %s: resolve mutated the impulse buffer\n", name);
|
||||
++g_failures;
|
||||
goto cleanup;
|
||||
}
|
||||
|
||||
double peak = 0.0, max_abs = 0.0, max_rel = 0.0, sum_sq = 0.0;
|
||||
double flux_fftw[3] = {0.0, 0.0, 0.0};
|
||||
double flux_spatial[3] = {0.0, 0.0, 0.0};
|
||||
size_t nan_count = 0;
|
||||
for (size_t i = 0; i < hdr_count; ++i) {
|
||||
const double reference = spatial_hdr[i];
|
||||
const double error = fabs(fftw_hdr[i] - reference);
|
||||
peak = fmax(peak, fabs(reference));
|
||||
max_abs = fmax(max_abs, error);
|
||||
sum_sq += error * error;
|
||||
if (!isfinite(fftw_hdr[i]) || !isfinite(reference))
|
||||
++nan_count;
|
||||
flux_fftw[i % 3] += fftw_hdr[i];
|
||||
flux_spatial[i % 3] += reference;
|
||||
}
|
||||
/* Mixed absolute/relative acceptance: the absolute floor covers the tiny
|
||||
* Moffat far-wing samples where the FFT and the direct sum disagree only by
|
||||
* roundoff, while the relative term checks significant samples. A crop,
|
||||
* wrap, channel, or normalization bug produces O(1) errors well above both. */
|
||||
const double rms = sqrt(sum_sq / (double)hdr_count);
|
||||
const double abs_tol = 1e-10 * fmax(1.0, peak);
|
||||
const double rel_tol = 1e-9;
|
||||
const double rel_floor = 1e-6 * fmax(peak, 1.0);
|
||||
double max_violation = 0.0;
|
||||
for (size_t i = 0; i < hdr_count; ++i) {
|
||||
const double reference = spatial_hdr[i];
|
||||
const double error = fabs(fftw_hdr[i] - reference);
|
||||
max_violation =
|
||||
fmax(max_violation, error - (abs_tol + rel_tol * fabs(reference)));
|
||||
if (fabs(reference) > rel_floor)
|
||||
max_rel = fmax(max_rel, error / fabs(reference));
|
||||
}
|
||||
int worst = -1;
|
||||
double worst_error = 0.0;
|
||||
for (size_t i = 0; i < hdr_count; ++i) {
|
||||
const double error = fabs(fftw_hdr[i] - spatial_hdr[i]);
|
||||
if (error > worst_error) {
|
||||
worst_error = error;
|
||||
worst = (int)(i / 3);
|
||||
}
|
||||
}
|
||||
double flux_rel = 0.0;
|
||||
for (int c = 0; c < 3; ++c) {
|
||||
const double denom = fmax(fabs(flux_spatial[c]), 1e-30);
|
||||
flux_rel = fmax(flux_rel, fabs(flux_fftw[c] - flux_spatial[c]) / denom);
|
||||
}
|
||||
|
||||
if (nan_count != 0 || max_violation > 0.0 || max_rel > rel_tol ||
|
||||
flux_rel > 1e-10) {
|
||||
fprintf(stderr,
|
||||
"FAIL: %s: peak=%.6g max_abs=%.3g (tol %.3g) max_rel=%.3g "
|
||||
"rms=%.3g flux_rel=%.3g nan=%zu worst_pixel=%d\n",
|
||||
name, peak, max_abs, abs_tol, max_rel, rms, flux_rel, nan_count,
|
||||
worst);
|
||||
++g_failures;
|
||||
} else {
|
||||
printf("ok %-28s peak=%.4g max_abs=%.3g max_rel=%.3g rms=%.3g\n", name,
|
||||
peak, max_abs, max_rel, rms);
|
||||
}
|
||||
|
||||
cleanup:
|
||||
free(fftw_hdr);
|
||||
free(spatial_hdr);
|
||||
fast_psf_accumulator_destroy(&acc);
|
||||
return g_failures == 0 ? 0 : -1;
|
||||
}
|
||||
|
||||
static void test_next_smooth_size(void)
|
||||
{
|
||||
struct {
|
||||
size_t input;
|
||||
long expected; /* -1 = must fail */
|
||||
} cases[] = {
|
||||
{1, 1}, {2, 2}, {3, 3}, {4, 4}, {5, 5}, {6, 6},
|
||||
{7, 7}, {8, 8}, {9, 9}, {11, 12}, {13, 14}, {15, 15},
|
||||
{121, 125}, {127, 128}, {200, 200}, {241, 243},
|
||||
{(size_t)INT_MAX + 1, -1},
|
||||
{0, -1},
|
||||
};
|
||||
for (size_t i = 0; i < sizeof cases / sizeof cases[0]; ++i) {
|
||||
size_t out = 0;
|
||||
const int rc = fast_psf_fftw_next_smooth_size(cases[i].input, &out);
|
||||
if (cases[i].expected < 0) {
|
||||
if (rc == 0)
|
||||
fprintf(stderr, "FAIL: next_smooth_size(%zu) unexpectedly returned %zu\n",
|
||||
cases[i].input, out), ++g_failures;
|
||||
} else if (rc != 0 || out != (size_t)cases[i].expected) {
|
||||
fprintf(stderr, "FAIL: next_smooth_size(%zu) = rc %d, %zu (want %ld)\n",
|
||||
cases[i].input, rc, out, cases[i].expected);
|
||||
++g_failures;
|
||||
}
|
||||
}
|
||||
size_t out = 0;
|
||||
if (fast_psf_fftw_next_smooth_size((size_t)1 << 30, &out) != 0 ||
|
||||
out > (size_t)INT_MAX)
|
||||
fail("next_smooth_size near INT_MAX did not return a valid FFT size");
|
||||
}
|
||||
|
||||
/* A single impulse must not wrap to the opposite edge, and an empty buffer must
|
||||
* leave the framebuffer untouched. */
|
||||
static void test_no_wraparound_and_empty(void)
|
||||
{
|
||||
PointSpreadFunction psf = default_psf();
|
||||
/* A compact kernel keeps the image much wider than the support so a genuine
|
||||
* circular-wrap bug would show up as an opposite-edge ghost. */
|
||||
psf.fwhm_pixels = 0.02;
|
||||
const double relative_tail = 1e-8;
|
||||
const int width = 12, height = 12, supersample = 2;
|
||||
FastPsfAccumulator acc = {0};
|
||||
if (fast_psf_accumulator_init(&acc, width, height, supersample,
|
||||
FAST_PSF_DEPOSIT_NEAREST, &psf, relative_tail,
|
||||
0.0, 1)) {
|
||||
fail("wraparound accumulator init");
|
||||
return;
|
||||
}
|
||||
const size_t count = (size_t)width * height * 3;
|
||||
double *hdr = calloc(count, sizeof *hdr);
|
||||
if (hdr == NULL) {
|
||||
fail("wraparound allocation");
|
||||
fast_psf_accumulator_destroy(&acc);
|
||||
return;
|
||||
}
|
||||
fast_psf_accumulator_deposit(&acc, 0.6, 0.6, (LinearRgb){1.0, 1.0, 1.0}, 1.0);
|
||||
if (fast_psf_accumulator_resolve(&acc, hdr, 2)) {
|
||||
fail("wraparound resolve");
|
||||
} else {
|
||||
const int far_x = width - 1, far_y = height - 1;
|
||||
const double ghost = hdr[3 * (far_y * width + far_x)];
|
||||
if (fabs(ghost) > 1e-15)
|
||||
fprintf(stderr, "FAIL: opposite-edge ghost value %.3g\n", ghost),
|
||||
++g_failures;
|
||||
if (hdr[0] <= 0.0)
|
||||
fail("impulse peak missing near the deposited corner");
|
||||
}
|
||||
/* Empty buffer: the same accumulator, after clearing, must not change HDR. */
|
||||
fast_psf_accumulator_clear(&acc);
|
||||
double background[3] = {0.25, 0.5, 0.75};
|
||||
for (size_t i = 0; i < count; ++i)
|
||||
hdr[i] = background[i % 3];
|
||||
if (fast_psf_accumulator_resolve(&acc, hdr, 2))
|
||||
fail("empty resolve");
|
||||
for (size_t i = 0; i < count; ++i) {
|
||||
if (hdr[i] != background[i % 3]) {
|
||||
fail("empty input changed the HDR framebuffer");
|
||||
break;
|
||||
}
|
||||
}
|
||||
free(hdr);
|
||||
fast_psf_accumulator_destroy(&acc);
|
||||
}
|
||||
|
||||
/* Channel separation: a pure-red impulse must leave green and blue at zero. */
|
||||
static void test_channel_isolation(void)
|
||||
{
|
||||
const PointSpreadFunction psf = default_psf();
|
||||
const int width = 20, height = 18;
|
||||
FastPsfAccumulator acc = {0};
|
||||
if (fast_psf_accumulator_init(&acc, width, height, 2,
|
||||
FAST_PSF_DEPOSIT_NEAREST, &psf, 1e-8, 0.0, 1)) {
|
||||
fail("channel isolation init");
|
||||
return;
|
||||
}
|
||||
const size_t count = (size_t)width * height * 3;
|
||||
double *hdr = calloc(count, sizeof *hdr);
|
||||
fast_psf_accumulator_deposit(&acc, 10.0, 9.0, (LinearRgb){1.0, 0.0, 0.0}, 1.0);
|
||||
if (fast_psf_accumulator_resolve(&acc, hdr, 2))
|
||||
fail("channel isolation resolve");
|
||||
for (size_t i = 0; i < count; i += 3) {
|
||||
if (hdr[i + 1] != 0.0 || hdr[i + 2] != 0.0) {
|
||||
fail("green/blue channel leaked into a pure-red impulse");
|
||||
break;
|
||||
}
|
||||
}
|
||||
free(hdr);
|
||||
fast_psf_accumulator_destroy(&acc);
|
||||
}
|
||||
|
||||
int main(void)
|
||||
{
|
||||
test_next_smooth_size();
|
||||
test_no_wraparound_and_empty();
|
||||
test_channel_isolation();
|
||||
|
||||
const int sizes[][2] = {{17, 13}, {16, 16}, {23, 31}, {33, 17}};
|
||||
const int supersamples[] = {1, 2, 3, 4};
|
||||
const FastPsfDeposit deposits[] = {FAST_PSF_DEPOSIT_NEAREST,
|
||||
FAST_PSF_DEPOSIT_BILINEAR};
|
||||
for (size_t s = 0; s < sizeof sizes / sizeof sizes[0]; ++s) {
|
||||
for (size_t n = 0; n < sizeof supersamples / sizeof supersamples[0]; ++n) {
|
||||
for (size_t d = 0; d < sizeof deposits / sizeof deposits[0]; ++d) {
|
||||
for (Scene scene = SCENE_CENTER_WHITE; scene <= SCENE_EMPTY;
|
||||
++scene) {
|
||||
char name[128];
|
||||
snprintf(name, sizeof name, "%dx%d N%d %s scene%d", sizes[s][0],
|
||||
sizes[s][1], supersamples[n],
|
||||
deposits[d] == FAST_PSF_DEPOSIT_NEAREST ? "near" : "bilin",
|
||||
(int)scene);
|
||||
if (run_compare(name, sizes[s][0], sizes[s][1], supersamples[n],
|
||||
deposits[d], scene, 0) != 0) {
|
||||
/* Keep going: collect all failures before summarizing. */
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
run_compare("prefilled background", 24, 20, 2, FAST_PSF_DEPOSIT_NEAREST,
|
||||
SCENE_MULTIPLE, 1);
|
||||
/* One axis (height) stays at the exact FFT size while the other grows. */
|
||||
run_compare("single-axis padding", 32, 8, 2, FAST_PSF_DEPOSIT_NEAREST,
|
||||
SCENE_CENTER_WHITE, 0);
|
||||
|
||||
if (g_failures != 0) {
|
||||
fprintf(stderr, "%d fast-PSF FFTW regression failure(s)\n", g_failures);
|
||||
return 1;
|
||||
}
|
||||
puts("fast PSF FFTW regressions passed");
|
||||
return 0;
|
||||
}
|
||||
Reference in new issue
Block a user