Files
GR-raytracing/tests/test_fast_psf_fftw.c
T
wyj 229f50cd86 Feat: Use FFTW linear convolution for CPU fast-mode PSF resolve
Replace the nested spatial global convolution in fast_psf_accumulator_resolve with a reusable double-precision FFTW linear convolution on the CPU PSF backend.

- Add the private src/fast_psf_fftw.{c,h} module: zero-padded R2C/C2R plans, cached kernel spectrum and planar scratch, exact 1/(Pwidth*Pheight) and 1/N^2 normalization, (R,R) crop, and additive HDR output.
- Keep the previous nested loops as fast_psf_accumulator_resolve_spatial_reference for tests/benchmarks only; it is not a runtime fallback.
- Cache the circular row spans on FastPsfAccumulator and report one-time plan, kernel transform, scratch, and per-frame stage timings.
- Require fftw3_omp for CPU builds; HIP and dummy builds do not link FFTW.
- Namespace test/helper binaries by spacetime and build tag, and reject make test / psf-capture for non-CPU backends.
- Add tests/test_fast_psf_fftw.c (FFTW versus spatial), tests/benchmark_fast_psf_fftw.c, an FFTW CLI smoke check, and the 2026-09-25 benchmark record.
2026-09-25 23:35:45 -04:00

379 lines
13 KiB
C

/* Dedicated FFTW-versus-spatial fast-mode convolution regression.
*
* Both resolvers run on the identical impulse buffer and must agree to double
* rounding. The spatial resolver is the production reference, not a fallback.
* Compile only for the CPU PSF backend. */
#ifndef FAST_PSF_FFTW
#error "test_fast_psf_fftw requires -DFAST_PSF_FFTW (PSF_BACKEND=cpu)"
#endif
#include "optics.h"
#include "fast_psf_fftw.h"
#include <limits.h>
#include <math.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#ifndef INT_MAX
#define INT_MAX 2147483647
#endif
static int g_failures = 0;
static void fail(const char *what)
{
fprintf(stderr, "FAIL: %s\n", what);
++g_failures;
}
static PointSpreadFunction default_psf(void)
{
PointSpreadFunction psf = {.fwhm_pixels = 2.7, .moffat_beta = 4.5};
return psf;
}
static uint64_t hash_bytes(const double *values, size_t count)
{
const unsigned char *bytes = (const unsigned char *)values;
const size_t nbytes = count * sizeof *values;
uint64_t hash = 1469598103934665603ULL;
for (size_t i = 0; i < nbytes; ++i) {
hash ^= bytes[i];
hash *= 1099511628211ULL;
}
return hash;
}
typedef enum {
SCENE_CENTER_WHITE,
SCENE_DISTINCT_RGB,
SCENE_EDGES,
SCENE_BOUNDARY_PHASES,
SCENE_MULTIPLE,
SCENE_DENSE_CELL,
SCENE_RANDOM,
SCENE_EMPTY,
} Scene;
static void fill_scene(FastPsfAccumulator *acc, Scene scene, int width,
int height)
{
const LinearRgb white = {1.0, 1.0, 1.0};
switch (scene) {
case SCENE_CENTER_WHITE:
fast_psf_accumulator_deposit(acc, 0.5 * width, 0.5 * height, white, 1.0);
break;
case SCENE_DISTINCT_RGB:
fast_psf_accumulator_deposit(acc, 0.5 * width, 0.5 * height,
(LinearRgb){1.0, 0.25, 0.05}, 0.75);
break;
case SCENE_EDGES:
fast_psf_accumulator_deposit(acc, 0.3, 0.3, white, 1.0);
fast_psf_accumulator_deposit(acc, width - 0.7, 0.4, white, 1.0);
fast_psf_accumulator_deposit(acc, 0.5, height - 0.6, white, 1.0);
fast_psf_accumulator_deposit(acc, width - 0.4, height - 0.3, white, 1.0);
break;
case SCENE_BOUNDARY_PHASES:
fast_psf_accumulator_deposit(acc, 10.499, 12.499, white, 1.0);
fast_psf_accumulator_deposit(acc, 10.501, 12.501, white, 1.0);
break;
case SCENE_MULTIPLE:
fast_psf_accumulator_deposit(acc, 5.5, 5.5, (LinearRgb){1.0, 0.0, 0.0}, 0.5);
fast_psf_accumulator_deposit(acc, 12.25, 7.75, (LinearRgb){0.0, 1.0, 0.0}, 0.25);
fast_psf_accumulator_deposit(acc, 8.1, 14.9, (LinearRgb){0.0, 0.0, 1.0}, 1.5);
break;
case SCENE_DENSE_CELL:
for (int i = 0; i < 32; ++i)
fast_psf_accumulator_deposit(acc, 9.0 + 0.01 * i, 9.0 + 0.013 * i,
(LinearRgb){1.0, 0.5, 0.25}, 0.05);
break;
case SCENE_RANDOM: {
uint64_t state = 0x9e3779b97f4a7c15ULL;
const size_t count = (size_t)acc->supersampled_width *
acc->supersampled_height * 3;
for (size_t i = 0; i < count; ++i) {
state = state * 6364136223846793005ULL + 1442695040888963407ULL;
acc->buffer[i] += (double)(state >> 40) / (double)(1ULL << 24) - 0.5;
}
break;
}
case SCENE_EMPTY:
break;
}
}
static int run_compare(const char *name, int width, int height,
int supersample, FastPsfDeposit deposit, Scene scene,
int prefill)
{
FastPsfAccumulator acc = {0};
const PointSpreadFunction psf = default_psf();
const double relative_tail = 1e-8;
if (fast_psf_accumulator_init(&acc, width, height, supersample, deposit, &psf,
relative_tail, 0.0, 1)) {
fprintf(stderr, "FAIL: %s: accumulator init failed\n", name);
++g_failures;
return -1;
}
if (!acc.fftw_enabled) {
fprintf(stderr, "FAIL: %s: FFTW path not enabled\n", name);
++g_failures;
fast_psf_accumulator_destroy(&acc);
return -1;
}
const size_t hdr_count = (size_t)width * height * 3;
const size_t buffer_count = (size_t)acc.supersampled_width *
acc.supersampled_height * 3;
double *fftw_hdr = malloc(hdr_count * sizeof *fftw_hdr);
double *spatial_hdr = malloc(hdr_count * sizeof *spatial_hdr);
if (fftw_hdr == NULL || spatial_hdr == NULL) {
fprintf(stderr, "FAIL: %s: HDR allocation failed\n", name);
++g_failures;
free(fftw_hdr);
free(spatial_hdr);
fast_psf_accumulator_destroy(&acc);
return -1;
}
for (size_t i = 0; i < hdr_count; ++i)
fftw_hdr[i] = spatial_hdr[i] = prefill ? 0.25 : 0.0;
fill_scene(&acc, scene, width, height);
const uint64_t buffer_before = hash_bytes(acc.buffer, buffer_count);
if (fast_psf_accumulator_resolve(&acc, fftw_hdr, 4)) {
fprintf(stderr, "FAIL: %s: FFTW resolve failed\n", name);
++g_failures;
goto cleanup;
}
if (fast_psf_accumulator_resolve_spatial_reference(&acc, spatial_hdr, 4)) {
fprintf(stderr, "FAIL: %s: spatial reference resolve failed\n", name);
++g_failures;
goto cleanup;
}
if (hash_bytes(acc.buffer, buffer_count) != buffer_before) {
fprintf(stderr, "FAIL: %s: resolve mutated the impulse buffer\n", name);
++g_failures;
goto cleanup;
}
double peak = 0.0, max_abs = 0.0, max_rel = 0.0, sum_sq = 0.0;
double flux_fftw[3] = {0.0, 0.0, 0.0};
double flux_spatial[3] = {0.0, 0.0, 0.0};
size_t nan_count = 0;
for (size_t i = 0; i < hdr_count; ++i) {
const double reference = spatial_hdr[i];
const double error = fabs(fftw_hdr[i] - reference);
peak = fmax(peak, fabs(reference));
max_abs = fmax(max_abs, error);
sum_sq += error * error;
if (!isfinite(fftw_hdr[i]) || !isfinite(reference))
++nan_count;
flux_fftw[i % 3] += fftw_hdr[i];
flux_spatial[i % 3] += reference;
}
/* Mixed absolute/relative acceptance: the absolute floor covers the tiny
* Moffat far-wing samples where the FFT and the direct sum disagree only by
* roundoff, while the relative term checks significant samples. A crop,
* wrap, channel, or normalization bug produces O(1) errors well above both. */
const double rms = sqrt(sum_sq / (double)hdr_count);
const double abs_tol = 1e-10 * fmax(1.0, peak);
const double rel_tol = 1e-9;
const double rel_floor = 1e-6 * fmax(peak, 1.0);
double max_violation = 0.0;
for (size_t i = 0; i < hdr_count; ++i) {
const double reference = spatial_hdr[i];
const double error = fabs(fftw_hdr[i] - reference);
max_violation =
fmax(max_violation, error - (abs_tol + rel_tol * fabs(reference)));
if (fabs(reference) > rel_floor)
max_rel = fmax(max_rel, error / fabs(reference));
}
int worst = -1;
double worst_error = 0.0;
for (size_t i = 0; i < hdr_count; ++i) {
const double error = fabs(fftw_hdr[i] - spatial_hdr[i]);
if (error > worst_error) {
worst_error = error;
worst = (int)(i / 3);
}
}
double flux_rel = 0.0;
for (int c = 0; c < 3; ++c) {
const double denom = fmax(fabs(flux_spatial[c]), 1e-30);
flux_rel = fmax(flux_rel, fabs(flux_fftw[c] - flux_spatial[c]) / denom);
}
if (nan_count != 0 || max_violation > 0.0 || max_rel > rel_tol ||
flux_rel > 1e-10) {
fprintf(stderr,
"FAIL: %s: peak=%.6g max_abs=%.3g (tol %.3g) max_rel=%.3g "
"rms=%.3g flux_rel=%.3g nan=%zu worst_pixel=%d\n",
name, peak, max_abs, abs_tol, max_rel, rms, flux_rel, nan_count,
worst);
++g_failures;
} else {
printf("ok %-28s peak=%.4g max_abs=%.3g max_rel=%.3g rms=%.3g\n", name,
peak, max_abs, max_rel, rms);
}
cleanup:
free(fftw_hdr);
free(spatial_hdr);
fast_psf_accumulator_destroy(&acc);
return g_failures == 0 ? 0 : -1;
}
static void test_next_smooth_size(void)
{
struct {
size_t input;
long expected; /* -1 = must fail */
} cases[] = {
{1, 1}, {2, 2}, {3, 3}, {4, 4}, {5, 5}, {6, 6},
{7, 7}, {8, 8}, {9, 9}, {11, 12}, {13, 14}, {15, 15},
{121, 125}, {127, 128}, {200, 200}, {241, 243},
{(size_t)INT_MAX + 1, -1},
{0, -1},
};
for (size_t i = 0; i < sizeof cases / sizeof cases[0]; ++i) {
size_t out = 0;
const int rc = fast_psf_fftw_next_smooth_size(cases[i].input, &out);
if (cases[i].expected < 0) {
if (rc == 0)
fprintf(stderr, "FAIL: next_smooth_size(%zu) unexpectedly returned %zu\n",
cases[i].input, out), ++g_failures;
} else if (rc != 0 || out != (size_t)cases[i].expected) {
fprintf(stderr, "FAIL: next_smooth_size(%zu) = rc %d, %zu (want %ld)\n",
cases[i].input, rc, out, cases[i].expected);
++g_failures;
}
}
size_t out = 0;
if (fast_psf_fftw_next_smooth_size((size_t)1 << 30, &out) != 0 ||
out > (size_t)INT_MAX)
fail("next_smooth_size near INT_MAX did not return a valid FFT size");
}
/* A single impulse must not wrap to the opposite edge, and an empty buffer must
* leave the framebuffer untouched. */
static void test_no_wraparound_and_empty(void)
{
PointSpreadFunction psf = default_psf();
/* A compact kernel keeps the image much wider than the support so a genuine
* circular-wrap bug would show up as an opposite-edge ghost. */
psf.fwhm_pixels = 0.02;
const double relative_tail = 1e-8;
const int width = 12, height = 12, supersample = 2;
FastPsfAccumulator acc = {0};
if (fast_psf_accumulator_init(&acc, width, height, supersample,
FAST_PSF_DEPOSIT_NEAREST, &psf, relative_tail,
0.0, 1)) {
fail("wraparound accumulator init");
return;
}
const size_t count = (size_t)width * height * 3;
double *hdr = calloc(count, sizeof *hdr);
if (hdr == NULL) {
fail("wraparound allocation");
fast_psf_accumulator_destroy(&acc);
return;
}
fast_psf_accumulator_deposit(&acc, 0.6, 0.6, (LinearRgb){1.0, 1.0, 1.0}, 1.0);
if (fast_psf_accumulator_resolve(&acc, hdr, 2)) {
fail("wraparound resolve");
} else {
const int far_x = width - 1, far_y = height - 1;
const double ghost = hdr[3 * (far_y * width + far_x)];
if (fabs(ghost) > 1e-15)
fprintf(stderr, "FAIL: opposite-edge ghost value %.3g\n", ghost),
++g_failures;
if (hdr[0] <= 0.0)
fail("impulse peak missing near the deposited corner");
}
/* Empty buffer: the same accumulator, after clearing, must not change HDR. */
fast_psf_accumulator_clear(&acc);
double background[3] = {0.25, 0.5, 0.75};
for (size_t i = 0; i < count; ++i)
hdr[i] = background[i % 3];
if (fast_psf_accumulator_resolve(&acc, hdr, 2))
fail("empty resolve");
for (size_t i = 0; i < count; ++i) {
if (hdr[i] != background[i % 3]) {
fail("empty input changed the HDR framebuffer");
break;
}
}
free(hdr);
fast_psf_accumulator_destroy(&acc);
}
/* Channel separation: a pure-red impulse must leave green and blue at zero. */
static void test_channel_isolation(void)
{
const PointSpreadFunction psf = default_psf();
const int width = 20, height = 18;
FastPsfAccumulator acc = {0};
if (fast_psf_accumulator_init(&acc, width, height, 2,
FAST_PSF_DEPOSIT_NEAREST, &psf, 1e-8, 0.0, 1)) {
fail("channel isolation init");
return;
}
const size_t count = (size_t)width * height * 3;
double *hdr = calloc(count, sizeof *hdr);
fast_psf_accumulator_deposit(&acc, 10.0, 9.0, (LinearRgb){1.0, 0.0, 0.0}, 1.0);
if (fast_psf_accumulator_resolve(&acc, hdr, 2))
fail("channel isolation resolve");
for (size_t i = 0; i < count; i += 3) {
if (hdr[i + 1] != 0.0 || hdr[i + 2] != 0.0) {
fail("green/blue channel leaked into a pure-red impulse");
break;
}
}
free(hdr);
fast_psf_accumulator_destroy(&acc);
}
int main(void)
{
test_next_smooth_size();
test_no_wraparound_and_empty();
test_channel_isolation();
const int sizes[][2] = {{17, 13}, {16, 16}, {23, 31}, {33, 17}};
const int supersamples[] = {1, 2, 3, 4};
const FastPsfDeposit deposits[] = {FAST_PSF_DEPOSIT_NEAREST,
FAST_PSF_DEPOSIT_BILINEAR};
for (size_t s = 0; s < sizeof sizes / sizeof sizes[0]; ++s) {
for (size_t n = 0; n < sizeof supersamples / sizeof supersamples[0]; ++n) {
for (size_t d = 0; d < sizeof deposits / sizeof deposits[0]; ++d) {
for (Scene scene = SCENE_CENTER_WHITE; scene <= SCENE_EMPTY;
++scene) {
char name[128];
snprintf(name, sizeof name, "%dx%d N%d %s scene%d", sizes[s][0],
sizes[s][1], supersamples[n],
deposits[d] == FAST_PSF_DEPOSIT_NEAREST ? "near" : "bilin",
(int)scene);
if (run_compare(name, sizes[s][0], sizes[s][1], supersamples[n],
deposits[d], scene, 0) != 0) {
/* Keep going: collect all failures before summarizing. */
}
}
}
}
}
run_compare("prefilled background", 24, 20, 2, FAST_PSF_DEPOSIT_NEAREST,
SCENE_MULTIPLE, 1);
/* One axis (height) stays at the exact FFT size while the other grows. */
run_compare("single-axis padding", 32, 8, 2, FAST_PSF_DEPOSIT_NEAREST,
SCENE_CENTER_WHITE, 0);
if (g_failures != 0) {
fprintf(stderr, "%d fast-PSF FFTW regression failure(s)\n", g_failures);
return 1;
}
puts("fast PSF FFTW regressions passed");
return 0;
}