Feat: Add optional three-channel sensor bloom model

Add an opt-in, post-processing limited-response model applied to the
finished linear HDR before tone mapping. Each RGB channel is processed
independently and isotropically: overflow above E spreads to the eight
neighbours with a fixed 9-point stencil, while the rest is absorbed or
lost at the image boundary. The synchronous ping-pong update uses a
monotonic bounding box and a row-parallel, deterministic reduction; the
conservative round bound reserves fp guard rounds inside a 4096 hard
limit and fails before touching HDR when exceeded.

Expose --sensor-bloom-limit E and --sensor-bloom-transfer e (both
required together, default disabled), validate them before expensive
initialization, and route every output path through the same hook in
write_frame_outputs: raw FITS first, bloom, tone-mapped PNG/PPM, then the
mesh overlay. The raw --hdr-output FITS therefore stays pre-bloom.

Add a standalone unit test (stencil, boundary loss, cascade reference,
symmetry, thread determinism, convergence limits, validation, allocation
failure), CLI integration and regression coverage, an isolated
sensor-bloom-bench target, and document the model in the design, usage,
README, and build docs.
This commit is contained in:
wyj committed 2026-09-27 04:39:38 -04:00
1 parent 85fce1bdeb
commit 9cd933d1f8
12 files changed
+1280 -7

No files matched your search

+121
View File
@@ -0,0 +1,121 @@
/* Isolated benchmark for the sensor-bloom model in src/sensor_bloom.c.
*
* This performs no ray tracing, catalog lookup, or PSF splatting; it fills a
* synthetic HDR framebuffer and times the post-processing model alone. The
* default 512x288 size keeps a full sweep bounded, and the optional size
* arguments are clamped so the benchmark cannot be turned into a production
* render. It is a non-default target and is not part of `make test`. */
#include "sensor_bloom.h"
#include <math.h>
#include <omp.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#define DEFAULT_WIDTH 512
#define DEFAULT_HEIGHT 288
#define MAX_DIMENSION 1024
#define RESPONSE_LIMIT 1.0
static uint32_t next_random(uint32_t *state) {
*state = *state * 1664525u + 1013904223u;
return *state;
}
static uint64_t checksum(const double *hdr, size_t count) {
const unsigned char *bytes = (const unsigned char *)hdr;
uint64_t hash = 1469598103934665603ull;
for (size_t i = 0; i < count * sizeof(double); ++i) {
hash ^= bytes[i];
hash *= 1099511628211ull;
}
return hash;
}
static void fill_central(double *hdr, int width, int height) {
const size_t index = ((size_t)(height / 2) * width + width / 2) * 3;
hdr[index + 0] = 20.0;
hdr[index + 1] = 20.0;
hdr[index + 2] = 20.0;
}
static void fill_scattered(double *hdr, int width, int height) {
int placed = 0;
for (int y = 0; y < height && placed < 64; y += height > 8 ? height / 8 : 1) {
for (int x = 0; x < width && placed < 64; x += width > 8 ? width / 8 : 1) {
const size_t index = ((size_t)y * width + x) * 3;
hdr[index + 0] = 20.0;
hdr[index + 1] = 16.0;
hdr[index + 2] = 12.0;
++placed;
}
}
}
static void fill_random_fraction(double *hdr, int width, int height) {
const size_t count = (size_t)width * height * 3;
uint32_t state = 20260927u;
for (size_t i = 0; i < count; ++i)
hdr[i] = 1.5 + (double)(next_random(&state) % 2500u) / 1000.0;
for (size_t i = 0; i < count; ++i)
if (next_random(&state) % 100u >= 5u)
hdr[i] = 0.0;
}
typedef void (*FillFunction)(double *, int, int);
static void run_case(const char *name, FillFunction fill, int width, int height,
double transfer) {
const size_t count = (size_t)width * height * 3;
double *hdr = calloc(count, sizeof *hdr);
if (hdr == NULL) {
fprintf(stderr, "allocation failed for %s\n", name);
exit(EXIT_FAILURE);
}
fill(hdr, width, height);
const SensorBloomSettings settings = {RESPONSE_LIMIT, transfer};
SensorBloomStats stats;
if (sensor_bloom_apply(hdr, width, height, &settings, &stats)) {
fprintf(stderr, "%s e=%.2f failed\n", name, transfer);
free(hdr);
exit(EXIT_FAILURE);
}
/* Report the model's own guard-inclusive conservative cap, not a duplicate
* of the bound formula. */
printf("%-10s e=%.2f size=%dx%d saturated=%zu predicted=%zu actual=%zu "
"bbox=%.4f elapsed=%.6fs checksum=%016llx\n",
name, transfer, width, height, stats.initially_saturated_channels,
stats.predicted_iterations, stats.iterations, stats.bbox_coverage,
stats.elapsed_seconds, (unsigned long long)checksum(hdr, count));
free(hdr);
}
int main(int argc, char **argv) {
int width = DEFAULT_WIDTH, height = DEFAULT_HEIGHT;
if (argc >= 2)
width = atoi(argv[1]);
if (argc >= 3)
height = atoi(argv[2]);
if (width < 1)
width = 1;
if (height < 1)
height = 1;
if (width > MAX_DIMENSION)
width = MAX_DIMENSION;
if (height > MAX_DIMENSION)
height = MAX_DIMENSION;
const int workers = omp_get_max_threads();
printf("sensor-bloom benchmark: size=%dx%d threads=%d limit=%.3f\n",
width, height, workers, RESPONSE_LIMIT);
const double transfers[] = {0.0, 0.5, 0.9};
for (size_t e = 0; e < sizeof transfers / sizeof transfers[0]; ++e) {
run_case("central", fill_central, width, height, transfers[e]);
run_case("scattered", fill_scattered, width, height, transfers[e]);
run_case("random5pct", fill_random_fraction, width, height, transfers[e]);
}
return EXIT_SUCCESS;
}
+89
View File
@@ -8,6 +8,28 @@ import sys
import tempfile
import zlib
def fits_max(path):
"""Largest positive sample in the renderer's three-plane float FITS."""
data = path.read_bytes()
cards, offset = [], 0
while True:
block = data[offset:offset + 2880]
assert len(block) == 2880, f'truncated FITS header: {path}'
offset += 2880
cards.extend(block[i:i + 80] for i in range(0, 2880, 80))
if any(card.startswith(b'END') for card in cards[-36:]):
break
values = {}
for card in cards:
if card[8:10] == b'= ':
values[card[:8].decode().strip()] = card[10:30].decode().strip()
shape = tuple(int(values[f'NAXIS{axis}']) for axis in (1, 2, 3))
count = shape[0] * shape[1] * shape[2]
payload = data[offset:offset + count * 4]
assert len(payload) == count * 4, f'truncated FITS payload: {path}'
return max(struct.unpack(f'>{count}f', payload))
BUILD = Path(sys.argv[1] if len(sys.argv) > 1 else 'build/Release').resolve()
TESTDIR = Path(sys.argv[2]).resolve() if len(sys.argv) > 2 else BUILD
ENV = dict(os.environ, OMP_NUM_THREADS='4')
@@ -89,6 +111,8 @@ with tempfile.TemporaryDirectory(prefix='gr-camera-cli-') as directory:
for option in ('--observer-position', '--observer-velocity', '--camera-roll-deg'):
assert option in help_text
assert '--tone-map' in help_text and '--tone-map-p' in help_text
assert '--sensor-bloom-limit' in help_text
assert '--sensor-bloom-transfer' in help_text
assert '--observer-inward-speed' not in help_text
assert '_mesh.' in help_text, help_text
# The synthetic grid is calibrated for the renderer's default exposure.
@@ -205,6 +229,17 @@ with tempfile.TemporaryDirectory(prefix='gr-camera-cli-') as directory:
(['--tone-map-p', 'nan'], None),
(['--tone-map-p', 'inf'], None),
(['--tone-map', 'reinhard', '--tone-map-p', 2], 'applies only'),
(['--sensor-bloom-limit', 1], 'specified together'),
(['--sensor-bloom-transfer', 0.5], 'specified together'),
(['--sensor-bloom-limit', 0], None),
(['--sensor-bloom-limit', -1], None),
(['--sensor-bloom-limit', 'nan'], None),
(['--sensor-bloom-limit', 'inf'], None),
(['--sensor-bloom-limit', 1, '--sensor-bloom-transfer', -0.1], None),
(['--sensor-bloom-limit', 1, '--sensor-bloom-transfer', 1], None),
(['--sensor-bloom-limit', 1, '--sensor-bloom-transfer', 1.5], None),
(['--sensor-bloom-limit', 1, '--sensor-bloom-transfer', 'nan'], None),
(['--sensor-bloom-limit', 1, '--sensor-bloom-transfer', 'inf'], None),
]
if backend == 'schwarzschild':
errors += [(['--observer-position', 1.5, 0, 0, '--observer-velocity', -0.5, 0, 0], 'capture cutoff'),
@@ -279,6 +314,60 @@ with tempfile.TemporaryDirectory(prefix='gr-camera-cli-') as directory:
assert image_payload(hdr_mesh_output) == baseline
assert image_payload(tmp / f'{backend}_hdr_mesh_mesh.{ext}') != baseline
# Sensor-bloom integration runs in every build. With linear HDR the
# response limit is derived from the rendered peak; without it the
# fixture's calibrated exposure puts the display shoulder near 1.0,
# which is guaranteed to saturate this scene, so the default ENABLE_HDR=0
# suite still exercises the output-pipeline hook. The limit is placed
# well below the peak so the clamped region falls in the tone map's
# sensitive range and the display comparison stays discriminating.
if hdr_available:
base_peak = fits_max(base_fits)
assert base_peak > 0.0
bloom_limit = base_peak / 100.0
else:
bloom_limit = 1.0
hdr_args = ['--hdr-output'] if hdr_available else []
bloom_output = tmp / f'{backend}_bloom.{ext}'
bloom_run = run(binary, *common, *hdr_args,
'--sensor-bloom-limit', bloom_limit,
'--sensor-bloom-transfer', 0.5,
'--output', bloom_output)
assert image_payload(bloom_output) != baseline
assert 'Sensor bloom:' in bloom_run.stderr, bloom_run.stderr
report = bloom_run.stderr.split('Sensor bloom:', 1)[1].splitlines()[0]
fields = dict(token.split('=', 1) for token in report.split() if '=' in token)
assert int(fields['saturated']) > 0, report
assert int(fields['iterations'].split('/')[0]) >= 1, report
assert float(fields['peak']) >= bloom_limit, report
if hdr_available:
# The raw FITS is written before the model runs, so it stays
# byte-identical to the baseline even though the display changes.
bloom_fits = tmp / f'{backend}_bloom_HDR.fits'
assert bloom_fits.exists(), bloom_run.stderr
diff = subprocess.run([sys.executable, str(FITSDIFF), str(base_fits),
str(bloom_fits)], capture_output=True, text=True)
assert diff.returncode == 0, diff.stdout + diff.stderr
assert 'mismatches=0 max_abs=0 max_rel=0' in diff.stdout, diff.stdout
# The mesh overlay is drawn on the already-bloomed frame, so the primary
# image is unchanged by --draw-mesh and the diagnostic lines do not feed
# back into the overflow model.
bloom_mesh_output = tmp / f'{backend}_bloom_mesh.{ext}'
run(binary, *common, *hdr_args, '--sensor-bloom-limit', bloom_limit,
'--sensor-bloom-transfer', 0.5, '--draw-mesh',
'--output', bloom_mesh_output)
assert image_payload(bloom_mesh_output) == image_payload(bloom_output)
bloom_mesh_sibling = tmp / f'{backend}_bloom_mesh_mesh.{ext}'
assert bloom_mesh_sibling.exists()
assert image_payload(bloom_mesh_sibling) != image_payload(bloom_output)
if hdr_available:
bloom_mesh_fits = tmp / f'{backend}_bloom_mesh_HDR.fits'
diff = subprocess.run([sys.executable, str(FITSDIFF), str(base_fits),
str(bloom_mesh_fits)], capture_output=True, text=True)
assert diff.returncode == 0, diff.stdout + diff.stderr
assert 'mismatches=0 max_abs=0 max_rel=0' in diff.stdout, diff.stdout
if backend == 'schwarzschild':
# Two inward-looking free-fall samples at r=6.2696 and r=3.1593.
# At 16:9 the latter frame finishes in generation 0, while the
+568
View File
@@ -0,0 +1,568 @@
/* Regression tests for the standalone sensor-bloom model in
* src/sensor_bloom.c.
*
* These link only the model: no catalog, ray tracing, image writer, FFTW, or
* GPU is involved, so they build in every ENABLE_HDR/PSF_BACKEND
* configuration. They call the production entry point and compare it against
* an independently written dense synchronous reference where a closed form is
* not simpler. */
#include "sensor_bloom.h"
#include <math.h>
#include <omp.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/resource.h>
#include <sys/wait.h>
#include <unistd.h>
/* The scratch malloc-failure path is covered by the ordinary non-ASan run of
* `allocation_failure_is_reported()` below. AddressSanitizer intercepts
* allocation and its runtime cannot mmap under a forced RLIMIT_AS, so that
* probe is skipped in ASan builds rather than being made to pass; the
* dimension-overflow validation is a separate check and does not substitute
* for it. */
#if defined(__SANITIZE_ADDRESS__)
#define SB_HAVE_ASAN 1
#elif defined(__has_feature)
#if __has_feature(address_sanitizer)
#define SB_HAVE_ASAN 1
#endif
#endif
static int failures = 0;
static void check(int condition, const char *message) {
if (!condition) {
fprintf(stderr, "FAIL: %s\n", message);
++failures;
}
}
static int close_to(double value, double expected, double tolerance) {
return fabs(value - expected) <= tolerance;
}
static double *alloc_rgb(int width, int height) {
return calloc((size_t)width * height * 3, sizeof(double));
}
static size_t at(int width, int x, int y, int c) {
return ((size_t)y * width + x) * 3 + (size_t)c;
}
static double sample(const double *rgb, int width, int x, int y, int c) {
return rgb[at(width, x, y, c)];
}
static void set(double *rgb, int width, int x, int y, int c, double value) {
rgb[at(width, x, y, c)] = value;
}
static int all_finite_and_bounded(const double *rgb, size_t count,
double limit) {
for (size_t i = 0; i < count; ++i)
if (!isfinite(rgb[i]) || rgb[i] > limit)
return 0;
return 1;
}
/* Independently written dense synchronous reference: every round reads the
* whole previous state and writes the whole next state, with the same 9-point
* weights. Only used to cross-check cascades. */
static double reference_overflow(const double *src, size_t index, double limit) {
const double difference = src[index] - limit;
return difference > 0.0 ? difference : 0.0;
}
static void reference_bloom(double *hdr, int width, int height, double limit,
double transfer) {
const double axial = 4.0 / 20.0, diagonal = 1.0 / 20.0;
const double tolerance = 1e-9 * limit;
const size_t count = (size_t)width * height * 3;
double *first = malloc(count * sizeof *first);
double *second = malloc(count * sizeof *second);
memcpy(first, hdr, count * sizeof *first);
memcpy(second, hdr, count * sizeof *second);
double *src = first, *dst = second;
for (int round = 0; round < 20000; ++round) {
double next_max = 0.0;
for (int y = 0; y < height; ++y) {
for (int x = 0; x < width; ++x) {
for (int c = 0; c < 3; ++c) {
const size_t index = at(width, x, y, c);
const double value = src[index];
double incoming = 0.0;
if (x > 0)
incoming += axial * reference_overflow(src, at(width, x - 1, y, c), limit);
if (x + 1 < width)
incoming += axial * reference_overflow(src, at(width, x + 1, y, c), limit);
if (y > 0)
incoming += axial * reference_overflow(src, at(width, x, y - 1, c), limit);
if (y + 1 < height)
incoming += axial * reference_overflow(src, at(width, x, y + 1, c), limit);
if (x > 0 && y > 0)
incoming += diagonal * reference_overflow(src, at(width, x - 1, y - 1, c), limit);
if (x + 1 < width && y > 0)
incoming += diagonal * reference_overflow(src, at(width, x + 1, y - 1, c), limit);
if (x > 0 && y + 1 < height)
incoming += diagonal * reference_overflow(src, at(width, x - 1, y + 1, c), limit);
if (x + 1 < width && y + 1 < height)
incoming += diagonal * reference_overflow(src, at(width, x + 1, y + 1, c), limit);
const double next = (value < limit ? value : limit) + transfer * incoming;
dst[index] = next;
const double overflow = next - limit;
if (overflow > next_max)
next_max = overflow;
}
}
}
double *swap = src;
src = dst;
dst = swap;
if (next_max <= tolerance)
break;
}
for (size_t i = 0; i < count; ++i)
if (src[i] > limit)
src[i] = limit;
memcpy(hdr, src, count * sizeof *hdr);
free(first);
free(second);
}
static void test_no_saturation(void) {
const int width = 8, height = 6;
const size_t count = (size_t)width * height * 3;
double *hdr = alloc_rgb(width, height);
for (size_t i = 0; i < count; ++i)
hdr[i] = 0.25 * (double)(i % 5);
double *original = malloc(count * sizeof *original);
memcpy(original, hdr, count * sizeof *original);
const SensorBloomSettings settings = {1.0, 0.5};
SensorBloomStats stats;
check(sensor_bloom_apply(hdr, width, height, &settings, &stats) == 0,
"no-saturation apply succeeds");
check(memcmp(hdr, original, count * sizeof *hdr) == 0,
"no-saturation buffer is byte-for-byte unchanged");
check(stats.iterations == 0, "no-saturation uses zero iterations");
check(stats.predicted_iterations == 0,
"no-saturation predicts zero iterations");
check(stats.initially_saturated_channels == 0,
"no-saturation reports no saturated channel");
check(stats.final_clamped_channels == 0,
"no-saturation clamps nothing");
free(original);
free(hdr);
}
static void test_zero_transfer(void) {
const int width = 5, height = 5;
double *hdr = alloc_rgb(width, height);
set(hdr, width, 2, 2, 0, 1.7);
const SensorBloomSettings settings = {1.0, 0.0};
SensorBloomStats stats;
check(sensor_bloom_apply(hdr, width, height, &settings, &stats) == 0,
"zero-transfer apply succeeds");
check(close_to(sample(hdr, width, 2, 2, 0), 1.0, 1e-15),
"zero-transfer clamps the saturated channel to E");
check(sample(hdr, width, 1, 2, 0) == 0.0 && sample(hdr, width, 3, 2, 0) == 0.0,
"zero-transfer leaves axial neighbours unchanged");
check(sample(hdr, width, 1, 1, 0) == 0.0 && sample(hdr, width, 3, 3, 0) == 0.0,
"zero-transfer leaves diagonal neighbours unchanged");
check(close_to(stats.absorbed_signal, 0.7, 1e-15),
"zero-transfer absorbs the full overflow");
check(stats.boundary_loss == 0.0, "zero-transfer loses nothing at the boundary");
check(stats.iterations >= 1, "zero-transfer iterates at least once");
free(hdr);
}
static void test_single_impulse_stencil(void) {
const int width = 5, height = 5;
const double limit = 1.0, transfer = 0.5;
double *hdr = alloc_rgb(width, height);
set(hdr, width, 2, 2, 0, 1.8);
set(hdr, width, 2, 2, 1, 1.5);
set(hdr, width, 2, 2, 2, 1.2);
const SensorBloomSettings settings = {limit, transfer};
SensorBloomStats stats;
check(sensor_bloom_apply(hdr, width, height, &settings, &stats) == 0,
"single-impulse apply succeeds");
check(sample(hdr, width, 2, 2, 0) == limit &&
sample(hdr, width, 2, 2, 1) == limit &&
sample(hdr, width, 2, 2, 2) == limit,
"single-impulse centre sits exactly at E");
const double red_axis = transfer * 0.8 * (4.0 / 20.0);
const double red_diagonal = transfer * 0.8 * (1.0 / 20.0);
const double green_axis = transfer * 0.5 * (4.0 / 20.0);
const double blue_axis = transfer * 0.2 * (4.0 / 20.0);
check(close_to(sample(hdr, width, 1, 2, 0), red_axis, 1e-15) &&
close_to(sample(hdr, width, 3, 2, 0), red_axis, 1e-15) &&
close_to(sample(hdr, width, 2, 1, 0), red_axis, 1e-15) &&
close_to(sample(hdr, width, 2, 3, 0), red_axis, 1e-15),
"axial neighbours receive e * D * 4/20");
check(close_to(sample(hdr, width, 1, 1, 0), red_diagonal, 1e-15) &&
close_to(sample(hdr, width, 3, 3, 0), red_diagonal, 1e-15),
"diagonal neighbours receive e * D * 1/20");
check(close_to(sample(hdr, width, 1, 2, 1), green_axis, 1e-15) &&
close_to(sample(hdr, width, 1, 2, 2), blue_axis, 1e-15),
"RGB channels transfer independently");
check(sample(hdr, width, 0, 0, 0) == 0.0 && sample(hdr, width, 4, 4, 2) == 0.0,
"far pixels stay untouched");
free(hdr);
}
static void test_boundary_loss(void) {
const int width = 5, height = 5;
const double limit = 1.0, transfer = 0.5;
double *hdr = alloc_rgb(width, height);
set(hdr, width, 0, 0, 0, 1.8);
const SensorBloomSettings settings = {limit, transfer};
SensorBloomStats stats;
check(sensor_bloom_apply(hdr, width, height, &settings, &stats) == 0,
"corner apply succeeds");
const double overflow = 0.8;
check(close_to(sample(hdr, width, 1, 0, 0), transfer * overflow * 0.2, 1e-15) &&
close_to(sample(hdr, width, 0, 1, 0), transfer * overflow * 0.2, 1e-15),
"in-image axial propagation at a corner uses the fixed weight");
check(close_to(sample(hdr, width, 1, 1, 0), transfer * overflow * 0.05, 1e-15),
"in-image diagonal propagation at a corner uses the fixed weight");
check(close_to(stats.boundary_loss,
transfer * overflow * (2.0 * 0.2 + 3.0 * 0.05), 1e-15),
"off-image weight is counted as boundary loss without renormalization");
check(close_to(stats.absorbed_signal, (1.0 - transfer) * overflow, 1e-15),
"absorbed signal is (1 - e) * D");
free(hdr);
}
static void test_cascade_against_reference(void) {
const int width = 9, height = 9;
const double limit = 0.5, transfer = 0.6;
const size_t count = (size_t)width * height * 3;
double *hdr = alloc_rgb(width, height);
set(hdr, width, 4, 4, 0, 10.0);
set(hdr, width, 4, 4, 1, 4.0);
set(hdr, width, 2, 6, 2, 3.0);
set(hdr, width, 7, 1, 0, 1.5);
double *expected = malloc(count * sizeof *expected);
memcpy(expected, hdr, count * sizeof *expected);
reference_bloom(expected, width, height, limit, transfer);
const SensorBloomSettings settings = {limit, transfer};
SensorBloomStats stats;
check(sensor_bloom_apply(hdr, width, height, &settings, &stats) == 0,
"cascade apply succeeds");
check(stats.iterations >= 2, "cascade needs at least two rounds");
double max_difference = 0.0;
for (size_t i = 0; i < count; ++i)
max_difference = fmax(max_difference, fabs(hdr[i] - expected[i]));
check(max_difference < 1e-12, "cascade matches the dense synchronous reference");
check(all_finite_and_bounded(hdr, count, limit),
"cascade output is finite and at most E");
free(expected);
free(hdr);
}
static void rotate_90(const double *src, double *dst, int size) {
for (int y = 0; y < size; ++y)
for (int x = 0; x < size; ++x)
for (int c = 0; c < 3; ++c)
set(dst, size, size - 1 - y, x, c, sample(src, size, x, y, c));
}
static void test_symmetry(void) {
const int size = 9;
const double limit = 1.0, transfer = 0.5;
double *hdr = alloc_rgb(size, size);
set(hdr, size, 4, 4, 0, 10.0);
const SensorBloomSettings settings = {limit, transfer};
SensorBloomStats stats;
check(sensor_bloom_apply(hdr, size, size, &settings, &stats) == 0,
"symmetry apply succeeds");
check(sample(hdr, size, 3, 4, 0) == sample(hdr, size, 5, 4, 0) &&
sample(hdr, size, 3, 4, 0) == sample(hdr, size, 4, 3, 0) &&
sample(hdr, size, 3, 4, 0) == sample(hdr, size, 4, 5, 0),
"four axial neighbours of a centred point are equal");
check(sample(hdr, size, 3, 3, 0) == sample(hdr, size, 5, 3, 0) &&
sample(hdr, size, 3, 3, 0) == sample(hdr, size, 3, 5, 0) &&
sample(hdr, size, 3, 3, 0) == sample(hdr, size, 5, 5, 0),
"four diagonal neighbours of a centred point are equal");
/* 90-degree rotational covariance: rotating the input then blooming must
* equal blooming then rotating the output. This starts from a fresh,
* asymmetric and still-saturated input; the already-bloomed centre-impulse
* buffer above has no overflow and would only exercise the no-op path. */
const size_t count = (size_t)size * size * 3;
double *input = alloc_rgb(size, size);
set(input, size, 2, 5, 0, 30.0);
set(input, size, 6, 3, 1, 8.0);
set(input, size, 4, 1, 2, 5.0);
set(input, size, 7, 7, 0, 4.0);
double *bloomed_input = malloc(count * sizeof *bloomed_input);
double *rotated_input = alloc_rgb(size, size);
rotate_90(input, rotated_input, size);
double *bloomed_rotated = malloc(count * sizeof *bloomed_rotated);
memcpy(bloomed_input, input, count * sizeof *bloomed_input);
memcpy(bloomed_rotated, rotated_input, count * sizeof *bloomed_rotated);
check(sensor_bloom_apply(bloomed_input, size, size, &settings, &stats) == 0 &&
stats.initially_saturated_channels > 0,
"covariance input is saturated and blooms");
check(sensor_bloom_apply(bloomed_rotated, size, size, &settings, &stats) == 0,
"rotated covariance input blooms");
double *expected = alloc_rgb(size, size);
rotate_90(bloomed_input, expected, size);
double max_difference = 0.0;
for (size_t i = 0; i < count; ++i)
max_difference = fmax(max_difference, fabs(bloomed_rotated[i] - expected[i]));
check(max_difference < 1e-9, "model is 90-degree rotation covariant");
free(expected);
free(bloomed_rotated);
free(rotated_input);
free(bloomed_input);
free(input);
free(hdr);
}
static uint32_t next_random(uint32_t *state) {
*state = *state * 1664525u + 1013904223u;
return *state;
}
static void test_thread_determinism(void) {
const int width = 40, height = 30;
const size_t count = (size_t)width * height * 3;
double *hdr = alloc_rgb(width, height);
uint32_t state = 12345u;
for (size_t i = 0; i < count; ++i)
hdr[i] = (double)(next_random(&state) % 3000u) / 1000.0;
double *first = malloc(count * sizeof *first);
double *second = malloc(count * sizeof *second);
memcpy(first, hdr, count * sizeof *first);
memcpy(second, hdr, count * sizeof *second);
const SensorBloomSettings settings = {1.5, 0.5};
SensorBloomStats stats_one, stats_four;
omp_set_num_threads(1);
check(sensor_bloom_apply(first, width, height, &settings, &stats_one) == 0,
"single-thread apply succeeds");
omp_set_num_threads(4);
check(sensor_bloom_apply(second, width, height, &settings, &stats_four) == 0,
"four-thread apply succeeds");
check(memcmp(first, second, count * sizeof *first) == 0,
"final HDR is byte-identical for 1 and 4 threads");
check(stats_one.absorbed_signal == stats_four.absorbed_signal &&
stats_one.boundary_loss == stats_four.boundary_loss &&
stats_one.iterations == stats_four.iterations &&
stats_one.residual_clamp_loss == stats_four.residual_clamp_loss,
"reported statistics are identical for 1 and 4 threads");
free(first);
free(second);
free(hdr);
}
static void test_convergence_and_hard_limit(void) {
const int width = 9, height = 9;
const size_t count = (size_t)width * height * 3;
double *hdr = alloc_rgb(width, height);
set(hdr, width, 4, 4, 0, 100.0);
const SensorBloomSettings settings = {1.0, 0.5};
SensorBloomStats stats;
check(sensor_bloom_apply(hdr, width, height, &settings, &stats) == 0,
"converging apply succeeds");
check(all_finite_and_bounded(hdr, count, 1.0),
"converged output is finite and at most E");
check(stats.iterations <= stats.predicted_iterations,
"actual iterations do not exceed the conservative bound");
/* A round bound just inside the hard limit must still converge. The reported
* cap includes the fp guard but must never exceed 4096, and it must bound the
* actual number of rounds. */
const int near_width = 5, near_height = 5;
const size_t near_count = (size_t)near_width * near_height * 3;
double *near = alloc_rgb(near_width, near_height);
set(near, near_width, 2, 2, 0, 1.0 + 7.4e-9);
const SensorBloomSettings near_settings = {1.0, 0.9995};
SensorBloomStats near_stats;
check(sensor_bloom_apply(near, near_width, near_height, &near_settings,
&near_stats) == 0,
"near-limit round bound still converges");
check(near_stats.predicted_iterations <= (size_t)4096,
"reported round cap never exceeds the 4096 hard limit");
check(near_stats.iterations <= near_stats.predicted_iterations,
"actual rounds never exceed the reported guard-inclusive cap");
check(all_finite_and_bounded(near, near_count, 1.0),
"near-limit output is finite and at most E");
free(near);
/* A bound one round above the hard limit (4097 with these parameters) must
* fail in the pre-check, before any HDR sample is touched. */
double *near_fail = alloc_rgb(near_width, near_height);
set(near_fail, near_width, 2, 2, 0, 1.0 + 7.758e-9);
double *near_original = malloc(near_count * sizeof *near_original);
memcpy(near_original, near_fail, near_count * sizeof *near_original);
const SensorBloomSettings near_hard = {1.0, 0.9995};
SensorBloomStats near_hard_stats;
check(sensor_bloom_apply(near_fail, near_width, near_height, &near_hard,
&near_hard_stats) == -1,
"round bound just above the hard limit fails");
check(memcmp(near_fail, near_original, near_count * sizeof *near_fail) == 0,
"near-limit failure leaves the buffer byte-for-byte unchanged");
free(near_original);
free(near_fail);
/* An unbounded-overflow parameter set (e = 1 - 1e-6) needs far more than the
* 4096-round hard limit and must fail before modifying the input. */
double *failing = alloc_rgb(width, height);
set(failing, width, 4, 4, 0, 2.0);
double *original = malloc(count * sizeof *original);
memcpy(original, failing, count * sizeof *original);
const SensorBloomSettings hard = {1.0, 0.999999};
SensorBloomStats hard_stats;
check(sensor_bloom_apply(failing, width, height, &hard, &hard_stats) == -1,
"round bound above the hard limit fails");
check(memcmp(failing, original, count * sizeof *failing) == 0,
"hard-limit failure leaves the buffer byte-for-byte unchanged");
check(hard_stats.iterations == 0,
"hard-limit failure reports no iterations");
check(hard_stats.predicted_iterations == 0 &&
hard_stats.final_clamped_channels == 0,
"hard-limit failure reports no predicted rounds or clamps");
free(original);
free(failing);
free(hdr);
}
static void test_input_validation(void) {
const int width = 4, height = 3;
const size_t count = (size_t)width * height * 3;
double *hdr = alloc_rgb(width, height);
double *original = malloc(count * sizeof *original);
const SensorBloomSettings valid = {1.0, 0.5};
SensorBloomStats stats;
check(sensor_bloom_apply(NULL, width, height, &valid, &stats) == -1,
"NULL framebuffer is rejected");
check(sensor_bloom_apply(hdr, width, height, NULL, &stats) == -1,
"NULL settings are rejected");
check(sensor_bloom_apply(hdr, 0, height, &valid, &stats) == -1,
"zero width is rejected");
check(sensor_bloom_apply(hdr, width, -1, &valid, &stats) == -1,
"negative height is rejected");
check(sensor_bloom_apply(hdr, width, height, &valid, NULL) == 0,
"NULL stats are permitted for an unsaturated buffer");
double dummy = 0.0;
check(sensor_bloom_apply(&dummy, INT32_MAX, INT32_MAX, &valid, &stats) == -1,
"multiplication-overflowing dimensions are rejected before scanning");
const double bad_limits[] = {0.0, -1.0, NAN, INFINITY, -INFINITY};
for (size_t i = 0; i < sizeof bad_limits / sizeof bad_limits[0]; ++i) {
const SensorBloomSettings bad = {bad_limits[i], 0.5};
check(sensor_bloom_apply(hdr, width, height, &bad, &stats) == -1,
"invalid response limit is rejected");
}
const double bad_transfers[] = {-0.1, 1.0, 1.5, NAN, INFINITY, -INFINITY};
for (size_t i = 0; i < sizeof bad_transfers / sizeof bad_transfers[0]; ++i) {
const SensorBloomSettings bad = {1.0, bad_transfers[i]};
check(sensor_bloom_apply(hdr, width, height, &bad, &stats) == -1,
"invalid transfer is rejected");
}
const double non_finite[] = {NAN, INFINITY, -INFINITY};
for (size_t i = 0; i < sizeof non_finite / sizeof non_finite[0]; ++i) {
for (size_t j = 0; j < count; ++j)
hdr[j] = 0.5;
hdr[count / 2] = non_finite[i];
memcpy(original, hdr, count * sizeof *original);
check(sensor_bloom_apply(hdr, width, height, &valid, &stats) == -1,
"non-finite HDR sample is rejected");
check(memcmp(hdr, original, count * sizeof *hdr) == 0,
"non-finite rejection leaves the buffer unchanged");
}
free(original);
free(hdr);
}
#ifndef SB_HAVE_ASAN
/* Forces the scratch allocation to fail by lowering RLIMIT_AS in a child after
* the input buffer is already mapped, then checks that the model reports the
* allocation failure instead of touching the framebuffer. */
static int allocation_failure_is_reported(void) {
const pid_t pid = fork();
if (pid < 0)
return 0;
if (pid == 0) {
const int width = 64, height = 64;
const size_t count = (size_t)width * height * 3;
double *hdr = calloc(count, sizeof *hdr);
if (hdr == NULL)
_exit(2);
hdr[0] = 2.0;
struct rlimit existing;
if (getrlimit(RLIMIT_AS, &existing))
_exit(2);
const struct rlimit tiny = {.rlim_cur = 1, .rlim_max = existing.rlim_max};
if (setrlimit(RLIMIT_AS, &tiny))
_exit(2);
const SensorBloomSettings settings = {1.0, 0.5};
SensorBloomStats stats;
_exit(sensor_bloom_apply(hdr, width, height, &settings, &stats) == -1 ? 0
: 1);
}
int status = 0;
if (waitpid(pid, &status, 0) < 0)
return 0;
return WIFEXITED(status) && WEXITSTATUS(status) == 0;
}
#endif
static void test_nonnegative_output(void) {
const int width = 12, height = 7;
const size_t count = (size_t)width * height * 3;
double *hdr = alloc_rgb(width, height);
uint32_t state = 99u;
for (size_t i = 0; i < count; ++i)
hdr[i] = (double)(next_random(&state) % 4000u) / 1000.0;
const SensorBloomSettings settings = {1.2, 0.5};
SensorBloomStats stats;
check(sensor_bloom_apply(hdr, width, height, &settings, &stats) == 0,
"nonnegative apply succeeds");
int nonnegative = 1;
for (size_t i = 0; i < count; ++i)
if (!(hdr[i] >= 0.0) || !isfinite(hdr[i]))
nonnegative = 0;
check(nonnegative, "nonnegative input produces nonnegative finite output");
check(all_finite_and_bounded(hdr, count, 1.2),
"nonnegative output respects the response limit");
free(hdr);
}
int main(void) {
#ifndef SB_HAVE_ASAN
/* Run the fork-based allocation-failure check before any OpenMP region has
* been created in the parent. */
check(allocation_failure_is_reported(),
"scratch allocation failure is reported without touching the input");
#endif
test_no_saturation();
test_zero_transfer();
test_single_impulse_stencil();
test_boundary_loss();
test_cascade_against_reference();
test_symmetry();
test_thread_determinism();
test_convergence_and_hard_limit();
test_input_validation();
test_nonnegative_output();
if (failures != 0) {
fprintf(stderr, "%d sensor-bloom assertion(s) failed\n", failures);
return EXIT_FAILURE;
}
puts("sensor-bloom tests passed");
return EXIT_SUCCESS;
}