Feat: Overlap movie output and cut per-frame fast-mode work
- Gather the union of every movie frame's all-sky tiles once and read them in bounded batches, replacing the per-frame prefetch scan and log. - Fuse the fast supersampled-buffer clear into the FFTW pack pass so a resolved frame starts clean with no serial memset. - Split parallel HDR->RGB8 tone mapping from PNG encoding. - Add a bounded single-producer/single-writer movie output queue used by both observer movies and multi-frame imported lens maps, with writer timing and error propagation. - Add --png-compression-level and --movie-output-workers shared|reserve-one. - Add --fast-fftw-plan estimate|measure|wisdom|wisdom-update with strict wisdom identity sidecars. - Add staged movie timing, regression tests, and docs.
This commit is contained in:
1 parent
a510fec23e
commit
04611e3e5a
20 files changed
+2282
-212
No files matched your search
+117
-48
@@ -380,6 +380,23 @@ int catalog_mark_source_triangle_tiles(
|
||||
&context);
|
||||
}
|
||||
|
||||
void catalog_tile_set_clear(CatalogTileSet *set)
|
||||
{
|
||||
if (set != NULL)
|
||||
memset(set, 0, sizeof *set);
|
||||
}
|
||||
|
||||
size_t catalog_tile_set_count(const CatalogTileSet *set)
|
||||
{
|
||||
size_t count = 0;
|
||||
if (set == NULL)
|
||||
return 0;
|
||||
for (size_t tile_id = 0; tile_id < CATALOG_ALL_SKY_TILE_COUNT; ++tile_id)
|
||||
if (set->requested[tile_id])
|
||||
++count;
|
||||
return count;
|
||||
}
|
||||
|
||||
typedef struct {
|
||||
size_t tile_id;
|
||||
Star *stars;
|
||||
@@ -387,15 +404,79 @@ typedef struct {
|
||||
int loaded;
|
||||
} CatalogPendingTile;
|
||||
|
||||
int catalog_prefetch_marked_tiles(
|
||||
StarCatalog *catalog,
|
||||
const unsigned char requested[CATALOG_ALL_SKY_TILE_COUNT],
|
||||
int worker_count, CatalogPrefetchStats *stats)
|
||||
/* Reads and commits up to `batch_tiles` unseen tiles from `ids` at a time.
|
||||
* Every batch allocates only its own bounded pending array, so an all-sky
|
||||
* union cannot stage an unbounded temporary copy of the star data. */
|
||||
static int prefetch_tile_ids(StarCatalog *catalog, const size_t *ids,
|
||||
size_t id_count, int worker_count,
|
||||
size_t batch_tiles, CatalogPrefetchStats *stats)
|
||||
{
|
||||
CatalogPendingTile *pending = NULL;
|
||||
size_t pending_count = 0;
|
||||
double load_start;
|
||||
int result = -1;
|
||||
if (batch_tiles == 0)
|
||||
batch_tiles = CATALOG_PREFETCH_DEFAULT_BATCH_TILES;
|
||||
if (batch_tiles > id_count)
|
||||
batch_tiles = id_count;
|
||||
for (size_t base = 0; base < id_count; base += batch_tiles) {
|
||||
const size_t chunk = id_count - base < batch_tiles
|
||||
? id_count - base
|
||||
: batch_tiles;
|
||||
CatalogPendingTile *pending = calloc(chunk, sizeof *pending);
|
||||
if (pending == NULL)
|
||||
return -1;
|
||||
size_t pending_count = 0;
|
||||
for (size_t i = 0; i < chunk; ++i) {
|
||||
const size_t tile_id = ids[base + i];
|
||||
if (catalog->tiles[tile_id].state != 0)
|
||||
continue;
|
||||
pending[pending_count++].tile_id = tile_id;
|
||||
}
|
||||
if (pending_count > 0) {
|
||||
int batch_workers = worker_count;
|
||||
if (batch_workers > (int)pending_count)
|
||||
batch_workers = (int)pending_count;
|
||||
const double load_start = omp_get_wtime();
|
||||
#pragma omp parallel for num_threads(batch_workers) schedule(static)
|
||||
for (size_t i = 0; i < pending_count; ++i) {
|
||||
const int ra_index =
|
||||
(int)(pending[i].tile_id % CATALOG_ALL_SKY_RA_TILES);
|
||||
const int dec_index =
|
||||
(int)(pending[i].tile_id / CATALOG_ALL_SKY_RA_TILES);
|
||||
pending[i].loaded = !load_tile_file(
|
||||
catalog, ra_index, dec_index, &pending[i].stars,
|
||||
&pending[i].count);
|
||||
}
|
||||
if (stats != NULL)
|
||||
stats->load_seconds += omp_get_wtime() - load_start;
|
||||
/* Only this serial commit mutates the shared catalog cache. */
|
||||
for (size_t i = 0; i < pending_count; ++i) {
|
||||
CatalogTile *tile = &catalog->tiles[pending[i].tile_id];
|
||||
if (pending[i].loaded) {
|
||||
tile->stars = pending[i].stars;
|
||||
tile->count = pending[i].count;
|
||||
tile->state = 1;
|
||||
catalog->count += tile->count;
|
||||
if (stats != NULL) {
|
||||
++stats->newly_loaded_tiles;
|
||||
stats->newly_loaded_stars += tile->count;
|
||||
}
|
||||
} else {
|
||||
tile->state = -1;
|
||||
if (stats != NULL)
|
||||
++stats->unavailable_tiles;
|
||||
}
|
||||
}
|
||||
}
|
||||
free(pending);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Collects the requested-but-unseen tile ids, then reads them in batches. */
|
||||
static int prefetch_bitmap(StarCatalog *catalog,
|
||||
const unsigned char requested[CATALOG_ALL_SKY_TILE_COUNT],
|
||||
int worker_count, size_t batch_tiles,
|
||||
CatalogPrefetchStats *stats)
|
||||
{
|
||||
size_t unseen = 0;
|
||||
if (stats != NULL)
|
||||
*stats = (CatalogPrefetchStats){0};
|
||||
if (catalog == NULL || requested == NULL || worker_count <= 0)
|
||||
@@ -408,54 +489,42 @@ int catalog_prefetch_marked_tiles(
|
||||
if (stats != NULL)
|
||||
++stats->requested_tiles;
|
||||
if (catalog->tiles[tile_id].state == 0)
|
||||
++pending_count;
|
||||
++unseen;
|
||||
}
|
||||
if (pending_count == 0)
|
||||
if (unseen == 0)
|
||||
return 0;
|
||||
pending = calloc(pending_count, sizeof *pending);
|
||||
if (pending == NULL)
|
||||
size_t *ids = malloc(unseen * sizeof *ids);
|
||||
if (ids == NULL)
|
||||
return -1;
|
||||
size_t pending_index = 0;
|
||||
size_t index = 0;
|
||||
for (size_t tile_id = 0; tile_id < CATALOG_ALL_SKY_TILE_COUNT; ++tile_id)
|
||||
if (requested[tile_id] && catalog->tiles[tile_id].state == 0)
|
||||
pending[pending_index++].tile_id = tile_id;
|
||||
if (worker_count > (int)pending_count)
|
||||
worker_count = (int)pending_count;
|
||||
load_start = omp_get_wtime();
|
||||
#pragma omp parallel for num_threads(worker_count) schedule(static)
|
||||
for (size_t i = 0; i < pending_count; ++i) {
|
||||
const int ra_index =
|
||||
(int)(pending[i].tile_id % CATALOG_ALL_SKY_RA_TILES);
|
||||
const int dec_index =
|
||||
(int)(pending[i].tile_id / CATALOG_ALL_SKY_RA_TILES);
|
||||
pending[i].loaded = !load_tile_file(
|
||||
catalog, ra_index, dec_index, &pending[i].stars, &pending[i].count);
|
||||
}
|
||||
if (stats != NULL)
|
||||
stats->load_seconds = omp_get_wtime() - load_start;
|
||||
/* Only this serial commit mutates the shared catalog cache. */
|
||||
for (size_t i = 0; i < pending_count; ++i) {
|
||||
CatalogTile *tile = &catalog->tiles[pending[i].tile_id];
|
||||
if (pending[i].loaded) {
|
||||
tile->stars = pending[i].stars;
|
||||
tile->count = pending[i].count;
|
||||
tile->state = 1;
|
||||
catalog->count += tile->count;
|
||||
if (stats != NULL) {
|
||||
++stats->newly_loaded_tiles;
|
||||
stats->newly_loaded_stars += tile->count;
|
||||
}
|
||||
} else {
|
||||
tile->state = -1;
|
||||
if (stats != NULL)
|
||||
++stats->unavailable_tiles;
|
||||
}
|
||||
}
|
||||
result = 0;
|
||||
free(pending);
|
||||
ids[index++] = tile_id;
|
||||
const int result =
|
||||
prefetch_tile_ids(catalog, ids, unseen, worker_count, batch_tiles, stats);
|
||||
free(ids);
|
||||
return result;
|
||||
}
|
||||
|
||||
int catalog_prefetch_marked_tiles(
|
||||
StarCatalog *catalog,
|
||||
const unsigned char requested[CATALOG_ALL_SKY_TILE_COUNT],
|
||||
int worker_count, CatalogPrefetchStats *stats)
|
||||
{
|
||||
return prefetch_bitmap(catalog, requested, worker_count,
|
||||
CATALOG_ALL_SKY_TILE_COUNT, stats);
|
||||
}
|
||||
|
||||
int catalog_prefetch_tile_set(StarCatalog *catalog, const CatalogTileSet *set,
|
||||
int worker_count, size_t batch_tiles,
|
||||
CatalogPrefetchStats *stats)
|
||||
{
|
||||
if (set == NULL)
|
||||
return -1;
|
||||
return prefetch_bitmap(catalog, set->requested, worker_count, batch_tiles,
|
||||
stats);
|
||||
}
|
||||
|
||||
int catalog_visit_source_triangle(StarCatalog *catalog,
|
||||
const double direction[3][3],
|
||||
int load_missing, CatalogTileVisitor visitor,
|
||||
|
||||
@@ -38,6 +38,16 @@ typedef struct {
|
||||
typedef int (*CatalogTileVisitor)(const Star *stars, size_t count,
|
||||
int fully_contained, void *context);
|
||||
|
||||
/* Caller-owned, fixed-size set of one-degree all-sky tiles. A movie gathers
|
||||
* the union of every frame's requested tiles into one of these, prefetches the
|
||||
* union once, and then reuses an immutable cache for all frames. */
|
||||
typedef struct {
|
||||
unsigned char requested[CATALOG_ALL_SKY_TILE_COUNT];
|
||||
} CatalogTileSet;
|
||||
|
||||
void catalog_tile_set_clear(CatalogTileSet *set);
|
||||
size_t catalog_tile_set_count(const CatalogTileSet *set);
|
||||
|
||||
typedef struct {
|
||||
size_t requested_tiles;
|
||||
size_t newly_loaded_tiles;
|
||||
@@ -69,6 +79,24 @@ int catalog_prefetch_marked_tiles(
|
||||
StarCatalog *catalog,
|
||||
const unsigned char requested[CATALOG_ALL_SKY_TILE_COUNT],
|
||||
int worker_count, CatalogPrefetchStats *stats);
|
||||
/* Default upper bound on the number of tiles read per parallel batch. The
|
||||
* bound keeps the temporary star-vector staging area finite for an all-sky
|
||||
* union that may legitimately request tens of thousands of tiles. */
|
||||
enum { CATALOG_PREFETCH_DEFAULT_BATCH_TILES = 512 };
|
||||
|
||||
/* Batched equivalent of catalog_prefetch_marked_tiles() over a tile set. Each
|
||||
* batch reads at most batch_tiles unseen tiles with worker_count OpenMP workers
|
||||
* and commits them to the cache serially at the batch boundary; already loaded
|
||||
* or unavailable tiles are never re-read. A batch_tiles value <= 0 selects
|
||||
* CATALOG_PREFETCH_DEFAULT_BATCH_TILES. Returns 0 on success and -1 on a real
|
||||
* allocation/internal failure; a missing tile file is still recorded as
|
||||
* unavailable and does not fail the call. */
|
||||
int catalog_prefetch_tile_set(
|
||||
StarCatalog *catalog,
|
||||
const CatalogTileSet *set,
|
||||
int worker_count,
|
||||
size_t batch_tiles,
|
||||
CatalogPrefetchStats *stats);
|
||||
/* Visit only 1-degree all-sky tiles that can intersect this source triangle.
|
||||
* Call with load_missing=1 before parallel rendering, then 0 inside workers. */
|
||||
int catalog_visit_source_triangle(StarCatalog *catalog,
|
||||
|
||||
+294
-13
@@ -1,12 +1,14 @@
|
||||
#include "fast_psf_fftw.h"
|
||||
|
||||
#include <fftw3.h>
|
||||
#include <errno.h>
|
||||
#include <limits.h>
|
||||
#include <math.h>
|
||||
#include <omp.h>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
|
||||
/* Process-global FFTW threading lifecycle. Plans are created only from the
|
||||
* serial control thread, and global cleanup happens only after the last live
|
||||
@@ -14,16 +16,73 @@
|
||||
* process-wide planner state. */
|
||||
static int g_threading_initialized = 0;
|
||||
static int g_live_states = 0;
|
||||
static int g_plan_measure = 0;
|
||||
static FastPsfFftwPlanMode g_plan_mode = FAST_PSF_FFTW_PLAN_ESTIMATE;
|
||||
static const char *g_wisdom_path = NULL;
|
||||
static int g_wisdom_imported = 0;
|
||||
static int g_wisdom_available = 0;
|
||||
|
||||
void fast_psf_fftw_set_plan_mode(int measure)
|
||||
{
|
||||
g_plan_measure = measure ? 1 : 0;
|
||||
g_plan_mode = measure ? FAST_PSF_FFTW_PLAN_MEASURE
|
||||
: FAST_PSF_FFTW_PLAN_ESTIMATE;
|
||||
g_wisdom_path = NULL;
|
||||
g_wisdom_imported = 0;
|
||||
g_wisdom_available = 0;
|
||||
}
|
||||
|
||||
int fast_psf_fftw_configure(FastPsfFftwPlanMode mode, const char *wisdom_path)
|
||||
{
|
||||
if ((mode == FAST_PSF_FFTW_PLAN_WISDOM ||
|
||||
mode == FAST_PSF_FFTW_PLAN_WISDOM_UPDATE) &&
|
||||
(wisdom_path == NULL || wisdom_path[0] == '\0')) {
|
||||
fprintf(stderr,
|
||||
"Fast FFTW: --fast-fftw-plan wisdom modes require "
|
||||
"--fast-fftw-wisdom FILE.\n");
|
||||
return -1;
|
||||
}
|
||||
g_plan_mode = mode;
|
||||
g_wisdom_path = wisdom_path;
|
||||
g_wisdom_imported = 0;
|
||||
g_wisdom_available = 0;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static const char *plan_mode_name(void)
|
||||
{
|
||||
switch (g_plan_mode) {
|
||||
case FAST_PSF_FFTW_PLAN_MEASURE:
|
||||
return "measure";
|
||||
case FAST_PSF_FFTW_PLAN_WISDOM:
|
||||
return "wisdom";
|
||||
case FAST_PSF_FFTW_PLAN_WISDOM_UPDATE:
|
||||
return "wisdom-update";
|
||||
case FAST_PSF_FFTW_PLAN_ESTIMATE:
|
||||
default:
|
||||
return "estimate";
|
||||
}
|
||||
}
|
||||
|
||||
static unsigned plan_flags(void)
|
||||
{
|
||||
return g_plan_measure ? FFTW_MEASURE : FFTW_ESTIMATE;
|
||||
switch (g_plan_mode) {
|
||||
case FAST_PSF_FFTW_PLAN_MEASURE:
|
||||
case FAST_PSF_FFTW_PLAN_WISDOM_UPDATE:
|
||||
return FFTW_MEASURE;
|
||||
case FAST_PSF_FFTW_PLAN_WISDOM:
|
||||
return FFTW_WISDOM_ONLY;
|
||||
case FAST_PSF_FFTW_PLAN_ESTIMATE:
|
||||
default:
|
||||
return FFTW_ESTIMATE;
|
||||
}
|
||||
}
|
||||
|
||||
/* Wisdom identity sidecar. FFTW's own wisdom already encodes the transform
|
||||
* size and precision, but not this build's supersampling or worker count, so a
|
||||
* validated <wisdom>.meta accompanies every exported file. */
|
||||
static int wisdom_meta_path(char *out, size_t cap, const char *wisdom_path)
|
||||
{
|
||||
const int written = snprintf(out, cap, "%s.meta", wisdom_path);
|
||||
return written < 0 || (size_t)written >= cap ? -1 : 0;
|
||||
}
|
||||
|
||||
struct FastPsfFftwState {
|
||||
@@ -48,6 +107,178 @@ struct FastPsfFftwState {
|
||||
fftw_plan inverse_plan;
|
||||
};
|
||||
|
||||
/* Required identity fields. Every one must appear exactly once and match; a
|
||||
* missing or bad field is a miss, not an implicit default. */
|
||||
enum {
|
||||
WISDOM_META_VERSION = 1u << 0,
|
||||
WISDOM_META_PRECISION = 1u << 1,
|
||||
WISDOM_META_SS_WIDTH = 1u << 2,
|
||||
WISDOM_META_SS_HEIGHT = 1u << 3,
|
||||
WISDOM_META_FFT_WIDTH = 1u << 4,
|
||||
WISDOM_META_FFT_HEIGHT = 1u << 5,
|
||||
WISDOM_META_SUPERSAMPLE = 1u << 6,
|
||||
WISDOM_META_RADIUS = 1u << 7,
|
||||
WISDOM_META_WORKERS = 1u << 8,
|
||||
WISDOM_META_ALL = (1u << 9) - 1
|
||||
};
|
||||
|
||||
static int strict_long(const char *text, long *out)
|
||||
{
|
||||
char *end;
|
||||
errno = 0;
|
||||
const long value = strtol(text, &end, 10);
|
||||
if (errno != 0 || end == text || *end != '\0')
|
||||
return -1;
|
||||
*out = value;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int wisdom_meta_matches(const FastPsfFftwState *state)
|
||||
{
|
||||
if (g_wisdom_path == NULL)
|
||||
return 0;
|
||||
char path[PATH_MAX];
|
||||
if (wisdom_meta_path(path, sizeof path, g_wisdom_path))
|
||||
return 0;
|
||||
FILE *file = fopen(path, "r");
|
||||
if (file == NULL)
|
||||
return 0;
|
||||
char line[256];
|
||||
unsigned seen = 0;
|
||||
int ok = 1;
|
||||
while (ok && fgets(line, sizeof line, file) != NULL) {
|
||||
char key[64], value[192];
|
||||
if (sscanf(line, "%63[^=]=%191s", key, value) != 2)
|
||||
continue;
|
||||
long expected = 0;
|
||||
unsigned bit = 0;
|
||||
int numeric = 0;
|
||||
if (!strcmp(key, "fftw_version")) {
|
||||
ok = strcmp(value, fftw_version) == 0;
|
||||
bit = WISDOM_META_VERSION;
|
||||
} else if (!strcmp(key, "precision")) {
|
||||
ok = strcmp(value, "double") == 0;
|
||||
bit = WISDOM_META_PRECISION;
|
||||
} else if (!strcmp(key, "ss_width")) {
|
||||
expected = state->ss_width; numeric = 1; bit = WISDOM_META_SS_WIDTH;
|
||||
} else if (!strcmp(key, "ss_height")) {
|
||||
expected = state->ss_height; numeric = 1; bit = WISDOM_META_SS_HEIGHT;
|
||||
} else if (!strcmp(key, "fft_width")) {
|
||||
expected = state->fft_width; numeric = 1; bit = WISDOM_META_FFT_WIDTH;
|
||||
} else if (!strcmp(key, "fft_height")) {
|
||||
expected = state->fft_height; numeric = 1; bit = WISDOM_META_FFT_HEIGHT;
|
||||
} else if (!strcmp(key, "supersample")) {
|
||||
expected = state->supersample; numeric = 1; bit = WISDOM_META_SUPERSAMPLE;
|
||||
} else if (!strcmp(key, "kernel_radius")) {
|
||||
expected = state->kernel_radius; numeric = 1; bit = WISDOM_META_RADIUS;
|
||||
} else if (!strcmp(key, "workers")) {
|
||||
expected = state->workers; numeric = 1; bit = WISDOM_META_WORKERS;
|
||||
} else {
|
||||
continue;
|
||||
}
|
||||
if (ok && numeric) {
|
||||
long parsed = 0;
|
||||
ok = strict_long(value, &parsed) == 0 && parsed == expected;
|
||||
}
|
||||
if (ok && (seen & bit) != 0)
|
||||
ok = 0; /* duplicate field: reject rather than accept twice */
|
||||
if (ok)
|
||||
seen |= bit;
|
||||
}
|
||||
fclose(file);
|
||||
return ok && (seen & WISDOM_META_ALL) == WISDOM_META_ALL;
|
||||
}
|
||||
|
||||
static int wisdom_write_meta(const FastPsfFftwState *state, const char *path)
|
||||
{
|
||||
FILE *file = fopen(path, "w");
|
||||
if (file == NULL)
|
||||
return -1;
|
||||
const int written = fprintf(
|
||||
file,
|
||||
"fftw_version=%s\nprecision=double\nss_width=%d\nss_height=%d\n"
|
||||
"fft_width=%d\nfft_height=%d\nsupersample=%d\nkernel_radius=%d\n"
|
||||
"workers=%d\n",
|
||||
fftw_version, state->ss_width, state->ss_height, state->fft_width,
|
||||
state->fft_height, state->supersample, state->kernel_radius,
|
||||
state->workers);
|
||||
const int closed = fclose(file);
|
||||
return written < 0 || closed != 0 ? -1 : 0;
|
||||
}
|
||||
|
||||
static int wisdom_import_and_verify(const FastPsfFftwState *state)
|
||||
{
|
||||
if (!g_wisdom_imported) {
|
||||
g_wisdom_imported = 1;
|
||||
if (g_wisdom_path != NULL &&
|
||||
fftw_import_wisdom_from_filename(g_wisdom_path) != 0)
|
||||
g_wisdom_available = 1;
|
||||
}
|
||||
if (!g_wisdom_available) {
|
||||
fprintf(stderr,
|
||||
"Fast FFTW: --fast-fftw-plan wisdom could not import wisdom from "
|
||||
"%s.\n",
|
||||
g_wisdom_path != NULL ? g_wisdom_path : "(unset)");
|
||||
return 0;
|
||||
}
|
||||
if (!wisdom_meta_matches(state)) {
|
||||
fprintf(stderr,
|
||||
"Fast FFTW: wisdom in %s does not match this render "
|
||||
"(size, supersample, worker count, version, or precision).\n",
|
||||
g_wisdom_path);
|
||||
return 0;
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* Writes the wisdom and its identity sidecar next to the requested path.
|
||||
* Each file is written to a unique temporary and renamed into place, so each
|
||||
* rename is individually atomic. The wisdom file and its sidecar are two
|
||||
* separate files and are not transactionally updated: the sidecar is committed
|
||||
* last, so a loader that sees a wisdom without a matching sidecar treats it as
|
||||
* a miss. Any failure returns -1 and makes wisdom-update initialization fail. */
|
||||
static int wisdom_export(const FastPsfFftwState *state)
|
||||
{
|
||||
if (g_plan_mode != FAST_PSF_FFTW_PLAN_WISDOM_UPDATE || g_wisdom_path == NULL)
|
||||
return 0;
|
||||
char meta_path[PATH_MAX];
|
||||
char wisdom_temp[PATH_MAX];
|
||||
char meta_temp[PATH_MAX];
|
||||
if (wisdom_meta_path(meta_path, sizeof meta_path, g_wisdom_path))
|
||||
return -1;
|
||||
const long pid = (long)getpid();
|
||||
if (snprintf(wisdom_temp, sizeof wisdom_temp, "%s.tmp.%ld", g_wisdom_path,
|
||||
pid) < 0 ||
|
||||
snprintf(meta_temp, sizeof meta_temp, "%s.tmp.%ld", meta_path, pid) < 0)
|
||||
return -1;
|
||||
if (wisdom_write_meta(state, meta_temp) != 0) {
|
||||
fprintf(stderr, "Fast FFTW: failed to write wisdom metadata %s.\n",
|
||||
meta_temp);
|
||||
unlink(meta_temp);
|
||||
return -1;
|
||||
}
|
||||
if (fftw_export_wisdom_to_filename(wisdom_temp) == 0) {
|
||||
fputs("Fast FFTW: failed to export FFTW wisdom.\n", stderr);
|
||||
unlink(wisdom_temp);
|
||||
unlink(meta_temp);
|
||||
return -1;
|
||||
}
|
||||
if (rename(wisdom_temp, g_wisdom_path) != 0) {
|
||||
fprintf(stderr, "Fast FFTW: failed to commit wisdom to %s.\n",
|
||||
g_wisdom_path);
|
||||
unlink(wisdom_temp);
|
||||
unlink(meta_temp);
|
||||
return -1;
|
||||
}
|
||||
if (rename(meta_temp, meta_path) != 0) {
|
||||
fprintf(stderr, "Fast FFTW: failed to commit wisdom metadata to %s.\n",
|
||||
meta_path);
|
||||
unlink(meta_temp);
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int checked_mul_size(size_t a, size_t b, size_t *out)
|
||||
{
|
||||
if (a != 0 && b > SIZE_MAX / a)
|
||||
@@ -151,7 +382,9 @@ FastPsfFftwState *fast_psf_fftw_create(int ss_width, int ss_height,
|
||||
state->supersample = supersample;
|
||||
state->kernel_radius = kernel_radius;
|
||||
state->workers = fft_workers;
|
||||
state->plan_measure = g_plan_measure;
|
||||
state->plan_measure =
|
||||
g_plan_mode == FAST_PSF_FFTW_PLAN_MEASURE ||
|
||||
g_plan_mode == FAST_PSF_FFTW_PLAN_WISDOM_UPDATE;
|
||||
|
||||
size_t real_bytes = 0, freq_bytes = 0, kernel_bytes = 0;
|
||||
size_t fft_width = 0, fft_height = 0;
|
||||
@@ -204,6 +437,10 @@ FastPsfFftwState *fast_psf_fftw_create(int ss_width, int ss_height,
|
||||
state->registered = 1;
|
||||
fftw_plan_with_nthreads(fft_workers);
|
||||
|
||||
if (g_plan_mode == FAST_PSF_FFTW_PLAN_WISDOM &&
|
||||
!wisdom_import_and_verify(state))
|
||||
goto fail;
|
||||
|
||||
const double plan_start = omp_get_wtime();
|
||||
stage = "FFTW scratch allocation";
|
||||
state->real_rgb = fftw_alloc_real(real_count);
|
||||
@@ -269,6 +506,9 @@ FastPsfFftwState *fast_psf_fftw_create(int ss_width, int ss_height,
|
||||
|
||||
state->fft_scale = 1.0 / ((double)fft_width * (double)fft_height);
|
||||
state->box_scale = 1.0 / ((double)supersample * (double)supersample);
|
||||
stage = "wisdom export";
|
||||
if (wisdom_export(state) != 0)
|
||||
goto fail;
|
||||
return state;
|
||||
|
||||
fail:
|
||||
@@ -279,17 +519,18 @@ fail:
|
||||
stage, ss_width, ss_height,
|
||||
(size_t)ss_width + 2 * radius,
|
||||
(size_t)ss_height + 2 * radius, fft_width, fft_height,
|
||||
state->plan_measure ? "measure" : "estimate", fft_workers,
|
||||
plan_mode_name(), fft_workers,
|
||||
real_bytes, freq_bytes, kernel_bytes,
|
||||
real_bytes + freq_bytes + kernel_bytes);
|
||||
fast_psf_fftw_destroy(state);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
int fast_psf_fftw_resolve(FastPsfFftwState *state,
|
||||
const double *interleaved_ss_rgb,
|
||||
double *interleaved_hdr_rgb,
|
||||
FastPsfFftwFrameTiming *timing)
|
||||
static int fast_psf_fftw_resolve_impl(FastPsfFftwState *state,
|
||||
const double *interleaved_ss_rgb,
|
||||
int clear_source,
|
||||
double *interleaved_hdr_rgb,
|
||||
FastPsfFftwFrameTiming *timing)
|
||||
{
|
||||
if (state == NULL || interleaved_ss_rgb == NULL ||
|
||||
interleaved_hdr_rgb == NULL)
|
||||
@@ -303,7 +544,15 @@ int fast_psf_fftw_resolve(FastPsfFftwState *state,
|
||||
const size_t real_count = 3 * plane;
|
||||
const size_t freq_count = 3 * complex_plane;
|
||||
|
||||
/* The pack pass writes every FFT cell and, when clear_source is set, both
|
||||
* consumes and zeroes its own source triplet. Each supersampled triplet is
|
||||
* touched by exactly one worker, so zeroing needs no atomic and no second
|
||||
* serial memset. Clearing is deliberately unconditional in the consuming
|
||||
* variant: even a later failure must not leave stale deposits behind. */
|
||||
double *source_mut = (double *)(uintptr_t)interleaved_ss_rgb;
|
||||
double start = omp_get_wtime();
|
||||
/* The padded border is never overwritten by packing, so it must be zeroed
|
||||
* every frame regardless of the consuming variant. */
|
||||
#pragma omp parallel for schedule(static) num_threads(state->workers) \
|
||||
if (state->workers > 1)
|
||||
for (size_t i = 0; i < real_count; ++i)
|
||||
@@ -315,9 +564,14 @@ int fast_psf_fftw_resolve(FastPsfFftwState *state,
|
||||
const size_t source =
|
||||
3 * ((size_t)y * state->ss_width + (size_t)x);
|
||||
const size_t target = (size_t)y * state->fft_width + (size_t)x;
|
||||
state->real_rgb[target] = interleaved_ss_rgb[source];
|
||||
state->real_rgb[plane + target] = interleaved_ss_rgb[source + 1];
|
||||
state->real_rgb[2 * plane + target] = interleaved_ss_rgb[source + 2];
|
||||
state->real_rgb[target] = source_mut[source];
|
||||
state->real_rgb[plane + target] = source_mut[source + 1];
|
||||
state->real_rgb[2 * plane + target] = source_mut[source + 2];
|
||||
if (clear_source) {
|
||||
source_mut[source] = 0.0;
|
||||
source_mut[source + 1] = 0.0;
|
||||
source_mut[source + 2] = 0.0;
|
||||
}
|
||||
}
|
||||
}
|
||||
local.zero_pack_seconds = omp_get_wtime() - start;
|
||||
@@ -377,6 +631,33 @@ int fast_psf_fftw_resolve(FastPsfFftwState *state,
|
||||
return 0;
|
||||
}
|
||||
|
||||
int fast_psf_fftw_resolve(FastPsfFftwState *state,
|
||||
const double *interleaved_ss_rgb,
|
||||
double *interleaved_hdr_rgb,
|
||||
FastPsfFftwFrameTiming *timing)
|
||||
{
|
||||
return fast_psf_fftw_resolve_impl(state, interleaved_ss_rgb, 0,
|
||||
interleaved_hdr_rgb, timing);
|
||||
}
|
||||
|
||||
int fast_psf_fftw_resolve_and_clear(FastPsfFftwState *state,
|
||||
double *interleaved_ss_rgb,
|
||||
double *interleaved_hdr_rgb,
|
||||
FastPsfFftwFrameTiming *timing)
|
||||
{
|
||||
if (state == NULL || interleaved_ss_rgb == NULL)
|
||||
return -1;
|
||||
const int result = fast_psf_fftw_resolve_impl(state, interleaved_ss_rgb, 1,
|
||||
interleaved_hdr_rgb, timing);
|
||||
if (result != 0) {
|
||||
/* The pack pass may not have run, so guarantee a clean source even on the
|
||||
* error path rather than letting a later frame inherit stale deposits. */
|
||||
const size_t count = (size_t)state->ss_width * state->ss_height * 3;
|
||||
memset(interleaved_ss_rgb, 0, count * sizeof *interleaved_ss_rgb);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
void fast_psf_fftw_report(const FastPsfFftwState *state, FILE *stream)
|
||||
{
|
||||
if (state == NULL || stream == NULL)
|
||||
@@ -387,7 +668,7 @@ void fast_psf_fftw_report(const FastPsfFftwState *state, FILE *stream)
|
||||
state->ss_height + 2 * state->kernel_radius,
|
||||
state->ss_width + 2 * state->kernel_radius, state->fft_height,
|
||||
state->fft_width, state->workers,
|
||||
state->plan_measure ? "measure" : "estimate", state->setup_seconds,
|
||||
plan_mode_name(), state->setup_seconds,
|
||||
state->kernel_seconds,
|
||||
(double)state->scratch_bytes / (1024.0 * 1024.0));
|
||||
}
|
||||
@@ -46,6 +46,16 @@ int fast_psf_fftw_resolve(FastPsfFftwState *state,
|
||||
double *interleaved_hdr_rgb,
|
||||
FastPsfFftwFrameTiming *timing);
|
||||
|
||||
/* Same convolution, but consumes `interleaved_ss_rgb`: each source triplet is
|
||||
* copied into the planar scratch and immediately zeroed by the same worker, so
|
||||
* the caller gets a clean accumulator back without a separate serial memset.
|
||||
* On failure the source is still zeroed so a later frame cannot inherit stale
|
||||
* deposits. Returns 0 on success and -1 on invalid state. */
|
||||
int fast_psf_fftw_resolve_and_clear(FastPsfFftwState *state,
|
||||
double *interleaved_ss_rgb,
|
||||
double *interleaved_hdr_rgb,
|
||||
FastPsfFftwFrameTiming *timing);
|
||||
|
||||
/* Safe for NULL and partially initialized states; releases plans before the
|
||||
* buffers they reference. Does not touch process-global FFTW state except on
|
||||
* the final live state. */
|
||||
@@ -64,4 +74,21 @@ int fast_psf_fftw_next_smooth_size(size_t min_extent, size_t *out);
|
||||
* production uses the default FFTW_ESTIMATE. */
|
||||
void fast_psf_fftw_set_plan_mode(int measure);
|
||||
|
||||
/* Plan strategy for fast-mode convolution plans.
|
||||
* ESTIMATE - FFTW_ESTIMATE (current default)
|
||||
* MEASURE - FFTW_MEASURE
|
||||
* WISDOM - import FILE and require FFTW_WISDOM_ONLY; a miss fails
|
||||
* WISDOM_UPDATE - FFTW_MEASURE, then atomically export wisdom to FILE */
|
||||
typedef enum {
|
||||
FAST_PSF_FFTW_PLAN_ESTIMATE = 0,
|
||||
FAST_PSF_FFTW_PLAN_MEASURE = 1,
|
||||
FAST_PSF_FFTW_PLAN_WISDOM = 2,
|
||||
FAST_PSF_FFTW_PLAN_WISDOM_UPDATE = 3
|
||||
} FastPsfFftwPlanMode;
|
||||
|
||||
/* Must be called before fast_psf_fftw_create(). `wisdom_path` may be NULL for
|
||||
* the estimate/measure modes. Returns 0 on success and -1 on an immediately
|
||||
* detectable configuration error (missing path for a wisdom mode). */
|
||||
int fast_psf_fftw_configure(FastPsfFftwPlanMode mode, const char *wisdom_path);
|
||||
|
||||
#endif
|
||||
+80
-21
@@ -1361,13 +1361,9 @@ static CatalogSplatStats splat_catalog_triangles(
|
||||
return stats;
|
||||
}
|
||||
|
||||
static void prefetch_catalog_for_mesh(const FrameLensMesh *mesh,
|
||||
StarCatalog *catalog,
|
||||
int worker_count,
|
||||
CatalogPrefetchStats *stats) {
|
||||
if (catalog->kind != STAR_CATALOG_ALL_SKY)
|
||||
return;
|
||||
unsigned char requested[CATALOG_ALL_SKY_TILE_COUNT] = {0};
|
||||
int frame_mark_catalog_tiles(const FrameLensMesh *mesh, CatalogTileSet *set) {
|
||||
if (mesh == NULL || set == NULL)
|
||||
return -1;
|
||||
for (size_t t = 0; t < mesh->triangle_count; ++t) {
|
||||
const LensVertex *vertex[3];
|
||||
if (!usable_triangle(mesh, &mesh->triangles[t], vertex))
|
||||
@@ -1376,9 +1372,29 @@ static void prefetch_catalog_for_mesh(const FrameLensMesh *mesh,
|
||||
{vertex[0]->n_infinity[0], vertex[0]->n_infinity[1], vertex[0]->n_infinity[2]},
|
||||
{vertex[1]->n_infinity[0], vertex[1]->n_infinity[1], vertex[1]->n_infinity[2]},
|
||||
{vertex[2]->n_infinity[0], vertex[2]->n_infinity[1], vertex[2]->n_infinity[2]}};
|
||||
(void)catalog_mark_source_triangle_tiles(direction, requested);
|
||||
(void)catalog_mark_source_triangle_tiles(direction, set->requested);
|
||||
}
|
||||
(void)catalog_prefetch_marked_tiles(catalog, requested, worker_count, stats);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void prefetch_catalog_for_mesh(const FrameLensMesh *mesh,
|
||||
StarCatalog *catalog,
|
||||
int worker_count,
|
||||
CatalogPrefetchStats *stats,
|
||||
MovieFrameTiming *timing) {
|
||||
if (catalog->kind != STAR_CATALOG_ALL_SKY)
|
||||
return;
|
||||
CatalogTileSet set;
|
||||
catalog_tile_set_clear(&set);
|
||||
const double mark_start = omp_get_wtime();
|
||||
(void)frame_mark_catalog_tiles(mesh, &set);
|
||||
if (timing != NULL)
|
||||
timing->catalog_mark_seconds = omp_get_wtime() - mark_start;
|
||||
const double load_start = omp_get_wtime();
|
||||
(void)catalog_prefetch_tile_set(catalog, &set, worker_count,
|
||||
CATALOG_PREFETCH_DEFAULT_BATCH_TILES, stats);
|
||||
if (timing != NULL)
|
||||
timing->catalog_load_seconds = omp_get_wtime() - load_start;
|
||||
}
|
||||
|
||||
size_t frame_splat_catalog(const FrameLensMesh *mesh,
|
||||
@@ -1392,10 +1408,12 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
|
||||
double psf_min_y,
|
||||
int limit_workers_by_memory,
|
||||
int catalog_load_workers,
|
||||
FrameCatalogPrefetchMode catalog_prefetch_mode,
|
||||
CatalogPrefetchStats *prefetch_stats,
|
||||
PsfSplatStats *psf_stats,
|
||||
const FrameSplatProgress *progress,
|
||||
FastPsfAccumulator *fast) {
|
||||
FastPsfAccumulator *fast,
|
||||
MovieFrameTiming *timing) {
|
||||
if (mesh == NULL || catalog == NULL || hdr == NULL || exposure <= 0.0 ||
|
||||
psf == NULL || width <= 0 || height <= 0 || catalog_load_workers <= 0 ||
|
||||
isnan(max_magnification) || max_magnification <= 0.0 ||
|
||||
@@ -1404,21 +1422,32 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
|
||||
psf_relative_tail >= 1.0 || !isfinite(psf_min_y) || psf_min_y < 0.0)
|
||||
return 0;
|
||||
|
||||
/* A bounded parallel read phase completes before splatting. Its serial cache
|
||||
* commit leaves immutable tile data for the OpenMP splat workers. */
|
||||
if (progress != NULL && progress->callback != NULL)
|
||||
progress->callback(progress->context, FRAME_SPLAT_PROGRESS_PREFETCH_BEGIN,
|
||||
0, mesh->triangle_count);
|
||||
prefetch_catalog_for_mesh(mesh, catalog, catalog_load_workers, prefetch_stats);
|
||||
if (progress != NULL && progress->callback != NULL)
|
||||
progress->callback(progress->context, FRAME_SPLAT_PROGRESS_PREFETCH_END,
|
||||
prefetch_stats == NULL ? 0 : prefetch_stats->requested_tiles,
|
||||
prefetch_stats == NULL ? 0 : prefetch_stats->requested_tiles);
|
||||
if (timing != NULL)
|
||||
*timing = (MovieFrameTiming){0};
|
||||
if (psf_stats != NULL)
|
||||
*psf_stats = (PsfSplatStats){0};
|
||||
|
||||
if (catalog_prefetch_mode == FRAME_CATALOG_PREFETCH_FRAME) {
|
||||
/* A bounded parallel read phase completes before splatting. Its serial
|
||||
* cache commit leaves immutable tile data for the OpenMP splat workers. */
|
||||
if (progress != NULL && progress->callback != NULL)
|
||||
progress->callback(progress->context, FRAME_SPLAT_PROGRESS_PREFETCH_BEGIN,
|
||||
0, mesh->triangle_count);
|
||||
prefetch_catalog_for_mesh(mesh, catalog, catalog_load_workers,
|
||||
prefetch_stats, timing);
|
||||
if (progress != NULL && progress->callback != NULL)
|
||||
progress->callback(progress->context, FRAME_SPLAT_PROGRESS_PREFETCH_END,
|
||||
prefetch_stats == NULL ? 0 : prefetch_stats->requested_tiles,
|
||||
prefetch_stats == NULL ? 0 : prefetch_stats->requested_tiles);
|
||||
} else if (prefetch_stats != NULL) {
|
||||
/* A movie-level union prefetch already ran: skip every per-frame tile scan,
|
||||
* prefetch worker region, and "Catalog prefetch" log. */
|
||||
*prefetch_stats = (CatalogPrefetchStats){0};
|
||||
}
|
||||
if (progress != NULL && progress->callback != NULL)
|
||||
progress->callback(progress->context, FRAME_SPLAT_PROGRESS_BEGIN, 0,
|
||||
mesh->triangle_count);
|
||||
const double splat_body_start = omp_get_wtime();
|
||||
|
||||
/* Fast mode replaces the per-event PSF splat with cheap delta deposits into
|
||||
* one shared supersampled buffer. A single global convolution plus an N x N
|
||||
@@ -1431,7 +1460,22 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
|
||||
int worker_count = omp_get_max_threads();
|
||||
if (worker_count < 1)
|
||||
worker_count = 1;
|
||||
fast_psf_accumulator_clear(fast);
|
||||
/* The FFTW consume path clears the shared buffer as it packs, so a CLEAN
|
||||
* accumulator needs no per-frame serial memset. The spatial-reference
|
||||
* fallback and any unexpected stale state still get an explicit clear. */
|
||||
#ifdef FAST_PSF_FFTW
|
||||
const int needs_clear =
|
||||
!fast->fftw_enabled || fast->buffer_state != FAST_PSF_BUFFER_CLEAN;
|
||||
#else
|
||||
const int needs_clear = 1;
|
||||
#endif
|
||||
if (needs_clear) {
|
||||
const double clear_start = omp_get_wtime();
|
||||
fast_psf_accumulator_clear(fast);
|
||||
if (timing != NULL)
|
||||
timing->fast_clear_seconds = omp_get_wtime() - clear_start;
|
||||
}
|
||||
const double deposit_start = omp_get_wtime();
|
||||
#ifdef GR_DEBUG
|
||||
double max_raw_magnification = 0.0;
|
||||
size_t magnification_clamped_triangles = 0;
|
||||
@@ -1458,8 +1502,21 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
|
||||
#endif
|
||||
}
|
||||
}
|
||||
if (timing != NULL)
|
||||
timing->catalog_splat_seconds = omp_get_wtime() - deposit_start;
|
||||
if (!failed && fast_psf_accumulator_resolve(fast, hdr, worker_count))
|
||||
failed = 1;
|
||||
#ifdef FAST_PSF_FFTW
|
||||
if (timing != NULL && fast->fftw_enabled) {
|
||||
timing->fftw_zero_pack_seconds = fast->fftw_last_timing.zero_pack_seconds;
|
||||
timing->fftw_forward_seconds = fast->fftw_last_timing.forward_seconds;
|
||||
timing->fftw_multiply_seconds = fast->fftw_last_timing.multiply_seconds;
|
||||
timing->fftw_inverse_seconds = fast->fftw_last_timing.inverse_seconds;
|
||||
timing->fftw_crop_seconds =
|
||||
fast->fftw_last_timing.crop_downsample_seconds;
|
||||
timing->fftw_total_seconds = fast->fftw_last_timing.total_seconds;
|
||||
}
|
||||
#endif
|
||||
copy_psf_splat_stats(psf_stats, (CatalogSplatStats){
|
||||
.images = images,
|
||||
.direct_fallbacks = direct_fallbacks,
|
||||
@@ -1799,6 +1856,8 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
|
||||
for (size_t worker = 0; worker < worker_count; ++worker)
|
||||
free(private_hdr[worker]);
|
||||
free(private_hdr);
|
||||
if (timing != NULL)
|
||||
timing->catalog_splat_seconds = omp_get_wtime() - splat_body_start;
|
||||
copy_psf_splat_stats(psf_stats, (CatalogSplatStats){
|
||||
.images = images,
|
||||
.direct_fallbacks = direct_fallbacks,
|
||||
|
||||
+35
-1
@@ -24,6 +24,26 @@ typedef struct {
|
||||
int evaluated;
|
||||
} LensTriangle;
|
||||
|
||||
/* Per-frame staged wall-clock breakdown for one movie frame. All fields are
|
||||
* seconds and are zero-initialized by the caller. Hot loops never call a
|
||||
* clock; each phase is timed exactly once around its bulk operation. */
|
||||
typedef struct {
|
||||
double catalog_mark_seconds;
|
||||
double catalog_load_seconds;
|
||||
double fast_clear_seconds;
|
||||
double catalog_splat_seconds;
|
||||
double fftw_zero_pack_seconds;
|
||||
double fftw_forward_seconds;
|
||||
double fftw_multiply_seconds;
|
||||
double fftw_inverse_seconds;
|
||||
double fftw_crop_seconds;
|
||||
double fftw_total_seconds;
|
||||
double tone_map_seconds;
|
||||
double png_encode_write_seconds;
|
||||
double writer_queue_wait_seconds;
|
||||
double frame_total_seconds;
|
||||
} MovieFrameTiming;
|
||||
|
||||
typedef struct {
|
||||
unsigned int max_level;
|
||||
double angle_absolute_rad;
|
||||
@@ -121,6 +141,18 @@ int frame_lens_mesh_refine_with_progress(
|
||||
const RefinementConfig *config, FrameRefinementProgressCallback callback,
|
||||
void *context);
|
||||
|
||||
/* Whether frame_splat_catalog() should run the per-frame catalog prefetch or
|
||||
* rely on a movie-level union prefetch that already completed. */
|
||||
typedef enum {
|
||||
FRAME_CATALOG_PREFETCH_FRAME,
|
||||
FRAME_CATALOG_PREFETCH_ALREADY_COMPLETE
|
||||
} FrameCatalogPrefetchMode;
|
||||
|
||||
/* Adds every one-degree tile that can intersect a usable source triangle of
|
||||
* the finalized mesh to `set`. Kept in the frame module so catalog.h does not
|
||||
* depend on FrameLensMesh. */
|
||||
int frame_mark_catalog_tiles(const FrameLensMesh *mesh, CatalogTileSet *set);
|
||||
|
||||
/* Each locally invertible escaped triangle contributes one image per contained
|
||||
* star. When `fast` is non-NULL, each image is deposited into its shared
|
||||
* supersampled buffer and the whole frame is resolved once at the end instead
|
||||
@@ -136,10 +168,12 @@ size_t frame_splat_catalog(const FrameLensMesh *mesh,
|
||||
double psf_min_y,
|
||||
int limit_workers_by_memory,
|
||||
int catalog_load_workers,
|
||||
FrameCatalogPrefetchMode catalog_prefetch_mode,
|
||||
CatalogPrefetchStats *prefetch_stats,
|
||||
PsfSplatStats *psf_stats,
|
||||
const FrameSplatProgress *progress,
|
||||
FastPsfAccumulator *fast);
|
||||
FastPsfAccumulator *fast,
|
||||
MovieFrameTiming *timing);
|
||||
void frame_draw_mesh(const FrameLensMesh *mesh, double *hdr, int width,
|
||||
int height, double gray, double opacity);
|
||||
void frame_lens_mesh_destroy(FrameLensMesh *mesh);
|
||||
|
||||
+447
-50
@@ -2,6 +2,7 @@
|
||||
#include "frame.h"
|
||||
#include "lens_map.h"
|
||||
#include "movie.h"
|
||||
#include "movie_output.h"
|
||||
#include "observer_track.h"
|
||||
#include "optics.h"
|
||||
#include "ray.h"
|
||||
@@ -59,9 +60,13 @@ typedef struct {
|
||||
double minkowski_proper_acceleration;
|
||||
double alcubierre_vs, alcubierre_radius, alcubierre_sigma;
|
||||
int catalog_load_workers;
|
||||
int movie_output_workers; /* 0 = shared, 1 = reserve one core for writer */
|
||||
FastPsfFftwPlanMode fast_fftw_plan_mode;
|
||||
const char *fast_fftw_wisdom_path;
|
||||
const char *blackbody_table_path;
|
||||
RefinementConfig refinement;
|
||||
ToneMapSettings tone_map;
|
||||
PngWriteSettings png;
|
||||
int sensor_bloom_enabled;
|
||||
int sensor_bloom_limit_specified;
|
||||
int sensor_bloom_transfer_specified;
|
||||
@@ -227,6 +232,36 @@ typedef struct {
|
||||
int draw_mesh;
|
||||
} FrameOutputPaths;
|
||||
|
||||
/* Parallel HDR -> sRGB8 conversion and the timed PNG/write step, split so the
|
||||
* movie timing can attribute each phase separately. */
|
||||
static int render_rgb8_image(const Settings *s, const double *hdr, int width,
|
||||
int height, unsigned char **out,
|
||||
MovieFrameTiming *timing) {
|
||||
unsigned char *rgb8 = malloc((size_t)width * height * 3);
|
||||
if (rgb8 == NULL)
|
||||
return -1;
|
||||
const double tone_start = omp_get_wtime();
|
||||
if (tone_map_srgb8_image(hdr, rgb8, width, height, &s->tone_map,
|
||||
omp_get_max_threads())) {
|
||||
free(rgb8);
|
||||
return -1;
|
||||
}
|
||||
if (timing != NULL)
|
||||
timing->tone_map_seconds += omp_get_wtime() - tone_start;
|
||||
*out = rgb8;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int write_rgb8_timed(const Settings *s, const char *path,
|
||||
const unsigned char *rgb8, int width, int height,
|
||||
MovieFrameTiming *timing) {
|
||||
const double write_start = omp_get_wtime();
|
||||
const int result = write_rgb8_image(path, rgb8, width, height, &s->png);
|
||||
if (timing != NULL)
|
||||
timing->png_encode_write_seconds += omp_get_wtime() - write_start;
|
||||
return result;
|
||||
}
|
||||
|
||||
/* Fills in the plain and mesh-overlay output paths for one frame, returning
|
||||
* -1 if a requested mesh sibling would overflow PATH_MAX. Callers choose the
|
||||
* timing; main() separately pre-validates the single-frame path before
|
||||
@@ -245,10 +280,41 @@ static int build_frame_output_paths(const Settings *s, const char *output_path,
|
||||
/* Canonical output order for every frame: clean HDR, clean tone-mapped image,
|
||||
* then the mesh overlay. The overlay reuses the already-consumed HDR buffer,
|
||||
* so no second full-size framebuffer is allocated and nothing is re-rendered. */
|
||||
/* Optional sensor bloom applied to the post-exposure linear HDR before the
|
||||
* tone map. Shared by the synchronous writer and the async movie producer. */
|
||||
static int apply_sensor_bloom(const Settings *s, double *hdr, int width,
|
||||
int height) {
|
||||
if (!s->sensor_bloom_enabled)
|
||||
return 0;
|
||||
const SensorBloomSettings bloom = {.response_limit = s->sensor_bloom_limit,
|
||||
.transfer = s->sensor_bloom_transfer};
|
||||
SensorBloomStats bloom_stats;
|
||||
if (sensor_bloom_apply(hdr, width, height, &bloom, &bloom_stats)) {
|
||||
fputs("Sensor bloom failed (invalid settings, non-finite HDR, allocation "
|
||||
"failure, or a round bound above the 4096 hard limit; lower "
|
||||
"--sensor-bloom-transfer, raise --sensor-bloom-limit, or reduce "
|
||||
"--exposure); aborting tone-mapped output.\n",
|
||||
stderr);
|
||||
return -1;
|
||||
}
|
||||
fprintf(stderr,
|
||||
"Sensor bloom: saturated=%zu clamped=%zu iterations=%zu/%zu "
|
||||
"peak=%.6g max_overflow=%.6g absorbed=%.6g boundary=%.6g "
|
||||
"residual=%.6g elapsed=%.6fs\n",
|
||||
bloom_stats.initially_saturated_channels,
|
||||
bloom_stats.final_clamped_channels, bloom_stats.iterations,
|
||||
bloom_stats.predicted_iterations, bloom_stats.peak_input,
|
||||
bloom_stats.initial_max_overflow, bloom_stats.absorbed_signal,
|
||||
bloom_stats.boundary_loss, bloom_stats.residual_clamp_loss,
|
||||
bloom_stats.elapsed_seconds);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int write_frame_outputs(const Settings *s, const FrameLensMesh *mesh,
|
||||
double *hdr, int width, int height, double fov_deg,
|
||||
const FrameOutputPaths *paths, size_t images,
|
||||
size_t stars, const char *note) {
|
||||
size_t stars, const char *note,
|
||||
MovieFrameTiming *timing) {
|
||||
#ifdef ENABLE_HDR_OUTPUT
|
||||
if (s->write_hdr_output &&
|
||||
write_hdr_fits(s->hdr_output_path, hdr, width, height, fov_deg)) {
|
||||
@@ -259,33 +325,16 @@ static int write_frame_outputs(const Settings *s, const FrameLensMesh *mesh,
|
||||
(void)s;
|
||||
(void)fov_deg;
|
||||
#endif
|
||||
if (s->sensor_bloom_enabled) {
|
||||
const SensorBloomSettings bloom = {
|
||||
.response_limit = s->sensor_bloom_limit,
|
||||
.transfer = s->sensor_bloom_transfer};
|
||||
SensorBloomStats bloom_stats;
|
||||
if (sensor_bloom_apply(hdr, width, height, &bloom, &bloom_stats)) {
|
||||
fputs("Sensor bloom failed (invalid settings, non-finite HDR, allocation "
|
||||
"failure, or a round bound above the 4096 hard limit; lower "
|
||||
"--sensor-bloom-transfer, raise --sensor-bloom-limit, or reduce "
|
||||
"--exposure); aborting tone-mapped output.\n",
|
||||
stderr);
|
||||
return -1;
|
||||
}
|
||||
fprintf(stderr,
|
||||
"Sensor bloom: saturated=%zu clamped=%zu iterations=%zu/%zu "
|
||||
"peak=%.6g max_overflow=%.6g absorbed=%.6g boundary=%.6g "
|
||||
"residual=%.6g elapsed=%.6fs\n",
|
||||
bloom_stats.initially_saturated_channels,
|
||||
bloom_stats.final_clamped_channels, bloom_stats.iterations,
|
||||
bloom_stats.predicted_iterations, bloom_stats.peak_input,
|
||||
bloom_stats.initial_max_overflow, bloom_stats.absorbed_signal,
|
||||
bloom_stats.boundary_loss, bloom_stats.residual_clamp_loss,
|
||||
bloom_stats.elapsed_seconds);
|
||||
if (apply_sensor_bloom(s, hdr, width, height))
|
||||
return -1;
|
||||
unsigned char *rgb8 = NULL;
|
||||
if (render_rgb8_image(s, hdr, width, height, &rgb8, timing)) {
|
||||
fputs("Tone mapping failed (allocation failure).\n", stderr);
|
||||
return -1;
|
||||
}
|
||||
const int write_result =
|
||||
write_tonemapped_image(paths->output_path, hdr, width, height,
|
||||
&s->tone_map);
|
||||
write_rgb8_timed(s, paths->output_path, rgb8, width, height, timing);
|
||||
free(rgb8);
|
||||
fprintf(stderr, "Rendered %zu images from %zu catalog stars to %s (%s%s)\n",
|
||||
images, stars, paths->output_path,
|
||||
write_result == 0 ? "ok" : "write failed", note);
|
||||
@@ -293,12 +342,16 @@ static int write_frame_outputs(const Settings *s, const FrameLensMesh *mesh,
|
||||
return -1;
|
||||
if (paths->draw_mesh) {
|
||||
frame_draw_mesh(mesh, hdr, width, height, 0.5, 0.5);
|
||||
if (write_tonemapped_image(paths->mesh_path, hdr, width, height,
|
||||
&s->tone_map)) {
|
||||
unsigned char *mesh_rgb8 = NULL;
|
||||
if (render_rgb8_image(s, hdr, width, height, &mesh_rgb8, timing) ||
|
||||
write_rgb8_timed(s, paths->mesh_path, mesh_rgb8, width, height,
|
||||
timing)) {
|
||||
free(mesh_rgb8);
|
||||
fprintf(stderr, "Failed to write mesh overlay image: %s\n",
|
||||
paths->mesh_path);
|
||||
return -1;
|
||||
}
|
||||
free(mesh_rgb8);
|
||||
fprintf(stderr, "Wrote mesh overlay image: %s\n", paths->mesh_path);
|
||||
}
|
||||
return 0;
|
||||
@@ -345,13 +398,16 @@ static int parse_args(int argc, char **argv, Settings *s,
|
||||
.alcubierre_radius = 5.0,
|
||||
.alcubierre_sigma = 1.0,
|
||||
.catalog_load_workers = 4,
|
||||
.movie_output_workers = 0,
|
||||
.fast_fftw_plan_mode = FAST_PSF_FFTW_PLAN_ESTIMATE,
|
||||
.refinement = {.angle_absolute_rad =
|
||||
1e-3 * 3.14159265358979323846 / 180.0,
|
||||
.angle_relative = 0.1,
|
||||
.jacobian_minimum = 1e-3,
|
||||
.min_edge_pixels = 0.5,
|
||||
.min_area_pixels2 = 0.25},
|
||||
.tone_map = {.op = TONE_MAP_SOFTCLIP, .p = 2.0}};
|
||||
.tone_map = {.op = TONE_MAP_SOFTCLIP, .p = 2.0},
|
||||
.png = {.compression_level = -1}};
|
||||
#ifdef SPACETIME_ALCUBIERRE
|
||||
/* The default bubble (R=5, sigma=1) has escape radius 25, so the generic
|
||||
* radius-30 camera would sit outside the active domain. */
|
||||
@@ -500,6 +556,35 @@ static int parse_args(int argc, char **argv, Settings *s,
|
||||
#endif
|
||||
} else if (!strcmp(argv[i], "--catalog-load-workers") && i + 1 < argc &&
|
||||
!parse_int(argv[++i], &s->catalog_load_workers)) {
|
||||
} else if (!strcmp(argv[i], "--png-compression-level") && i + 1 < argc) {
|
||||
char *end;
|
||||
errno = 0;
|
||||
const long level = strtol(argv[++i], &end, 10);
|
||||
if (errno || *end || level < 0 || level > 9)
|
||||
return -1;
|
||||
s->png.compression_level = (int)level;
|
||||
} else if (!strcmp(argv[i], "--movie-output-workers") && i + 1 < argc) {
|
||||
const char *mode = argv[++i];
|
||||
if (!strcmp(mode, "shared"))
|
||||
s->movie_output_workers = 0;
|
||||
else if (!strcmp(mode, "reserve-one"))
|
||||
s->movie_output_workers = 1;
|
||||
else
|
||||
return -1;
|
||||
} else if (!strcmp(argv[i], "--fast-fftw-plan") && i + 1 < argc) {
|
||||
const char *mode = argv[++i];
|
||||
if (!strcmp(mode, "estimate"))
|
||||
s->fast_fftw_plan_mode = FAST_PSF_FFTW_PLAN_ESTIMATE;
|
||||
else if (!strcmp(mode, "measure"))
|
||||
s->fast_fftw_plan_mode = FAST_PSF_FFTW_PLAN_MEASURE;
|
||||
else if (!strcmp(mode, "wisdom"))
|
||||
s->fast_fftw_plan_mode = FAST_PSF_FFTW_PLAN_WISDOM;
|
||||
else if (!strcmp(mode, "wisdom-update"))
|
||||
s->fast_fftw_plan_mode = FAST_PSF_FFTW_PLAN_WISDOM_UPDATE;
|
||||
else
|
||||
return -1;
|
||||
} else if (!strcmp(argv[i], "--fast-fftw-wisdom") && i + 1 < argc) {
|
||||
s->fast_fftw_wisdom_path = argv[++i];
|
||||
} else if (!strcmp(argv[i], "--blackbody-table") && i + 1 < argc) {
|
||||
s->blackbody_table_path = argv[++i];
|
||||
} else if (!strcmp(argv[i], "--write-minkowski-accel-track") &&
|
||||
@@ -591,7 +676,16 @@ static void print_help(const char *program) {
|
||||
" --fast-supersample N Fast-mode supersample factor, 1..8 (default: 2)\n"
|
||||
" --fast-deposit MODE nearest (exact FWHM/beta, 0.5/N px quantization) or\n"
|
||||
" bilinear (exact centroid, broadens FWHM); default: nearest\n"
|
||||
" --fast-fftw-plan MODE estimate (default), measure, wisdom, or wisdom-update\n"
|
||||
" (CPU mode only)\n"
|
||||
" --fast-fftw-wisdom FILE FFTW wisdom file for the wisdom plan modes\n",
|
||||
stdout);
|
||||
fputs(
|
||||
" --catalog-load-workers N All-sky catalog loader workers (default: 4)\n"
|
||||
" --png-compression-level N libpng/zlib compression level 0..9\n"
|
||||
" (default: libpng default; use 1 for preview movies)\n"
|
||||
" --movie-output-workers MODE shared (renderer keeps all threads) or reserve-one\n"
|
||||
" (renderer leaves one core for the async writer; default shared)\n"
|
||||
" --blackbody-table FILE Explicit GRBBLUT3 table\n"
|
||||
" (default: assets/blackbody/cie1931_2deg_xyz_1024.grbblut)\n",
|
||||
stdout);
|
||||
@@ -967,11 +1061,12 @@ static int render_observer_frame(const Settings *s, StarCatalog *catalog,
|
||||
s->psf_relative_tail,
|
||||
s->psf_min_y,
|
||||
spacetime_limits_render_workers_by_memory(spacetime),
|
||||
s->catalog_load_workers, &prefetch, &psf_stats,
|
||||
s->catalog_load_workers, FRAME_CATALOG_PREFETCH_FRAME, &prefetch,
|
||||
&psf_stats,
|
||||
&(FrameSplatProgress){report_splat_progress,
|
||||
s->verbose ? report_splat_worker_progress : NULL,
|
||||
&progress},
|
||||
s->fast_psf);
|
||||
s->fast_psf, NULL);
|
||||
if (images == SIZE_MAX) {
|
||||
frame_lens_mesh_destroy(&mesh);
|
||||
free(hdr);
|
||||
@@ -985,7 +1080,7 @@ static int render_observer_frame(const Settings *s, StarCatalog *catalog,
|
||||
}
|
||||
const int result = write_frame_outputs(
|
||||
s, &mesh, hdr, s->width, s->height, s->horizontal_fov_deg, &output_paths,
|
||||
images, catalog->count, "");
|
||||
images, catalog->count, "", NULL);
|
||||
report_psf_splat(s, &psf_stats);
|
||||
frame_lens_mesh_destroy(&mesh);
|
||||
free(hdr);
|
||||
@@ -1102,12 +1197,128 @@ static int trace_movie_generation(Movie *movie, const Settings *s,
|
||||
return 1;
|
||||
}
|
||||
|
||||
enum { MOVIE_TIMING_COUNT = 14 };
|
||||
|
||||
/* Running per-phase totals and maxima across the movie's frames. No frame
|
||||
* arrays are retained, so memory stays O(1) in the frame count. */
|
||||
typedef struct {
|
||||
double sum[MOVIE_TIMING_COUNT];
|
||||
double max[MOVIE_TIMING_COUNT];
|
||||
size_t frames;
|
||||
} MovieTimingAccumulator;
|
||||
|
||||
static const char *const movie_timing_names[MOVIE_TIMING_COUNT] = {
|
||||
"catalog_mark", "catalog_load", "fast_clear", "splat", "fftw_zero_pack",
|
||||
"fftw_forward", "fftw_multiply", "fftw_inverse", "fftw_crop", "fftw_total",
|
||||
"tone_map", "output", "wait", "total"};
|
||||
|
||||
static void movie_timing_add(MovieTimingAccumulator *acc,
|
||||
const MovieFrameTiming *t) {
|
||||
const double values[MOVIE_TIMING_COUNT] = {
|
||||
t->catalog_mark_seconds, t->catalog_load_seconds,
|
||||
t->fast_clear_seconds, t->catalog_splat_seconds,
|
||||
t->fftw_zero_pack_seconds, t->fftw_forward_seconds,
|
||||
t->fftw_multiply_seconds, t->fftw_inverse_seconds,
|
||||
t->fftw_crop_seconds, t->fftw_total_seconds,
|
||||
t->tone_map_seconds, t->png_encode_write_seconds,
|
||||
t->writer_queue_wait_seconds, t->frame_total_seconds};
|
||||
for (size_t i = 0; i < MOVIE_TIMING_COUNT; ++i) {
|
||||
acc->sum[i] += values[i];
|
||||
if (values[i] > acc->max[i])
|
||||
acc->max[i] = values[i];
|
||||
}
|
||||
++acc->frames;
|
||||
}
|
||||
|
||||
static void report_movie_timing_frame(const Settings *s, size_t frame_id,
|
||||
const MovieFrameTiming *t) {
|
||||
if (!s->verbose)
|
||||
return;
|
||||
fprintf(stderr,
|
||||
"Movie frame %zu timing: catalog_mark=%.4f catalog_load=%.4f "
|
||||
"fast_clear=%.4f splat=%.4f fftw=%.4f tone_map=%.4f output=%.4f "
|
||||
"wait=%.4f total=%.4f\n",
|
||||
frame_id, t->catalog_mark_seconds, t->catalog_load_seconds,
|
||||
t->fast_clear_seconds, t->catalog_splat_seconds, t->fftw_total_seconds,
|
||||
t->tone_map_seconds, t->png_encode_write_seconds,
|
||||
t->writer_queue_wait_seconds, t->frame_total_seconds);
|
||||
}
|
||||
|
||||
static void report_movie_timing_summary(const MovieTimingAccumulator *acc) {
|
||||
if (acc->frames == 0)
|
||||
return;
|
||||
fprintf(stderr, "Movie timing total (%zu frames):", acc->frames);
|
||||
for (size_t i = 0; i < MOVIE_TIMING_COUNT; ++i)
|
||||
fprintf(stderr, " %s=%.4f", movie_timing_names[i], acc->sum[i]);
|
||||
fputc('\n', stderr);
|
||||
fprintf(stderr, "Movie timing avg (%zu frames):", acc->frames);
|
||||
for (size_t i = 0; i < MOVIE_TIMING_COUNT; ++i)
|
||||
fprintf(stderr, " %s=%.4f", movie_timing_names[i],
|
||||
acc->sum[i] / (double)acc->frames);
|
||||
fputc('\n', stderr);
|
||||
fprintf(stderr, "Movie timing max (%zu frames):", acc->frames);
|
||||
for (size_t i = 0; i < MOVIE_TIMING_COUNT; ++i)
|
||||
fprintf(stderr, " %s=%.4f", movie_timing_names[i], acc->max[i]);
|
||||
fputc('\n', stderr);
|
||||
}
|
||||
|
||||
/* Producer half of the async movie output: finishes every HDR-side step
|
||||
* (sensor bloom, tone map, optional mesh overlay, optional frame log) and
|
||||
* fills a job that carries only 8-bit RGB buffers. HDR can then be freed
|
||||
* immediately after submit. */
|
||||
static int prepare_movie_output_job(const Settings *s,
|
||||
const FrameLensMesh *mesh, double *hdr,
|
||||
int width, int height,
|
||||
const FrameOutputPaths *paths,
|
||||
size_t frame_id, size_t images,
|
||||
size_t stars, const char *note,
|
||||
const PsfSplatStats *psf_stats,
|
||||
MovieFrameTiming *timing,
|
||||
MovieOutputJob *job) {
|
||||
memset(job, 0, sizeof *job);
|
||||
job->frame_id = frame_id;
|
||||
job->width = width;
|
||||
job->height = height;
|
||||
job->images = images;
|
||||
job->catalog_stars = stars;
|
||||
job->draw_mesh = paths->draw_mesh;
|
||||
snprintf(job->output_path, sizeof job->output_path, "%s", paths->output_path);
|
||||
if (note != NULL)
|
||||
snprintf(job->note, sizeof job->note, "%s", note);
|
||||
if (paths->draw_mesh)
|
||||
snprintf(job->mesh_path, sizeof job->mesh_path, "%s", paths->mesh_path);
|
||||
if (psf_stats != NULL) {
|
||||
job->psf_stats = *psf_stats;
|
||||
job->has_psf_stats = 1;
|
||||
job->fast_mode = s->fast_mode;
|
||||
}
|
||||
if (apply_sensor_bloom(s, hdr, width, height))
|
||||
return -1;
|
||||
if (render_rgb8_image(s, hdr, width, height, &job->clean_rgb8, timing))
|
||||
return -1;
|
||||
if (paths->draw_mesh) {
|
||||
frame_draw_mesh(mesh, hdr, width, height, 0.5, 0.5);
|
||||
if (render_rgb8_image(s, hdr, width, height, &job->mesh_rgb8, timing)) {
|
||||
free(job->clean_rgb8);
|
||||
job->clean_rgb8 = NULL;
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int render_movie(const Settings *s, StarCatalog *catalog,
|
||||
const SpacetimeSource *spacetime) {
|
||||
ObserverTrack track = {0};
|
||||
Movie movie = {0};
|
||||
const GeodesicTraceConfig trace = trace_config(s);
|
||||
MovieOutputQueue output_queue;
|
||||
int output_queue_ready = 0;
|
||||
int result = -1;
|
||||
/* True end-to-end movie wall, starting before track load, ray tracing, and
|
||||
* the catalog union prefetch. The later pipeline wall measures only the
|
||||
* per-frame render + output stage. */
|
||||
const double movie_wall_start = omp_get_wtime();
|
||||
if (s->observer_track_path == NULL ||
|
||||
observer_track_load_csv(&track, s->observer_track_path) ||
|
||||
(s->movie_track_samples ? movie_init_track_samples(&movie, &track) :
|
||||
@@ -1143,7 +1354,44 @@ static int render_movie(const Settings *s, StarCatalog *catalog,
|
||||
fprintf(stderr, "Wrote lens-map movie: %s (%zu frames)\n",
|
||||
s->lens_map_output_path, movie.frame_count);
|
||||
}
|
||||
/* An all-sky movie prefetches the union of every finalized frame's requested
|
||||
* tiles exactly once. An in-memory catalog never uses tiles, so it must not
|
||||
* pay for a full triangle scan across every frame at all. */
|
||||
if (catalog->kind == STAR_CATALOG_ALL_SKY) {
|
||||
CatalogTileSet movie_tiles;
|
||||
catalog_tile_set_clear(&movie_tiles);
|
||||
const double mark_start = omp_get_wtime();
|
||||
for (size_t i = 0; i < movie.frame_count; ++i)
|
||||
(void)frame_mark_catalog_tiles(&movie.frames[i].mesh, &movie_tiles);
|
||||
const double mark_seconds = omp_get_wtime() - mark_start;
|
||||
CatalogPrefetchStats movie_prefetch = {0};
|
||||
const double load_start = omp_get_wtime();
|
||||
if (catalog_prefetch_tile_set(catalog, &movie_tiles,
|
||||
s->catalog_load_workers,
|
||||
CATALOG_PREFETCH_DEFAULT_BATCH_TILES,
|
||||
&movie_prefetch)) {
|
||||
fputs("Movie-level catalog prefetch failed.\n", stderr);
|
||||
goto done;
|
||||
}
|
||||
const double load_seconds = omp_get_wtime() - load_start;
|
||||
fprintf(stderr,
|
||||
"Movie catalog prefetch: %zu unique requested, %zu newly loaded "
|
||||
"(%zu stars), %zu unavailable in %.3f s (mark %.3f, load+commit "
|
||||
"%.3f); %d loader workers\n",
|
||||
catalog_tile_set_count(&movie_tiles),
|
||||
movie_prefetch.newly_loaded_tiles, movie_prefetch.newly_loaded_stars,
|
||||
movie_prefetch.unavailable_tiles, mark_seconds + load_seconds,
|
||||
mark_seconds, load_seconds, s->catalog_load_workers);
|
||||
}
|
||||
if (movie_output_queue_init(&output_queue, 2, &s->png)) {
|
||||
fputs("Movie output queue initialization failed.\n", stderr);
|
||||
goto done;
|
||||
}
|
||||
output_queue_ready = 1;
|
||||
const double endpoint_start = omp_get_wtime();
|
||||
MovieTimingAccumulator timing_acc = {0};
|
||||
for (size_t i = 0; i < movie.frame_count; ++i) {
|
||||
const double frame_start = omp_get_wtime();
|
||||
char output_path[PATH_MAX];
|
||||
FrameOutputPaths output_paths;
|
||||
if (frame_output_path(output_path, s, movie.frames[i].frame_id) ||
|
||||
@@ -1154,6 +1402,7 @@ static int render_movie(const Settings *s, StarCatalog *catalog,
|
||||
goto done;
|
||||
CatalogPrefetchStats prefetch = {0};
|
||||
PsfSplatStats psf_stats = {0};
|
||||
MovieFrameTiming frame_timing = {0};
|
||||
RenderProgress progress = {
|
||||
.verbose = s->verbose,
|
||||
.frame_id = movie.frames[i].frame_id,
|
||||
@@ -1166,25 +1415,49 @@ static int render_movie(const Settings *s, StarCatalog *catalog,
|
||||
s->psf_relative_tail,
|
||||
s->psf_min_y,
|
||||
spacetime_limits_render_workers_by_memory(spacetime),
|
||||
s->catalog_load_workers, &prefetch, &psf_stats,
|
||||
s->catalog_load_workers, FRAME_CATALOG_PREFETCH_ALREADY_COMPLETE,
|
||||
&prefetch, &psf_stats,
|
||||
&(FrameSplatProgress){report_splat_progress,
|
||||
s->verbose ? report_splat_worker_progress : NULL,
|
||||
&progress},
|
||||
s->fast_psf);
|
||||
s->fast_psf, &frame_timing);
|
||||
if (images == SIZE_MAX) {
|
||||
free(hdr);
|
||||
goto done;
|
||||
}
|
||||
const int write_result = write_frame_outputs(
|
||||
s, &movie.frames[i].mesh, hdr, s->width, s->height,
|
||||
s->horizontal_fov_deg, &output_paths, images, catalog->count, "");
|
||||
free(hdr);
|
||||
report_psf_splat(s, &psf_stats);
|
||||
if (write_result)
|
||||
MovieOutputJob job;
|
||||
if (prepare_movie_output_job(s, &movie.frames[i].mesh, hdr, s->width,
|
||||
s->height, &output_paths,
|
||||
movie.frames[i].frame_id, images,
|
||||
catalog->count, "", &psf_stats, &frame_timing,
|
||||
&job)) {
|
||||
free(hdr);
|
||||
goto done;
|
||||
}
|
||||
free(hdr);
|
||||
double queue_wait = 0.0;
|
||||
if (movie_output_queue_submit(&output_queue, &job, &queue_wait)) {
|
||||
/* The queue rejected the job; buffers still belong to this caller. */
|
||||
free(job.clean_rgb8);
|
||||
free(job.mesh_rgb8);
|
||||
goto done;
|
||||
}
|
||||
frame_timing.writer_queue_wait_seconds = queue_wait;
|
||||
frame_timing.frame_total_seconds = omp_get_wtime() - frame_start;
|
||||
report_movie_timing_frame(s, movie.frames[i].frame_id, &frame_timing);
|
||||
movie_timing_add(&timing_acc, &frame_timing);
|
||||
}
|
||||
result = 0;
|
||||
result = movie_output_queue_finish(&output_queue) == 0 ? 0 : -1;
|
||||
const double now = omp_get_wtime();
|
||||
report_movie_timing_summary(&timing_acc);
|
||||
movie_output_queue_report(&output_queue, stderr);
|
||||
fprintf(stderr,
|
||||
"Movie output pipeline wall: %.4f s (producer frame-time sum %.4f s)\n",
|
||||
now - endpoint_start, timing_acc.sum[MOVIE_TIMING_COUNT - 1]);
|
||||
fprintf(stderr, "Movie end-to-end wall: %.4f s\n", now - movie_wall_start);
|
||||
done:
|
||||
if (output_queue_ready)
|
||||
movie_output_queue_destroy(&output_queue);
|
||||
movie_destroy(&movie);
|
||||
observer_track_destroy(&track);
|
||||
return result;
|
||||
@@ -1227,7 +1500,58 @@ static int render_lens_map(const Settings *s, StarCatalog *catalog) {
|
||||
fast_psf_accumulator_report(&local_fast, stderr);
|
||||
fast_psf_accumulator_set_verbose(&local_fast, s->verbose);
|
||||
}
|
||||
/* A multi-frame imported map runs the same movie-level union prefetch and
|
||||
* bounded async output queue as a tracked observer movie: one immutable tile
|
||||
* cache, overlapping PNG writes. */
|
||||
FrameCatalogPrefetchMode map_prefetch_mode = FRAME_CATALOG_PREFETCH_FRAME;
|
||||
if (map.frame_count > 1) {
|
||||
if (catalog->kind == STAR_CATALOG_ALL_SKY) {
|
||||
CatalogTileSet map_tiles;
|
||||
catalog_tile_set_clear(&map_tiles);
|
||||
const double mark_start = omp_get_wtime();
|
||||
for (size_t i = 0; i < map.frame_count; ++i)
|
||||
(void)frame_mark_catalog_tiles(&map.frames[i].mesh, &map_tiles);
|
||||
const double mark_seconds = omp_get_wtime() - mark_start;
|
||||
CatalogPrefetchStats movie_prefetch = {0};
|
||||
const double load_start = omp_get_wtime();
|
||||
if (catalog_prefetch_tile_set(catalog, &map_tiles,
|
||||
s->catalog_load_workers,
|
||||
CATALOG_PREFETCH_DEFAULT_BATCH_TILES,
|
||||
&movie_prefetch)) {
|
||||
fast_psf_accumulator_destroy(&local_fast);
|
||||
lens_map_destroy(&map);
|
||||
return -1;
|
||||
}
|
||||
const double load_seconds = omp_get_wtime() - load_start;
|
||||
fprintf(stderr,
|
||||
"Movie catalog prefetch: %zu unique requested, %zu newly loaded "
|
||||
"(%zu stars), %zu unavailable in %.3f s (mark %.3f, load+commit "
|
||||
"%.3f); %d loader workers\n",
|
||||
catalog_tile_set_count(&map_tiles),
|
||||
movie_prefetch.newly_loaded_tiles,
|
||||
movie_prefetch.newly_loaded_stars,
|
||||
movie_prefetch.unavailable_tiles, mark_seconds + load_seconds,
|
||||
mark_seconds, load_seconds, s->catalog_load_workers);
|
||||
}
|
||||
map_prefetch_mode = FRAME_CATALOG_PREFETCH_ALREADY_COMPLETE;
|
||||
}
|
||||
#ifndef PSF_BACKEND_DUMMY
|
||||
MovieOutputQueue output_queue;
|
||||
int output_queue_ready = 0;
|
||||
if (map.frame_count > 1) {
|
||||
if (movie_output_queue_init(&output_queue, 2, &s->png)) {
|
||||
fputs("Movie output queue initialization failed.\n", stderr);
|
||||
fast_psf_accumulator_destroy(&local_fast);
|
||||
lens_map_destroy(&map);
|
||||
return -1;
|
||||
}
|
||||
output_queue_ready = 1;
|
||||
}
|
||||
MovieTimingAccumulator timing_acc = {0};
|
||||
const double pipeline_start = omp_get_wtime();
|
||||
#endif
|
||||
for (size_t i = 0; i < map.frame_count; ++i) {
|
||||
const double frame_start = omp_get_wtime();
|
||||
const char *output_path = s->output_path;
|
||||
char movie_path[PATH_MAX];
|
||||
#ifndef PSF_BACKEND_DUMMY
|
||||
@@ -1256,6 +1580,7 @@ static int render_lens_map(const Settings *s, StarCatalog *catalog) {
|
||||
if (hdr == NULL) { result = -1; break; }
|
||||
CatalogPrefetchStats prefetch = {0};
|
||||
PsfSplatStats psf_stats = {0};
|
||||
MovieFrameTiming frame_timing = {0};
|
||||
RenderProgress progress = {.verbose = s->verbose,
|
||||
.frame_id = (size_t)map.frames[i].frame_id,
|
||||
.prefetch = &prefetch,
|
||||
@@ -1265,24 +1590,50 @@ static int render_lens_map(const Settings *s, StarCatalog *catalog) {
|
||||
&map.frames[i].mesh, catalog, hdr, map.width, map.height, s->exposure,
|
||||
&s->psf, &s->psf_cache, s->max_magnification, s->max_cache_psf_flux,
|
||||
s->psf_relative_tail, s->psf_min_y, 0, s->catalog_load_workers,
|
||||
&prefetch, &psf_stats,
|
||||
map_prefetch_mode, &prefetch, &psf_stats,
|
||||
&(FrameSplatProgress){report_splat_progress,
|
||||
s->verbose ? report_splat_worker_progress : NULL,
|
||||
&progress},
|
||||
fast);
|
||||
fast, &frame_timing);
|
||||
if (images == SIZE_MAX) {
|
||||
free(hdr);
|
||||
result = -1;
|
||||
break;
|
||||
}
|
||||
#ifndef PSF_BACKEND_DUMMY
|
||||
const int write_result = write_frame_outputs(
|
||||
s, &map.frames[i].mesh, hdr, map.width, map.height,
|
||||
map.horizontal_fov_deg, &output_paths, images, catalog->count,
|
||||
"; imported lens map");
|
||||
free(hdr);
|
||||
report_psf_splat(s, &psf_stats);
|
||||
if (write_result) { result = -1; break; }
|
||||
if (output_queue_ready) {
|
||||
MovieOutputJob job;
|
||||
if (prepare_movie_output_job(s, &map.frames[i].mesh, hdr, map.width,
|
||||
map.height, &output_paths,
|
||||
(size_t)map.frames[i].frame_id, images,
|
||||
catalog->count, "; imported lens map",
|
||||
&psf_stats, &frame_timing, &job)) {
|
||||
free(hdr);
|
||||
result = -1;
|
||||
break;
|
||||
}
|
||||
free(hdr);
|
||||
double queue_wait = 0.0;
|
||||
if (movie_output_queue_submit(&output_queue, &job, &queue_wait)) {
|
||||
free(job.clean_rgb8);
|
||||
free(job.mesh_rgb8);
|
||||
result = -1;
|
||||
break;
|
||||
}
|
||||
frame_timing.writer_queue_wait_seconds = queue_wait;
|
||||
frame_timing.frame_total_seconds = omp_get_wtime() - frame_start;
|
||||
report_movie_timing_frame(s, (size_t)map.frames[i].frame_id,
|
||||
&frame_timing);
|
||||
movie_timing_add(&timing_acc, &frame_timing);
|
||||
} else {
|
||||
const int write_result = write_frame_outputs(
|
||||
s, &map.frames[i].mesh, hdr, map.width, map.height,
|
||||
map.horizontal_fov_deg, &output_paths, images, catalog->count,
|
||||
"; imported lens map", NULL);
|
||||
free(hdr);
|
||||
report_psf_splat(s, &psf_stats);
|
||||
if (write_result) { result = -1; break; }
|
||||
}
|
||||
#else
|
||||
if (s->draw_mesh)
|
||||
fputs("Dummy PSF backend ignores --draw-mesh.\n", stderr);
|
||||
@@ -1293,6 +1644,17 @@ static int render_lens_map(const Settings *s, StarCatalog *catalog) {
|
||||
report_psf_splat(s, &psf_stats);
|
||||
#endif
|
||||
}
|
||||
#ifndef PSF_BACKEND_DUMMY
|
||||
if (output_queue_ready) {
|
||||
if (movie_output_queue_finish(&output_queue))
|
||||
result = -1;
|
||||
report_movie_timing_summary(&timing_acc);
|
||||
movie_output_queue_report(&output_queue, stderr);
|
||||
fprintf(stderr, "Movie output pipeline wall: %.4f s\n",
|
||||
omp_get_wtime() - pipeline_start);
|
||||
movie_output_queue_destroy(&output_queue);
|
||||
}
|
||||
#endif
|
||||
fast_psf_accumulator_destroy(&local_fast);
|
||||
lens_map_destroy(&map);
|
||||
return result;
|
||||
@@ -1388,6 +1750,30 @@ int main(int argc, char **argv) {
|
||||
fputs("--fast-supersample must be between 1 and 8.\n", stderr);
|
||||
return 2;
|
||||
}
|
||||
#ifndef FAST_PSF_FFTW
|
||||
if (settings.fast_fftw_plan_mode != FAST_PSF_FFTW_PLAN_ESTIMATE) {
|
||||
fputs("--fast-fftw-plan and --fast-fftw-wisdom require the CPU PSF "
|
||||
"backend built with FFTW.\n",
|
||||
stderr);
|
||||
return 2;
|
||||
}
|
||||
#else
|
||||
if (fast_psf_fftw_configure(settings.fast_fftw_plan_mode,
|
||||
settings.fast_fftw_wisdom_path))
|
||||
return 2;
|
||||
#endif
|
||||
#ifndef ENABLE_PNG
|
||||
if (settings.png.compression_level != -1) {
|
||||
fputs("--png-compression-level requires a libpng build.\n", stderr);
|
||||
return 2;
|
||||
}
|
||||
#endif
|
||||
if (settings.movie_output_workers == 1 && settings.frames_dir == NULL) {
|
||||
fputs("--movie-output-workers reserve-one requires a movie output "
|
||||
"sequence (--frames-dir).\n",
|
||||
stderr);
|
||||
return 2;
|
||||
}
|
||||
if (settings.fast_mode)
|
||||
fputs("Fast mode is a preview approximation; --max-cache-psf-flux is "
|
||||
"ignored and one global kernel is used for every event.\n",
|
||||
@@ -1486,6 +1872,17 @@ int main(int argc, char **argv) {
|
||||
return 1;
|
||||
}
|
||||
fprintf(stderr, "Blackbody backend: %s\n", blackbody_backend_name());
|
||||
/* `reserve-one` must be applied before any fast-mode accumulator is built:
|
||||
* the FFTW plans in fast_psf_accumulator_init() snapshot omp_get_max_threads()
|
||||
* at creation time, so shrinking the pool later would leave the plans (and
|
||||
* their worker count) unchanged. It only applies when a movie output queue
|
||||
* will actually run (observer movie or multi-frame imported map). */
|
||||
if (settings.movie_output_workers == 1 && settings.frames_dir != NULL) {
|
||||
int render_workers = omp_get_max_threads() - 1;
|
||||
if (render_workers < 1)
|
||||
render_workers = 1;
|
||||
omp_set_num_threads(render_workers);
|
||||
}
|
||||
FastPsfAccumulator fast_accumulator = {0};
|
||||
if (settings.fast_mode && settings.lens_map_input_path == NULL) {
|
||||
if (fast_psf_accumulator_init(&fast_accumulator, settings.width,
|
||||
|
||||
@@ -0,0 +1,223 @@
|
||||
#include "movie_output.h"
|
||||
|
||||
#include <omp.h>
|
||||
#include <pthread.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
/* Default writer: clean image, then optional mesh overlay, then the frame's
|
||||
* comment lines. This is the only place that prints a movie frame's log, so
|
||||
* frames stay in submit order; individual lines are not lock-protected and may
|
||||
* still interleave with concurrent producer --verbose output. */
|
||||
static int movie_output_default_write(void *context, const MovieOutputJob *job,
|
||||
const PngWriteSettings *settings) {
|
||||
(void)context;
|
||||
if (write_rgb8_image(job->output_path, job->clean_rgb8, job->width,
|
||||
job->height, settings))
|
||||
return -1;
|
||||
fprintf(stderr, "Rendered %zu images from %zu catalog stars to %s (ok%s)\n",
|
||||
job->images, job->catalog_stars, job->output_path, job->note);
|
||||
if (job->draw_mesh && job->mesh_rgb8 != NULL) {
|
||||
if (write_rgb8_image(job->mesh_path, job->mesh_rgb8, job->width,
|
||||
job->height, settings))
|
||||
return -1;
|
||||
fprintf(stderr, "Wrote mesh overlay image: %s\n", job->mesh_path);
|
||||
}
|
||||
if (job->has_psf_stats) {
|
||||
if (job->fast_mode)
|
||||
fprintf(stderr,
|
||||
"Fast PSF splats: deposited %zu, wing-clipped %zu, discarded "
|
||||
"below min-Y %zu\n",
|
||||
job->psf_stats.cached_splats, job->psf_stats.cached_wing_clipped,
|
||||
job->psf_stats.discarded_below_min_y);
|
||||
else
|
||||
psf_kernel_cache_report(NULL, &job->psf_stats, stderr);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void *movie_output_writer_main(void *opaque) {
|
||||
MovieOutputQueue *queue = opaque;
|
||||
pthread_mutex_lock(&queue->mutex);
|
||||
for (;;) {
|
||||
while (queue->count == 0 && !queue->producer_done)
|
||||
pthread_cond_wait(&queue->not_empty, &queue->mutex);
|
||||
if (queue->count == 0 && queue->producer_done)
|
||||
break;
|
||||
const size_t slot = queue->head;
|
||||
const MovieOutputJob job = queue->jobs[slot];
|
||||
/* Ownership moved into the local copy; clear the slot so destroy() cannot
|
||||
* free the same buffers a second time. */
|
||||
queue->jobs[slot].clean_rgb8 = NULL;
|
||||
queue->jobs[slot].mesh_rgb8 = NULL;
|
||||
queue->head = (queue->head + 1) % queue->capacity;
|
||||
--queue->count;
|
||||
pthread_cond_signal(&queue->not_full);
|
||||
pthread_mutex_unlock(&queue->mutex);
|
||||
|
||||
const double write_start = omp_get_wtime();
|
||||
const int result = queue->write(queue->write_context, &job, &queue->settings);
|
||||
const double write_seconds = omp_get_wtime() - write_start;
|
||||
|
||||
pthread_mutex_lock(&queue->mutex);
|
||||
queue->writer_total_seconds += write_seconds;
|
||||
if (write_seconds > queue->writer_max_seconds)
|
||||
queue->writer_max_seconds = write_seconds;
|
||||
++queue->written_jobs;
|
||||
if (result != 0) {
|
||||
if (!queue->failed)
|
||||
queue->failed = 1;
|
||||
pthread_cond_broadcast(&queue->not_full);
|
||||
}
|
||||
pthread_mutex_unlock(&queue->mutex);
|
||||
free(job.clean_rgb8);
|
||||
free(job.mesh_rgb8);
|
||||
pthread_mutex_lock(&queue->mutex);
|
||||
}
|
||||
pthread_mutex_unlock(&queue->mutex);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
int movie_output_queue_init(MovieOutputQueue *queue, size_t capacity,
|
||||
const PngWriteSettings *settings) {
|
||||
if (queue == NULL)
|
||||
return -1;
|
||||
memset(queue, 0, sizeof *queue);
|
||||
if (capacity < 1)
|
||||
capacity = 1;
|
||||
queue->jobs = calloc(capacity, sizeof *queue->jobs);
|
||||
if (queue->jobs == NULL)
|
||||
return -1;
|
||||
queue->capacity = capacity;
|
||||
queue->settings = settings != NULL ? *settings : (PngWriteSettings){-1};
|
||||
queue->write = movie_output_default_write;
|
||||
/* Destroy exactly what was initialized if a later step fails. */
|
||||
if (pthread_mutex_init(&queue->mutex, NULL) != 0) {
|
||||
free(queue->jobs);
|
||||
queue->jobs = NULL;
|
||||
return -1;
|
||||
}
|
||||
if (pthread_cond_init(&queue->not_full, NULL) != 0) {
|
||||
pthread_mutex_destroy(&queue->mutex);
|
||||
free(queue->jobs);
|
||||
queue->jobs = NULL;
|
||||
return -1;
|
||||
}
|
||||
if (pthread_cond_init(&queue->not_empty, NULL) != 0) {
|
||||
pthread_cond_destroy(&queue->not_full);
|
||||
pthread_mutex_destroy(&queue->mutex);
|
||||
free(queue->jobs);
|
||||
queue->jobs = NULL;
|
||||
return -1;
|
||||
}
|
||||
if (pthread_create(&queue->writer, NULL, movie_output_writer_main, queue) != 0) {
|
||||
pthread_cond_destroy(&queue->not_empty);
|
||||
pthread_cond_destroy(&queue->not_full);
|
||||
pthread_mutex_destroy(&queue->mutex);
|
||||
free(queue->jobs);
|
||||
queue->jobs = NULL;
|
||||
return -1;
|
||||
}
|
||||
queue->thread_started = 1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
void movie_output_queue_set_writer(MovieOutputQueue *queue,
|
||||
MovieOutputWriteFn writer, void *context) {
|
||||
if (queue == NULL || writer == NULL)
|
||||
return;
|
||||
queue->write = writer;
|
||||
queue->write_context = context;
|
||||
}
|
||||
|
||||
int movie_output_queue_submit(MovieOutputQueue *queue, MovieOutputJob *job,
|
||||
double *wait_seconds) {
|
||||
if (queue == NULL || job == NULL)
|
||||
return -1;
|
||||
double waited = 0.0;
|
||||
pthread_mutex_lock(&queue->mutex);
|
||||
while (queue->count == queue->capacity) {
|
||||
if (queue->failed)
|
||||
break;
|
||||
const double wait_start = omp_get_wtime();
|
||||
pthread_cond_wait(&queue->not_full, &queue->mutex);
|
||||
waited += omp_get_wtime() - wait_start;
|
||||
}
|
||||
if (queue->failed) {
|
||||
pthread_mutex_unlock(&queue->mutex);
|
||||
if (wait_seconds != NULL)
|
||||
*wait_seconds += waited;
|
||||
return -1;
|
||||
}
|
||||
queue->jobs[queue->tail] = *job;
|
||||
queue->tail = (queue->tail + 1) % queue->capacity;
|
||||
++queue->count;
|
||||
pthread_cond_signal(&queue->not_empty);
|
||||
pthread_mutex_unlock(&queue->mutex);
|
||||
if (wait_seconds != NULL)
|
||||
*wait_seconds += waited;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int movie_output_queue_failed(const MovieOutputQueue *queue) {
|
||||
if (queue == NULL)
|
||||
return 1;
|
||||
MovieOutputQueue *mutable_queue = (MovieOutputQueue *)queue;
|
||||
pthread_mutex_lock(&mutable_queue->mutex);
|
||||
const int failed = mutable_queue->failed;
|
||||
pthread_mutex_unlock(&mutable_queue->mutex);
|
||||
return failed;
|
||||
}
|
||||
|
||||
int movie_output_queue_finish(MovieOutputQueue *queue) {
|
||||
if (queue == NULL)
|
||||
return -1;
|
||||
pthread_mutex_lock(&queue->mutex);
|
||||
queue->producer_done = 1;
|
||||
pthread_cond_signal(&queue->not_empty);
|
||||
const int need_join = queue->thread_started && !queue->joined;
|
||||
if (need_join)
|
||||
queue->joined = 1;
|
||||
pthread_mutex_unlock(&queue->mutex);
|
||||
if (need_join) {
|
||||
const double drain_start = omp_get_wtime();
|
||||
pthread_join(queue->writer, NULL);
|
||||
pthread_mutex_lock(&queue->mutex);
|
||||
queue->drain_seconds = omp_get_wtime() - drain_start;
|
||||
pthread_mutex_unlock(&queue->mutex);
|
||||
}
|
||||
return movie_output_queue_failed(queue) ? -1 : 0;
|
||||
}
|
||||
|
||||
void movie_output_queue_report(const MovieOutputQueue *queue, FILE *stream) {
|
||||
if (queue == NULL || stream == NULL)
|
||||
return;
|
||||
const double average = queue->written_jobs > 0
|
||||
? queue->writer_total_seconds /
|
||||
(double)queue->written_jobs
|
||||
: 0.0;
|
||||
fprintf(stream,
|
||||
"Movie writer summary: jobs=%zu total=%.4f avg=%.4f max=%.4f "
|
||||
"drain=%.4f s\n",
|
||||
queue->written_jobs, queue->writer_total_seconds, average,
|
||||
queue->writer_max_seconds, queue->drain_seconds);
|
||||
}
|
||||
|
||||
void movie_output_queue_destroy(MovieOutputQueue *queue) {
|
||||
if (queue == NULL)
|
||||
return;
|
||||
if (queue->thread_started && !queue->joined)
|
||||
(void)movie_output_queue_finish(queue);
|
||||
/* Any job that was accepted but never written is released here. */
|
||||
if (queue->jobs != NULL)
|
||||
for (size_t i = 0; i < queue->capacity; ++i) {
|
||||
free(queue->jobs[i].clean_rgb8);
|
||||
free(queue->jobs[i].mesh_rgb8);
|
||||
}
|
||||
pthread_cond_destroy(&queue->not_empty);
|
||||
pthread_cond_destroy(&queue->not_full);
|
||||
pthread_mutex_destroy(&queue->mutex);
|
||||
free(queue->jobs);
|
||||
queue->jobs = NULL;
|
||||
}
|
||||
@@ -0,0 +1,104 @@
|
||||
#ifndef MOVIE_OUTPUT_H
|
||||
#define MOVIE_OUTPUT_H
|
||||
|
||||
#include "optics.h"
|
||||
|
||||
#include <limits.h>
|
||||
#include <pthread.h>
|
||||
#include <stddef.h>
|
||||
|
||||
/* Bounded, single-producer/single-writer movie output queue.
|
||||
*
|
||||
* The producer (the render loop) performs all HDR work and the tone map before
|
||||
* submitting; a job therefore carries finished 8-bit RGB buffers, never a
|
||||
* double HDR framebuffer. The writer thread encodes/writes them in submit
|
||||
* order while the producer renders the next frame.
|
||||
*
|
||||
* Ownership contract for submit(): on success the queue owns clean_rgb8 and
|
||||
* mesh_rgb8 and frees them after writing; on failure they remain owned by the
|
||||
* caller.
|
||||
*
|
||||
* `capacity` bounds the queued jobs only; the writer may additionally hold one
|
||||
* already-popped job, so the true in-memory bound is capacity + 1 jobs. With
|
||||
* the production default of 2 that is three jobs. */
|
||||
|
||||
typedef struct {
|
||||
size_t frame_id;
|
||||
char output_path[PATH_MAX];
|
||||
char mesh_path[PATH_MAX];
|
||||
int draw_mesh;
|
||||
|
||||
unsigned char *clean_rgb8;
|
||||
unsigned char *mesh_rgb8; /* NULL when draw_mesh is false */
|
||||
int width;
|
||||
int height;
|
||||
|
||||
size_t images;
|
||||
size_t catalog_stars;
|
||||
PsfSplatStats psf_stats;
|
||||
int fast_mode;
|
||||
int has_psf_stats;
|
||||
|
||||
/* Extra suffix appended to the "Rendered ..." line (for example
|
||||
* "; imported lens map"). The writer emits a frame's log lines in submit
|
||||
* order; because there is no shared log lock, an individual line may still
|
||||
* interleave with concurrent producer --verbose output. */
|
||||
char note[64];
|
||||
} MovieOutputJob;
|
||||
|
||||
/* Optional custom writer. Returns 0 on success; the default writer writes
|
||||
* clean_rgb8 to output_path and, when draw_mesh is set, mesh_rgb8 to
|
||||
* mesh_path, then prints the "Rendered ... (<note>)" and PSF lines. The queue
|
||||
* owns and frees clean_rgb8/mesh_rgb8 after the writer returns. */
|
||||
typedef int (*MovieOutputWriteFn)(void *context, const MovieOutputJob *job,
|
||||
const PngWriteSettings *settings);
|
||||
|
||||
typedef struct MovieOutputQueue {
|
||||
pthread_mutex_t mutex;
|
||||
pthread_cond_t not_full;
|
||||
pthread_cond_t not_empty;
|
||||
pthread_t writer;
|
||||
|
||||
MovieOutputJob *jobs;
|
||||
size_t capacity;
|
||||
size_t head;
|
||||
size_t tail;
|
||||
size_t count;
|
||||
|
||||
int producer_done;
|
||||
int failed;
|
||||
int joined;
|
||||
int thread_started;
|
||||
|
||||
PngWriteSettings settings;
|
||||
MovieOutputWriteFn write;
|
||||
void *write_context;
|
||||
|
||||
/* Writer-side accounting, updated by the writer thread under the mutex and
|
||||
* readable after finish(). */
|
||||
double writer_total_seconds;
|
||||
double writer_max_seconds;
|
||||
double drain_seconds; /* producer_done -> writer join, measured by finish() */
|
||||
size_t written_jobs;
|
||||
} MovieOutputQueue;
|
||||
|
||||
/* capacity is clamped to at least 1; the production default is 2. */
|
||||
int movie_output_queue_init(MovieOutputQueue *queue, size_t capacity,
|
||||
const PngWriteSettings *settings);
|
||||
/* Test seam: replace the writer before the first submit. */
|
||||
void movie_output_queue_set_writer(MovieOutputQueue *queue,
|
||||
MovieOutputWriteFn writer, void *context);
|
||||
int movie_output_queue_submit(MovieOutputQueue *queue, MovieOutputJob *job,
|
||||
double *wait_seconds);
|
||||
/* Returns -1 if the writer has already recorded an error. */
|
||||
int movie_output_queue_failed(const MovieOutputQueue *queue);
|
||||
/* Signals end of input, drains every accepted job, then joins the writer.
|
||||
* Records drain_seconds around the join. Safe to call more than once. */
|
||||
int movie_output_queue_finish(MovieOutputQueue *queue);
|
||||
/* Prints jobs, total/average/max writer time, and final drain time. Call
|
||||
* after finish() for complete accounting. */
|
||||
void movie_output_queue_report(const MovieOutputQueue *queue, FILE *stream);
|
||||
/* Finish if needed, join, and release synchronization resources. */
|
||||
void movie_output_queue_destroy(MovieOutputQueue *queue);
|
||||
|
||||
#endif
|
||||
+77
-34
@@ -534,7 +534,8 @@ int fast_psf_accumulator_init(FastPsfAccumulator *accumulator, int width,
|
||||
.use_reference = use_reference,
|
||||
.weights = weights,
|
||||
.row_span = row_span,
|
||||
.buffer = buffer};
|
||||
.buffer = buffer,
|
||||
.buffer_state = FAST_PSF_BUFFER_CLEAN};
|
||||
/* k[m] is the pixel-area integral of I(z/N; alpha, beta) over ss cell m,
|
||||
* i.e. the final-normalized Moffat evaluated at the supersampled scale.
|
||||
* Its total weight is ~N^2, so the N x N box average restores unit flux
|
||||
@@ -577,7 +578,8 @@ void fast_psf_accumulator_clear(FastPsfAccumulator *accumulator)
|
||||
return;
|
||||
memset(accumulator->buffer, 0,
|
||||
(size_t)accumulator->supersampled_width *
|
||||
accumulator->supersampled_height * 3 * sizeof(double));
|
||||
accumulator->supersampled_height * 3 * sizeof(double));
|
||||
accumulator->buffer_state = FAST_PSF_BUFFER_CLEAN;
|
||||
}
|
||||
|
||||
int fast_psf_accumulator_deposit(FastPsfAccumulator *accumulator, double x,
|
||||
@@ -593,6 +595,7 @@ int fast_psf_accumulator_deposit(FastPsfAccumulator *accumulator, double x,
|
||||
accumulator->relative_tail_fraction, accumulator->min_y);
|
||||
if (support_radius == 0.0)
|
||||
return 3;
|
||||
accumulator->buffer_state = FAST_PSF_BUFFER_DIRTY;
|
||||
const int wing_clipped =
|
||||
support_radius >
|
||||
(double)accumulator->radius_pixels / accumulator->supersample
|
||||
@@ -689,9 +692,17 @@ int fast_psf_accumulator_resolve(FastPsfAccumulator *accumulator,
|
||||
if (accumulator->fftw_enabled && accumulator->fftw != NULL) {
|
||||
FastPsfFftwFrameTiming timing = {0};
|
||||
const double start = omp_get_wtime();
|
||||
if (fast_psf_fftw_resolve(accumulator->fftw, accumulator->buffer, hdr,
|
||||
&timing))
|
||||
/* The consuming resolve zeroes the shared source buffer as it packs,
|
||||
* so a successful call leaves the accumulator CLEAN for the next
|
||||
* frame with no separate serial clear. A failed call still returns
|
||||
* an all-zero buffer (fast_psf_fftw_resolve_and_clear guarantees it). */
|
||||
if (fast_psf_fftw_resolve_and_clear(accumulator->fftw,
|
||||
accumulator->buffer, hdr, &timing)) {
|
||||
accumulator->buffer_state = FAST_PSF_BUFFER_CLEAN;
|
||||
return -1;
|
||||
}
|
||||
accumulator->buffer_state = FAST_PSF_BUFFER_CLEAN;
|
||||
accumulator->fftw_last_timing = timing;
|
||||
accumulator->fftw_frame_seconds = omp_get_wtime() - start;
|
||||
if (accumulator->verbose)
|
||||
fprintf(stderr,
|
||||
@@ -818,50 +829,86 @@ unsigned char tone_map_srgb8_channel(double hdr_value,
|
||||
return (unsigned char)lround(255.0 * clamp(display, 0.0, 1.0));
|
||||
}
|
||||
|
||||
int tone_map_srgb8_image(const double *hdr, unsigned char *rgb8, int width,
|
||||
int height, const ToneMapSettings *settings,
|
||||
int worker_count)
|
||||
{
|
||||
if (hdr == NULL || rgb8 == NULL || width <= 0 || height <= 0)
|
||||
return -1;
|
||||
const int threads = worker_count > 0 ? worker_count : 1;
|
||||
/* Row-outer and statically scheduled: each output byte is written by one
|
||||
* worker, and the byte stream is identical for any worker_count. */
|
||||
#pragma omp parallel for schedule(static) num_threads(threads) if (threads > 1)
|
||||
for (int row = 0; row < height; ++row) {
|
||||
const size_t base = (size_t)row * (size_t)width * 3;
|
||||
for (int column = 0; column < width * 3; ++column)
|
||||
rgb8[base + (size_t)column] =
|
||||
tone_map_srgb8_channel(hdr[base + (size_t)column], settings);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
#ifndef ENABLE_PNG
|
||||
static int write_tonemapped_ppm(const char *path, const double *hdr, int width,
|
||||
int height, const ToneMapSettings *settings)
|
||||
static int write_ppm_rgb8(const char *path, const unsigned char *rgb8, int width,
|
||||
int height)
|
||||
{
|
||||
FILE *file = fopen(path, "wb");
|
||||
if (file == NULL) return -1;
|
||||
fprintf(file, "P6\n%d %d\n255\n", width, height);
|
||||
for (int i = 0; i < width * height * 3; ++i) {
|
||||
const unsigned char value = tone_map_srgb8_channel(hdr[i], settings);
|
||||
if (fwrite(&value, 1, 1, file) != 1) { fclose(file); return -1; }
|
||||
}
|
||||
return fclose(file) == 0 ? 0 : -1;
|
||||
const size_t bytes = (size_t)width * height * 3;
|
||||
int ok = fprintf(file, "P6\n%d %d\n255\n", width, height) >= 0 &&
|
||||
fwrite(rgb8, 1, bytes, file) == bytes;
|
||||
if (fclose(file) != 0)
|
||||
ok = 0;
|
||||
return ok ? 0 : -1;
|
||||
}
|
||||
#endif
|
||||
|
||||
int write_rgb8_image(const char *path, const unsigned char *rgb8, int width,
|
||||
int height, const PngWriteSettings *settings)
|
||||
{
|
||||
if (path == NULL || rgb8 == NULL || width <= 0 || height <= 0)
|
||||
return -1;
|
||||
const size_t path_length = strlen(path);
|
||||
#ifdef ENABLE_PNG
|
||||
static int write_tonemapped_png(const char *path, const double *hdr, int width,
|
||||
int height, const ToneMapSettings *settings)
|
||||
if (path_length >= 4 && strcmp(path + path_length - 4, ".png") == 0)
|
||||
return write_png_rgb8(path, rgb8, width, height, settings);
|
||||
fputs("PNG output is enabled; use a .png output path.\n", stderr);
|
||||
return -1;
|
||||
#else
|
||||
(void)settings;
|
||||
if (path_length >= 4 && strcmp(path + path_length - 4, ".ppm") == 0)
|
||||
return write_ppm_rgb8(path, rgb8, width, height);
|
||||
fputs("PNG output is unavailable; use a .ppm output path or rebuild with libpng.\n",
|
||||
stderr);
|
||||
return -1;
|
||||
#endif
|
||||
}
|
||||
|
||||
#ifdef ENABLE_PNG
|
||||
int write_png_rgb8(const char *path, const unsigned char *rgb8, int width,
|
||||
int height, const PngWriteSettings *settings)
|
||||
{
|
||||
FILE *file = fopen(path, "wb");
|
||||
png_structp png = NULL;
|
||||
png_infop info = NULL;
|
||||
unsigned char *pixels = NULL;
|
||||
int result = -1;
|
||||
if (file == NULL) return -1;
|
||||
png = png_create_write_struct(PNG_LIBPNG_VER_STRING, NULL, NULL, NULL);
|
||||
if (png == NULL) goto done;
|
||||
info = png_create_info_struct(png);
|
||||
if (info == NULL || setjmp(png_jmpbuf(png))) goto done;
|
||||
pixels = malloc((size_t)width * height * 3);
|
||||
if (pixels == NULL) goto done;
|
||||
for (int i = 0; i < width * height * 3; ++i)
|
||||
pixels[i] = tone_map_srgb8_channel(hdr[i], settings);
|
||||
if (settings != NULL && settings->compression_level >= 0)
|
||||
png_set_compression_level(png, settings->compression_level);
|
||||
png_init_io(png, file);
|
||||
png_set_IHDR(png, info, (png_uint_32)width, (png_uint_32)height, 8,
|
||||
PNG_COLOR_TYPE_RGB, PNG_INTERLACE_NONE,
|
||||
PNG_COMPRESSION_TYPE_DEFAULT, PNG_FILTER_TYPE_DEFAULT);
|
||||
png_write_info(png, info);
|
||||
for (int row = 0; row < height; ++row)
|
||||
png_write_row(png, &pixels[(size_t)row * width * 3]);
|
||||
png_write_row(png, &rgb8[(size_t)row * width * 3]);
|
||||
png_write_end(png, info);
|
||||
result = 0;
|
||||
done:
|
||||
free(pixels);
|
||||
png_destroy_write_struct(&png, &info);
|
||||
if (fclose(file) != 0) result = -1;
|
||||
return result;
|
||||
@@ -871,19 +918,15 @@ done:
|
||||
int write_tonemapped_image(const char *path, const double *hdr, int width,
|
||||
int height, const ToneMapSettings *settings)
|
||||
{
|
||||
const size_t path_length = strlen(path);
|
||||
#ifdef ENABLE_PNG
|
||||
if (path_length >= 4 && strcmp(path + path_length - 4, ".png") == 0)
|
||||
return write_tonemapped_png(path, hdr, width, height, settings);
|
||||
fputs("PNG output is enabled; use a .png output path.\n", stderr);
|
||||
return -1;
|
||||
#else
|
||||
if (path_length >= 4 && strcmp(path + path_length - 4, ".ppm") == 0)
|
||||
return write_tonemapped_ppm(path, hdr, width, height, settings);
|
||||
fputs("PNG output is unavailable; use a .ppm output path or rebuild with libpng.\n",
|
||||
stderr);
|
||||
return -1;
|
||||
#endif
|
||||
if (path == NULL || hdr == NULL || width <= 0 || height <= 0)
|
||||
return -1;
|
||||
unsigned char *rgb8 = malloc((size_t)width * height * 3);
|
||||
if (rgb8 == NULL) return -1;
|
||||
int result = -1;
|
||||
if (tone_map_srgb8_image(hdr, rgb8, width, height, settings, 0) == 0)
|
||||
result = write_rgb8_image(path, rgb8, width, height, NULL);
|
||||
free(rgb8);
|
||||
return result;
|
||||
}
|
||||
|
||||
#ifdef ENABLE_HDR_OUTPUT
|
||||
|
||||
+39
-1
@@ -1,6 +1,8 @@
|
||||
#ifndef OPTICS_H
|
||||
#define OPTICS_H
|
||||
|
||||
#include "fast_psf_fftw.h"
|
||||
|
||||
#include <stddef.h>
|
||||
#include <stdio.h>
|
||||
|
||||
@@ -55,6 +57,15 @@ typedef struct FastPsfFftwState FastPsfFftwState;
|
||||
* benchmarks/fast_mode_deposit_2026-09-18.md). The kernel is the pixel-area
|
||||
* integral of the Moffat with alpha scaled by `supersample` and the same beta,
|
||||
* so the box average keeps the requested FWHM/beta semantics. */
|
||||
|
||||
/* Lifecycle of the shared supersampled buffer. A fresh accumulator is CLEAN.
|
||||
* Deposits mark it DIRTY; a successful FFTW resolve consumes and restores it to
|
||||
* CLEAN so the next frame cannot inherit stale deposits. */
|
||||
typedef enum {
|
||||
FAST_PSF_BUFFER_CLEAN = 0,
|
||||
FAST_PSF_BUFFER_DIRTY = 1
|
||||
} FastPsfBufferState;
|
||||
|
||||
typedef struct {
|
||||
double fwhm_pixels, moffat_beta, alpha_pixels, alpha_supersampled;
|
||||
double relative_tail_fraction, min_y;
|
||||
@@ -72,7 +83,9 @@ typedef struct {
|
||||
double fftw_setup_seconds;
|
||||
double fftw_kernel_seconds;
|
||||
double fftw_frame_seconds; /* most recent resolve */
|
||||
FastPsfFftwFrameTiming fftw_last_timing; /* breakdown of most recent resolve */
|
||||
#endif
|
||||
FastPsfBufferState buffer_state; /* shared-buffer lifecycle state */
|
||||
int verbose;
|
||||
} FastPsfAccumulator;
|
||||
|
||||
@@ -190,7 +203,32 @@ int psf_prepare_cached_event(PsfCachedEvent *event, double x, double y,
|
||||
void splat_prepared_cached_event(double *hdr, int width, int height,
|
||||
const PsfCachedEvent *event,
|
||||
const PsfKernelCache *cache);
|
||||
/* Writes PNG when built with libpng; non-libpng builds use PPM fallback. */
|
||||
/* Write settings shared by the PNG encoder and the movie output queue. A
|
||||
* negative compression_level selects the libpng/zlib default, preserving the
|
||||
* historical byte output. */
|
||||
typedef struct {
|
||||
int compression_level;
|
||||
} PngWriteSettings;
|
||||
|
||||
/* Parallel HDR -> 8-bit sRGB conversion. Row-outer and static, so the exact
|
||||
* output bytes are independent of worker_count; worker_count <= 1 runs
|
||||
* serially. Every output byte is written by exactly one worker. */
|
||||
int tone_map_srgb8_image(const double *hdr, unsigned char *rgb8, int width,
|
||||
int height, const ToneMapSettings *settings,
|
||||
int worker_count);
|
||||
|
||||
/* Writes an already-tone-mapped sRGB8 buffer. write_png_rgb8 needs libpng and
|
||||
* is declared only when available; write_rgb8_image dispatches on the path
|
||||
* extension and uses the PPM fallback in non-libpng builds. */
|
||||
int write_rgb8_image(const char *path, const unsigned char *rgb8, int width,
|
||||
int height, const PngWriteSettings *settings);
|
||||
#ifdef ENABLE_PNG
|
||||
int write_png_rgb8(const char *path, const unsigned char *rgb8, int width,
|
||||
int height, const PngWriteSettings *settings);
|
||||
#endif
|
||||
|
||||
/* Compatibility wrapper: allocates an RGB8 buffer, applies the tone map, then
|
||||
* writes it. Writes PNG when built with libpng; non-libpng builds use PPM. */
|
||||
int write_tonemapped_image(const char *path, const double *hdr, int width,
|
||||
int height, const ToneMapSettings *settings);
|
||||
#ifdef ENABLE_HDR_OUTPUT
|
||||
|
||||
Reference in new issue
Block a user