Files
GR-raytracing/benchmarks/quadratic_precision/fixtures/probe.c
T
wyj 7c980c35aa Benchmark: Compare quadratic entry precision and arithmetic cost
Generate plain long-double and experimental FMA variants from the production entry kernel. Compare exact-rational references, geometric validation, fallback counts and repeated timings with self-contained fixtures.

Validate cached build dependencies and reject stale timing output. Count all unconfirmed entry outcomes independently of reference classification.
2026-10-09 00:15:02 -04:00

615 lines
21 KiB
C

/* Quadratic precision benchmark probe.
*
* Compiled once per generated variant, with
* -DPROBE_VARIANT_SOURCE=".../ld_fma.c" -DPROBE_VARIANT_NAME="ld_fma"
* The probe #includes the generated full module, so the private static
* entry_quadratic_coeffs / entry_solve and the public asymptotic_route_camera
* are the *actual generated* code. No production file is modified.
*
* Modes:
* accuracy -- run every kernel case through the generated kernel and every
* route case through the generated public pre-route; emit CSVs.
* microbench -- timed repeated kernel assembly + entry_solve (serial).
* routebench -- timed repeated public pre-route (OpenMP, static schedule).
*/
#include PROBE_VARIANT_SOURCE
#include <float.h>
#include <math.h>
#include <omp.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <time.h>
#include "quadratic_cases.h"
#ifndef PROBE_VARIANT_NAME
#define PROBE_VARIANT_NAME "unknown"
#endif
/* ------------------------------------------------------------------ */
/* Exact-value coefficient printing: long double uses %La (full mantissa),
* double uses %a. _Generic selects the right printer without knowing the
* generated EntryQuadratic field type. */
static void coeff_ld(FILE *f, long double v) { fprintf(f, "%La", v); }
static void coeff_d(FILE *f, double v) { fprintf(f, "%a", v); }
#define PRINT_COEFF(f, v) \
_Generic((v), long double : coeff_ld, double : coeff_d)((f), (v))
/* ------------------------------------------------------------------ */
/* Fixture source: flat identity metric, one Minkowski end whose worldtube is
* a constant-velocity (optionally linearly growing/shrinking) sphere.
*
* model 0 (input-stable): center = c0 + v*(t - t0), R = R0 + rr*(t - t0)
* model 1 (production-style): center = v*t, R = R0 + rr*t
*
* model 0 evaluates exactly at the camera: center(t0) == c0, R(t0) == R0, so
* the kernel inputs are the frozen case values. model 1 mirrors the
* Alcubierre-style callback (used only with rr == 0). */
typedef struct {
double c0[3];
double v[3];
double R0;
double rr;
double t0;
double valid_t_min;
int model;
} FixtureContext;
static double fixture_center(const FixtureContext *ctx, int i, double t) {
if (ctx->model == 0)
return ctx->c0[i] + ctx->v[i] * (t - ctx->t0);
return ctx->v[i] * t;
}
static double fixture_radius(const FixtureContext *ctx, double t) {
if (ctx->model == 0)
return ctx->R0 + ctx->rr * (t - ctx->t0);
return ctx->R0 + ctx->rr * t;
}
static SpacetimePointStatus fixture_eval(const SpacetimeSource *source,
double t, const double x[3],
MetricData *metric) {
(void)source;
(void)t;
(void)x;
*metric = (MetricData){.alpha = 1.0,
.gamma = {{1.0, 0.0, 0.0},
{0.0, 1.0, 0.0},
{0.0, 0.0, 1.0}}};
return SPACETIME_POINT_OK;
}
static SpacetimeRayStatus fixture_classify(const SpacetimeSource *source,
double t, const double x[3]) {
(void)source;
(void)t;
(void)x;
return SPACETIME_RAY_ACTIVE;
}
static size_t fixture_end_count(const SpacetimeSource *source) {
(void)source;
return 1;
}
static int fixture_end(const SpacetimeSource *source, size_t index,
SpacetimeAsymptoticEnd *out) {
const FixtureContext *ctx = source->context;
if (index != 0)
return -1;
*out = (SpacetimeAsymptoticEnd){
.end_id = 0,
.exterior_kind = ASYMPTOTIC_EXTERIOR_MINKOWSKI,
.mass = 0.0,
.frame_origin = {0.0, 0.0, 0.0},
.frame_axes = {{1.0, 0.0, 0.0}, {0.0, 1.0, 0.0}, {0.0, 0.0, 1.0}}};
return 0;
}
static int fixture_worldtube(const SpacetimeSource *source,
SpacetimeEndId end_id, double t,
SpacetimeEscapeWorldtubeSample *out) {
const FixtureContext *ctx = source->context;
if (end_id != 0)
return -1;
if (!isfinite(t) || t < ctx->valid_t_min) {
*out = (SpacetimeEscapeWorldtubeSample){.valid = 0};
return 0;
}
*out = (SpacetimeEscapeWorldtubeSample){
.center = {fixture_center(ctx, 0, t), fixture_center(ctx, 1, t),
fixture_center(ctx, 2, t)},
.velocity = {ctx->v[0], ctx->v[1], ctx->v[2]},
.radius = fixture_radius(ctx, t),
.radius_rate = ctx->rr,
.velocity_constant = 1,
.valid = 1};
return 0;
}
static double fixture_next_segment(const SpacetimeSource *source,
SpacetimeEndId end_id, double t) {
(void)source;
(void)end_id;
(void)t;
return NAN;
}
static void fixture_destroy(SpacetimeSource *source) {
source->context = NULL;
source->ops = NULL;
}
static const SpacetimeOps fixture_ops = {
.eval = fixture_eval,
.classify = fixture_classify,
.asymptotic_end_count = fixture_end_count,
.asymptotic_end = fixture_end,
.escape_worldtube_sample = fixture_worldtube,
.escape_worldtube_next_segment = fixture_next_segment,
.destroy = fixture_destroy,
};
/* ------------------------------------------------------------------ */
static void print_environment(void) {
fprintf(stdout,
"{\"variant\":\"%s\",\"sizeof_long_double\":%zu,\"LDBL_MANT_DIG\":%d,"
"\"DBL_MANT_DIG\":%d,\"LDBL_MAX_EXP\":%d,\"hardware_threads\":%d,"
"\"omp_max_threads\":%d}\n",
PROBE_VARIANT_NAME, sizeof(long double), LDBL_MANT_DIG, DBL_MANT_DIG,
LDBL_MAX_EXP, omp_get_num_procs(), omp_get_max_threads());
}
/* ------------------------------------------------------------------ */
/* accuracy mode */
static void mode_accuracy(const char *kernel_out, const char *route_out) {
FILE *kf = fopen(kernel_out, "w");
if (kf == NULL) {
fprintf(stderr, "cannot open %s\n", kernel_out);
exit(2);
}
fprintf(kf,
"id,category,status,sigma,x0,x1,x2,c0,c1,c2,w0,w1,w2,v0,v1,v2,R0,rr,"
"a,b,c\n");
for (int i = 0; i < quad_kernel_case_count; ++i) {
const QuadKernelCase *c = &quad_kernel_cases[i];
EntryQuadratic k = entry_quadratic_coeffs(c->x, c->c, c->w, c->v, c->R0,
c->rr);
double s = -1.0;
EntrySolveResult r = entry_solve(&k, &s);
fprintf(kf, "%d,%s,%d,%a", c->id, c->category, (int)r, s);
for (int j = 0; j < 3; ++j)
fprintf(kf, ",%a", c->x[j]);
for (int j = 0; j < 3; ++j)
fprintf(kf, ",%a", c->c[j]);
for (int j = 0; j < 3; ++j)
fprintf(kf, ",%a", c->w[j]);
for (int j = 0; j < 3; ++j)
fprintf(kf, ",%a", c->v[j]);
fprintf(kf, ",%a,%a,", c->R0, c->rr);
PRINT_COEFF(kf, k.a);
fputc(',', kf);
PRINT_COEFF(kf, k.b);
fputc(',', kf);
PRINT_COEFF(kf, k.c);
fputc('\n', kf);
}
fclose(kf);
FILE *rf = fopen(route_out, "w");
if (rf == NULL) {
fprintf(stderr, "cannot open %s\n", route_out);
exit(2);
}
fprintf(rf,
"id,category,status,kind,failure_reason,fallback_evals,pi_match,"
"lcam_match,F,tol,why,x0,x1,x2,Pi0,Pi1,Pi2,ninf0,ninf1,ninf2,"
"canon_t,canonx0,canonx1,canonx2,canonw0,canonw1,canonw2,"
"cbx0,cbx1,cbx2,cbr,cbok,ct00,ct01,ct02,rt0,cb0ok,"
"kern_ok,kern_status,kern_sigma,kern_a,kern_b,kern_c,"
"logcamera,logentry,"
"t0,oc0,oc1,oc2,d0,d1,d2,fc0,fc1,fc2,v0,v1,v2,"
"R0,rr,model,reason_name\n");
for (int i = 0; i < quad_route_case_count; ++i) {
const QuadRouteCase *c = &quad_route_cases[i];
FixtureContext ctx = {.c0 = {c->c0[0], c->c0[1], c->c0[2]},
.v = {c->v[0], c->v[1], c->v[2]},
.R0 = c->R0,
.rr = c->rr,
.t0 = c->t0,
.valid_t_min = c->valid_t_min,
.model = c->model};
SpacetimeSource source = {.ops = &fixture_ops, .context = &ctx};
ObserverState obs = {0};
obs.coordinate_time = c->t0;
for (int j = 0; j < 3; ++j)
obs.coordinate_position[j] = c->obs[j];
obs.tetrad[0][0] = 1.0;
obs.tetrad[1][1] = 1.0;
obs.tetrad[2][2] = 1.0;
obs.tetrad[3][3] = 1.0;
int canon_ok = 0;
AsymptoticPhotonState canon = {0};
GeodesicRayState st = {0};
{
MetricData metric;
if (spacetime_eval(&source, c->t0, obs.coordinate_position, &metric) ==
SPACETIME_POINT_OK &&
geodesic_initialize_past_ray_metric(&metric, &obs, c->dir, &st) == 0 &&
asymptotic_canonical_from_backend(&source, 0, &metric, c->t0, st.x,
st.Pi, st.log_alpha_p0,
&canon) == 0)
canon_ok = 1;
}
AsymptoticRoute route;
AsymptoticStatus status =
asymptotic_route_camera(&source, &obs, c->dir, &route);
int pi_match = 1, lcam_match = 1;
if (canon_ok && status == ASYMPTOTIC_OK &&
route.kind == ASYMPTOTIC_ROUTE_ENTRY) {
for (int j = 0; j < 3; ++j)
if (route.Pi[j] != st.Pi[j])
pi_match = 0;
if (route.log_alpha_p0_camera != st.log_alpha_p0)
lcam_match = 0;
}
double F = NAN, tol = NAN;
RayReason why = RAY_REASON_NONE;
if (status == ASYMPTOTIC_OK && route.kind == ASYMPTOTIC_ROUTE_ENTRY)
asymptotic_entry_geometry(&source, route.end_id, route.activate_t,
route.x, &F, &tol, &why);
SpacetimeEscapeWorldtubeSample cb = {0};
int cbok = 0;
{
const double tt = (status == ASYMPTOTIC_OK &&
route.kind == ASYMPTOTIC_ROUTE_ENTRY)
? route.activate_t
: c->t0;
if (spacetime_escape_worldtube_sample(&source, route.end_id, tt, &cb) ==
0 &&
cb.valid)
cbok = 1;
}
/* Worldtube sample at the segment start used by the quadratic kernel. */
SpacetimeEscapeWorldtubeSample cb0 = {0};
int cb0ok = 0;
if (spacetime_escape_worldtube_sample(&source, route.end_id, c->t0,
&cb0) == 0 &&
cb0.valid)
cb0ok = 1;
/* Reproduce the first-segment public kernel classification at the actual
* canonical inputs, so a reference ENTER / kernel MISS / route ESCAPED is
* directly visible and not confused with canonical normalisation. */
int kern_ok = 0;
int kern_status = -1;
double kern_sigma = -1.0;
EntryQuadratic kk = {0};
SpacetimeAsymptoticEnd end_desc;
if (canon_ok && cb0ok &&
spacetime_asymptotic_end(&source, 0, &end_desc) == 0) {
double c_frame[3], v_frame[3];
backend_position_to_frame(&end_desc, cb0.center, c_frame);
backend_vector_to_frame(&end_desc, cb0.velocity, v_frame);
kk = entry_quadratic_coeffs(canon.x, c_frame, canon.w, v_frame,
cb0.radius, cb0.radius_rate);
kern_status = (int)entry_solve(&kk, &kern_sigma);
kern_ok = 1;
}
fprintf(rf, "%d,%s,%d,%d,%d,%u,%d,%d,%a,%a,%d", c->id, c->category,
(int)status, (int)route.kind, (int)route.failure_reason,
route.entry_fallback_evaluations, pi_match, lcam_match, F, tol,
(int)why);
for (int j = 0; j < 3; ++j)
fprintf(rf, ",%a", route.x[j]);
for (int j = 0; j < 3; ++j)
fprintf(rf, ",%a", route.Pi[j]);
for (int j = 0; j < 3; ++j)
fprintf(rf, ",%a", route.n_infinity[j]);
fprintf(rf, ",%a", canon.t);
for (int j = 0; j < 3; ++j)
fprintf(rf, ",%a", canon.x[j]);
for (int j = 0; j < 3; ++j)
fprintf(rf, ",%a", canon.w[j]);
for (int j = 0; j < 3; ++j)
fprintf(rf, ",%a", cb.center[j]);
fprintf(rf, ",%a,%d", cb.radius, cbok);
for (int j = 0; j < 3; ++j)
fprintf(rf, ",%a", cb0.center[j]);
fprintf(rf, ",%a,%d", cb0.radius, cb0ok);
fprintf(rf, ",%d,%d,%a,", kern_ok, kern_status, kern_sigma);
PRINT_COEFF(rf, kk.a);
fputc(',', rf);
PRINT_COEFF(rf, kk.b);
fputc(',', rf);
PRINT_COEFF(rf, kk.c);
fprintf(rf, ",%a,%a", route.log_alpha_p0_camera, route.log_alpha_p0);
fprintf(rf, ",%a", c->t0);
for (int j = 0; j < 3; ++j)
fprintf(rf, ",%a", c->obs[j]);
for (int j = 0; j < 3; ++j)
fprintf(rf, ",%a", c->dir[j]);
for (int j = 0; j < 3; ++j)
fprintf(rf, ",%a", c->c0[j]);
for (int j = 0; j < 3; ++j)
fprintf(rf, ",%a", c->v[j]);
fprintf(rf, ",%a,%a,%d,%s\n", c->R0, c->rr, c->model,
ray_reason_name(route.failure_reason));
}
fclose(rf);
}
/* ------------------------------------------------------------------ */
/* microbench mode: coefficient assembly + entry_solve, serial. */
static volatile double g_kernel_sink;
static __attribute__((noinline)) double kernel_batch(const QuadKernelCase *cs,
int n, long reps) {
double acc = 0.0;
for (long r = 0; r < reps; ++r) {
for (int i = 0; i < n; ++i) {
/* Force the index through an opaque register so the compiler cannot
* hoist the pure coefficient solve out of the repetition loop or prove
* the loaded fixture invariant. Identical for every variant. */
int idx = i;
__asm__ __volatile__("" : "+r"(idx) : : "memory");
const QuadKernelCase *c = cs + idx;
EntryQuadratic k = entry_quadratic_coeffs(c->x, c->c, c->w, c->v,
c->R0, c->rr);
double s = 0.0;
acc += (double)entry_solve(&k, &s) + s * 1e-300;
}
}
return acc;
}
static void mode_microbench(long target_calls, const char *out) {
const int n = quad_kernel_case_count;
long reps = target_calls / n;
if (reps < 1)
reps = 1;
/* Warm-up outside the timed region. */
g_kernel_sink += kernel_batch(quad_kernel_cases, n, 1);
const double t0 = omp_get_wtime();
const double acc = kernel_batch(quad_kernel_cases, n, reps);
const double t1 = omp_get_wtime();
g_kernel_sink += acc;
const long calls = (long)n * reps;
const double seconds = t1 - t0;
FILE *f = fopen(out, "w");
if (f == NULL) {
fprintf(stderr, "cannot open %s\n", out);
exit(2);
}
fprintf(f,
"{\"variant\":\"%s\",\"mode\":\"microbench\",\"cases\":%d,"
"\"reps\":%ld,\"calls\":%ld,\"seconds\":%.9f,\"ns_per_call\":%.6f,"
"\"sink\":%.17g}\n",
PROBE_VARIANT_NAME, n, reps, calls, seconds,
seconds * 1e9 / (double)calls, g_kernel_sink);
fclose(f);
fprintf(stdout, "microbench %s: %ld calls in %.6f s (%.2f ns/call)\n",
PROBE_VARIANT_NAME, calls, seconds, seconds * 1e9 / (double)calls);
}
/* ------------------------------------------------------------------ */
/* routebench mode: public asymptotic_route_camera, OpenMP static. */
typedef struct {
FixtureContext *ctxs;
SpacetimeSource *srcs;
ObserverState *obss;
const QuadRouteCase *rcs;
int n;
} Preloaded;
typedef struct {
long entry, escaped, inside, time_exhausted, invalid, unsupported, other;
long fallback_sum;
} RouteCounts;
static volatile double g_route_sink;
static __attribute__((noinline)) void route_batch(const Preloaded *p, long reps,
int threads, double *seconds,
RouteCounts *counts) {
const int n = p->n;
long entry = 0, escaped = 0, inside = 0, texh = 0, inv = 0, unsup = 0,
other = 0, fallback = 0;
const long total = (long)n * reps;
const double t0 = omp_get_wtime();
#pragma omp parallel num_threads(threads) reduction(+ : entry, escaped, inside, texh, inv, unsup, other, fallback)
{
#pragma omp for schedule(static)
for (long k = 0; k < total; ++k) {
long kk = k;
__asm__ __volatile__("" : "+r"(kk) : : "memory");
const int i = (int)(kk % n);
AsymptoticRoute route;
const AsymptoticStatus st = asymptotic_route_camera(
&p->srcs[i], &p->obss[i], p->rcs[i].dir, &route);
fallback += (long)route.entry_fallback_evaluations;
if (st == ASYMPTOTIC_OK) {
switch (route.kind) {
case ASYMPTOTIC_ROUTE_ENTRY:
++entry;
break;
case ASYMPTOTIC_ROUTE_ESCAPED:
++escaped;
break;
case ASYMPTOTIC_ROUTE_INSIDE:
++inside;
break;
case ASYMPTOTIC_ROUTE_TIME_RANGE_EXHAUSTED:
++texh;
break;
default:
++other;
break;
}
} else if (st == ASYMPTOTIC_UNSUPPORTED) {
++unsup;
} else {
++inv;
}
}
}
*seconds = omp_get_wtime() - t0;
counts->entry = entry;
counts->escaped = escaped;
counts->inside = inside;
counts->time_exhausted = texh;
counts->invalid = inv;
counts->unsupported = unsup;
counts->other = other;
counts->fallback_sum = fallback;
}
static void mode_routebench(long target_calls, int threads, double max_seconds,
const char *out) {
const int n = quad_route_case_count;
Preloaded p;
p.n = n;
p.rcs = quad_route_cases;
p.ctxs = malloc(sizeof(FixtureContext) * (size_t)n);
p.srcs = malloc(sizeof(SpacetimeSource) * (size_t)n);
p.obss = malloc(sizeof(ObserverState) * (size_t)n);
if (!p.ctxs || !p.srcs || !p.obss) {
fprintf(stderr, "allocation failure\n");
exit(2);
}
for (int i = 0; i < n; ++i) {
const QuadRouteCase *c = &quad_route_cases[i];
p.ctxs[i] = (FixtureContext){.c0 = {c->c0[0], c->c0[1], c->c0[2]},
.v = {c->v[0], c->v[1], c->v[2]},
.R0 = c->R0,
.rr = c->rr,
.t0 = c->t0,
.valid_t_min = c->valid_t_min,
.model = c->model};
p.srcs[i] = (SpacetimeSource){.ops = &fixture_ops, .context = &p.ctxs[i]};
p.obss[i] = (ObserverState){0};
p.obss[i].coordinate_time = c->t0;
for (int j = 0; j < 3; ++j)
p.obss[i].coordinate_position[j] = c->obs[j];
p.obss[i].tetrad[0][0] = 1.0;
p.obss[i].tetrad[1][1] = 1.0;
p.obss[i].tetrad[2][2] = 1.0;
p.obss[i].tetrad[3][3] = 1.0;
}
/* Calibrate with one pass, then size reps to the call/time budgets. */
double cal_seconds = 0.0;
RouteCounts cal_counts;
route_batch(&p, 1, threads, &cal_seconds, &cal_counts);
const double per_call = cal_seconds / (double)n;
long reps = target_calls / n;
if (reps < 1)
reps = 1;
if (per_call > 0.0) {
const long by_time = (long)(max_seconds / (per_call * (double)n));
if (by_time < 1)
reps = 1;
else if (reps > by_time)
reps = by_time;
}
double seconds = 0.0;
RouteCounts counts;
route_batch(&p, reps, threads, &seconds, &counts);
g_route_sink += (double)counts.entry + (double)counts.escaped;
FILE *f = fopen(out, "w");
if (f == NULL) {
fprintf(stderr, "cannot open %s\n", out);
exit(2);
}
fprintf(f,
"{\"variant\":\"%s\",\"mode\":\"routebench\",\"cases\":%d,"
"\"reps\":%ld,\"calls\":%ld,\"threads\":%d,\"cal_seconds\":%.9f,"
"\"seconds\":%.9f,\"ns_per_call\":%.6f,\"entry\":%ld,\"escaped\":%ld,"
"\"inside\":%ld,\"time_exhausted\":%ld,\"invalid\":%ld,"
"\"unsupported\":%ld,\"other\":%ld,\"fallback_sum\":%ld}\n",
PROBE_VARIANT_NAME, n, reps, (long)n * reps, threads, cal_seconds,
seconds, seconds * 1e9 / (double)((long)n * reps), counts.entry,
counts.escaped, counts.inside, counts.time_exhausted, counts.invalid,
counts.unsupported, counts.other, counts.fallback_sum);
fclose(f);
fprintf(stdout,
"routebench %s: %ld calls in %.6f s on %d threads (%.2f ns/call)\n",
PROBE_VARIANT_NAME, (long)n * reps, seconds, threads,
seconds * 1e9 / (double)((long)n * reps));
free(p.ctxs);
free(p.srcs);
free(p.obss);
}
/* ------------------------------------------------------------------ */
static void usage(const char *argv0) {
fprintf(stderr,
"usage:\n"
" %s accuracy --kernel-out K.csv --route-out R.csv\n"
" %s microbench --out F.json [--target-calls N]\n"
" %s routebench --out F.json [--target-calls N] [--threads T]"
" [--max-seconds S]\n",
argv0, argv0, argv0);
}
static const char *arg_value(int argc, char **argv, const char *flag) {
for (int i = 1; i + 1 < argc; ++i)
if (strcmp(argv[i], flag) == 0)
return argv[i + 1];
return NULL;
}
int main(int argc, char **argv) {
print_environment();
if (argc < 2) {
usage(argv[0]);
return 1;
}
if (strcmp(argv[1], "accuracy") == 0) {
const char *k = arg_value(argc, argv, "--kernel-out");
const char *r = arg_value(argc, argv, "--route-out");
if (!k || !r) {
usage(argv[0]);
return 1;
}
mode_accuracy(k, r);
return 0;
}
if (strcmp(argv[1], "microbench") == 0) {
const char *out = arg_value(argc, argv, "--out");
const char *tc = arg_value(argc, argv, "--target-calls");
if (!out) {
usage(argv[0]);
return 1;
}
mode_microbench(tc ? atol(tc) : 10000000L, out);
return 0;
}
if (strcmp(argv[1], "routebench") == 0) {
const char *out = arg_value(argc, argv, "--out");
const char *tc = arg_value(argc, argv, "--target-calls");
const char *th = arg_value(argc, argv, "--threads");
const char *ms = arg_value(argc, argv, "--max-seconds");
if (!out) {
usage(argv[0]);
return 1;
}
mode_routebench(tc ? atol(tc) : 1000000L, th ? atoi(th) : 4,
ms ? atof(ms) : 20.0, out);
return 0;
}
usage(argv[0]);
return 1;
}