Generate plain long-double and experimental FMA variants from the production entry kernel. Compare exact-rational references, geometric validation, fallback counts and repeated timings with self-contained fixtures. Validate cached build dependencies and reject stale timing output. Count all unconfirmed entry outcomes independently of reference classification.
615 lines
21 KiB
C
615 lines
21 KiB
C
/* Quadratic precision benchmark probe.
|
|
*
|
|
* Compiled once per generated variant, with
|
|
* -DPROBE_VARIANT_SOURCE=".../ld_fma.c" -DPROBE_VARIANT_NAME="ld_fma"
|
|
* The probe #includes the generated full module, so the private static
|
|
* entry_quadratic_coeffs / entry_solve and the public asymptotic_route_camera
|
|
* are the *actual generated* code. No production file is modified.
|
|
*
|
|
* Modes:
|
|
* accuracy -- run every kernel case through the generated kernel and every
|
|
* route case through the generated public pre-route; emit CSVs.
|
|
* microbench -- timed repeated kernel assembly + entry_solve (serial).
|
|
* routebench -- timed repeated public pre-route (OpenMP, static schedule).
|
|
*/
|
|
#include PROBE_VARIANT_SOURCE
|
|
|
|
#include <float.h>
|
|
#include <math.h>
|
|
#include <omp.h>
|
|
#include <stdio.h>
|
|
#include <stdlib.h>
|
|
#include <string.h>
|
|
#include <time.h>
|
|
|
|
#include "quadratic_cases.h"
|
|
|
|
#ifndef PROBE_VARIANT_NAME
|
|
#define PROBE_VARIANT_NAME "unknown"
|
|
#endif
|
|
|
|
/* ------------------------------------------------------------------ */
|
|
/* Exact-value coefficient printing: long double uses %La (full mantissa),
|
|
* double uses %a. _Generic selects the right printer without knowing the
|
|
* generated EntryQuadratic field type. */
|
|
static void coeff_ld(FILE *f, long double v) { fprintf(f, "%La", v); }
|
|
static void coeff_d(FILE *f, double v) { fprintf(f, "%a", v); }
|
|
#define PRINT_COEFF(f, v) \
|
|
_Generic((v), long double : coeff_ld, double : coeff_d)((f), (v))
|
|
|
|
/* ------------------------------------------------------------------ */
|
|
/* Fixture source: flat identity metric, one Minkowski end whose worldtube is
|
|
* a constant-velocity (optionally linearly growing/shrinking) sphere.
|
|
*
|
|
* model 0 (input-stable): center = c0 + v*(t - t0), R = R0 + rr*(t - t0)
|
|
* model 1 (production-style): center = v*t, R = R0 + rr*t
|
|
*
|
|
* model 0 evaluates exactly at the camera: center(t0) == c0, R(t0) == R0, so
|
|
* the kernel inputs are the frozen case values. model 1 mirrors the
|
|
* Alcubierre-style callback (used only with rr == 0). */
|
|
typedef struct {
|
|
double c0[3];
|
|
double v[3];
|
|
double R0;
|
|
double rr;
|
|
double t0;
|
|
double valid_t_min;
|
|
int model;
|
|
} FixtureContext;
|
|
|
|
static double fixture_center(const FixtureContext *ctx, int i, double t) {
|
|
if (ctx->model == 0)
|
|
return ctx->c0[i] + ctx->v[i] * (t - ctx->t0);
|
|
return ctx->v[i] * t;
|
|
}
|
|
|
|
static double fixture_radius(const FixtureContext *ctx, double t) {
|
|
if (ctx->model == 0)
|
|
return ctx->R0 + ctx->rr * (t - ctx->t0);
|
|
return ctx->R0 + ctx->rr * t;
|
|
}
|
|
|
|
static SpacetimePointStatus fixture_eval(const SpacetimeSource *source,
|
|
double t, const double x[3],
|
|
MetricData *metric) {
|
|
(void)source;
|
|
(void)t;
|
|
(void)x;
|
|
*metric = (MetricData){.alpha = 1.0,
|
|
.gamma = {{1.0, 0.0, 0.0},
|
|
{0.0, 1.0, 0.0},
|
|
{0.0, 0.0, 1.0}}};
|
|
return SPACETIME_POINT_OK;
|
|
}
|
|
|
|
static SpacetimeRayStatus fixture_classify(const SpacetimeSource *source,
|
|
double t, const double x[3]) {
|
|
(void)source;
|
|
(void)t;
|
|
(void)x;
|
|
return SPACETIME_RAY_ACTIVE;
|
|
}
|
|
|
|
static size_t fixture_end_count(const SpacetimeSource *source) {
|
|
(void)source;
|
|
return 1;
|
|
}
|
|
|
|
static int fixture_end(const SpacetimeSource *source, size_t index,
|
|
SpacetimeAsymptoticEnd *out) {
|
|
const FixtureContext *ctx = source->context;
|
|
if (index != 0)
|
|
return -1;
|
|
*out = (SpacetimeAsymptoticEnd){
|
|
.end_id = 0,
|
|
.exterior_kind = ASYMPTOTIC_EXTERIOR_MINKOWSKI,
|
|
.mass = 0.0,
|
|
.frame_origin = {0.0, 0.0, 0.0},
|
|
.frame_axes = {{1.0, 0.0, 0.0}, {0.0, 1.0, 0.0}, {0.0, 0.0, 1.0}}};
|
|
return 0;
|
|
}
|
|
|
|
static int fixture_worldtube(const SpacetimeSource *source,
|
|
SpacetimeEndId end_id, double t,
|
|
SpacetimeEscapeWorldtubeSample *out) {
|
|
const FixtureContext *ctx = source->context;
|
|
if (end_id != 0)
|
|
return -1;
|
|
if (!isfinite(t) || t < ctx->valid_t_min) {
|
|
*out = (SpacetimeEscapeWorldtubeSample){.valid = 0};
|
|
return 0;
|
|
}
|
|
*out = (SpacetimeEscapeWorldtubeSample){
|
|
.center = {fixture_center(ctx, 0, t), fixture_center(ctx, 1, t),
|
|
fixture_center(ctx, 2, t)},
|
|
.velocity = {ctx->v[0], ctx->v[1], ctx->v[2]},
|
|
.radius = fixture_radius(ctx, t),
|
|
.radius_rate = ctx->rr,
|
|
.velocity_constant = 1,
|
|
.valid = 1};
|
|
return 0;
|
|
}
|
|
|
|
static double fixture_next_segment(const SpacetimeSource *source,
|
|
SpacetimeEndId end_id, double t) {
|
|
(void)source;
|
|
(void)end_id;
|
|
(void)t;
|
|
return NAN;
|
|
}
|
|
|
|
static void fixture_destroy(SpacetimeSource *source) {
|
|
source->context = NULL;
|
|
source->ops = NULL;
|
|
}
|
|
|
|
static const SpacetimeOps fixture_ops = {
|
|
.eval = fixture_eval,
|
|
.classify = fixture_classify,
|
|
.asymptotic_end_count = fixture_end_count,
|
|
.asymptotic_end = fixture_end,
|
|
.escape_worldtube_sample = fixture_worldtube,
|
|
.escape_worldtube_next_segment = fixture_next_segment,
|
|
.destroy = fixture_destroy,
|
|
};
|
|
|
|
/* ------------------------------------------------------------------ */
|
|
static void print_environment(void) {
|
|
fprintf(stdout,
|
|
"{\"variant\":\"%s\",\"sizeof_long_double\":%zu,\"LDBL_MANT_DIG\":%d,"
|
|
"\"DBL_MANT_DIG\":%d,\"LDBL_MAX_EXP\":%d,\"hardware_threads\":%d,"
|
|
"\"omp_max_threads\":%d}\n",
|
|
PROBE_VARIANT_NAME, sizeof(long double), LDBL_MANT_DIG, DBL_MANT_DIG,
|
|
LDBL_MAX_EXP, omp_get_num_procs(), omp_get_max_threads());
|
|
}
|
|
|
|
/* ------------------------------------------------------------------ */
|
|
/* accuracy mode */
|
|
static void mode_accuracy(const char *kernel_out, const char *route_out) {
|
|
FILE *kf = fopen(kernel_out, "w");
|
|
if (kf == NULL) {
|
|
fprintf(stderr, "cannot open %s\n", kernel_out);
|
|
exit(2);
|
|
}
|
|
fprintf(kf,
|
|
"id,category,status,sigma,x0,x1,x2,c0,c1,c2,w0,w1,w2,v0,v1,v2,R0,rr,"
|
|
"a,b,c\n");
|
|
for (int i = 0; i < quad_kernel_case_count; ++i) {
|
|
const QuadKernelCase *c = &quad_kernel_cases[i];
|
|
EntryQuadratic k = entry_quadratic_coeffs(c->x, c->c, c->w, c->v, c->R0,
|
|
c->rr);
|
|
double s = -1.0;
|
|
EntrySolveResult r = entry_solve(&k, &s);
|
|
fprintf(kf, "%d,%s,%d,%a", c->id, c->category, (int)r, s);
|
|
for (int j = 0; j < 3; ++j)
|
|
fprintf(kf, ",%a", c->x[j]);
|
|
for (int j = 0; j < 3; ++j)
|
|
fprintf(kf, ",%a", c->c[j]);
|
|
for (int j = 0; j < 3; ++j)
|
|
fprintf(kf, ",%a", c->w[j]);
|
|
for (int j = 0; j < 3; ++j)
|
|
fprintf(kf, ",%a", c->v[j]);
|
|
fprintf(kf, ",%a,%a,", c->R0, c->rr);
|
|
PRINT_COEFF(kf, k.a);
|
|
fputc(',', kf);
|
|
PRINT_COEFF(kf, k.b);
|
|
fputc(',', kf);
|
|
PRINT_COEFF(kf, k.c);
|
|
fputc('\n', kf);
|
|
}
|
|
fclose(kf);
|
|
|
|
FILE *rf = fopen(route_out, "w");
|
|
if (rf == NULL) {
|
|
fprintf(stderr, "cannot open %s\n", route_out);
|
|
exit(2);
|
|
}
|
|
fprintf(rf,
|
|
"id,category,status,kind,failure_reason,fallback_evals,pi_match,"
|
|
"lcam_match,F,tol,why,x0,x1,x2,Pi0,Pi1,Pi2,ninf0,ninf1,ninf2,"
|
|
"canon_t,canonx0,canonx1,canonx2,canonw0,canonw1,canonw2,"
|
|
"cbx0,cbx1,cbx2,cbr,cbok,ct00,ct01,ct02,rt0,cb0ok,"
|
|
"kern_ok,kern_status,kern_sigma,kern_a,kern_b,kern_c,"
|
|
"logcamera,logentry,"
|
|
"t0,oc0,oc1,oc2,d0,d1,d2,fc0,fc1,fc2,v0,v1,v2,"
|
|
"R0,rr,model,reason_name\n");
|
|
for (int i = 0; i < quad_route_case_count; ++i) {
|
|
const QuadRouteCase *c = &quad_route_cases[i];
|
|
FixtureContext ctx = {.c0 = {c->c0[0], c->c0[1], c->c0[2]},
|
|
.v = {c->v[0], c->v[1], c->v[2]},
|
|
.R0 = c->R0,
|
|
.rr = c->rr,
|
|
.t0 = c->t0,
|
|
.valid_t_min = c->valid_t_min,
|
|
.model = c->model};
|
|
SpacetimeSource source = {.ops = &fixture_ops, .context = &ctx};
|
|
ObserverState obs = {0};
|
|
obs.coordinate_time = c->t0;
|
|
for (int j = 0; j < 3; ++j)
|
|
obs.coordinate_position[j] = c->obs[j];
|
|
obs.tetrad[0][0] = 1.0;
|
|
obs.tetrad[1][1] = 1.0;
|
|
obs.tetrad[2][2] = 1.0;
|
|
obs.tetrad[3][3] = 1.0;
|
|
|
|
int canon_ok = 0;
|
|
AsymptoticPhotonState canon = {0};
|
|
GeodesicRayState st = {0};
|
|
{
|
|
MetricData metric;
|
|
if (spacetime_eval(&source, c->t0, obs.coordinate_position, &metric) ==
|
|
SPACETIME_POINT_OK &&
|
|
geodesic_initialize_past_ray_metric(&metric, &obs, c->dir, &st) == 0 &&
|
|
asymptotic_canonical_from_backend(&source, 0, &metric, c->t0, st.x,
|
|
st.Pi, st.log_alpha_p0,
|
|
&canon) == 0)
|
|
canon_ok = 1;
|
|
}
|
|
|
|
AsymptoticRoute route;
|
|
AsymptoticStatus status =
|
|
asymptotic_route_camera(&source, &obs, c->dir, &route);
|
|
|
|
int pi_match = 1, lcam_match = 1;
|
|
if (canon_ok && status == ASYMPTOTIC_OK &&
|
|
route.kind == ASYMPTOTIC_ROUTE_ENTRY) {
|
|
for (int j = 0; j < 3; ++j)
|
|
if (route.Pi[j] != st.Pi[j])
|
|
pi_match = 0;
|
|
if (route.log_alpha_p0_camera != st.log_alpha_p0)
|
|
lcam_match = 0;
|
|
}
|
|
|
|
double F = NAN, tol = NAN;
|
|
RayReason why = RAY_REASON_NONE;
|
|
if (status == ASYMPTOTIC_OK && route.kind == ASYMPTOTIC_ROUTE_ENTRY)
|
|
asymptotic_entry_geometry(&source, route.end_id, route.activate_t,
|
|
route.x, &F, &tol, &why);
|
|
|
|
SpacetimeEscapeWorldtubeSample cb = {0};
|
|
int cbok = 0;
|
|
{
|
|
const double tt = (status == ASYMPTOTIC_OK &&
|
|
route.kind == ASYMPTOTIC_ROUTE_ENTRY)
|
|
? route.activate_t
|
|
: c->t0;
|
|
if (spacetime_escape_worldtube_sample(&source, route.end_id, tt, &cb) ==
|
|
0 &&
|
|
cb.valid)
|
|
cbok = 1;
|
|
}
|
|
/* Worldtube sample at the segment start used by the quadratic kernel. */
|
|
SpacetimeEscapeWorldtubeSample cb0 = {0};
|
|
int cb0ok = 0;
|
|
if (spacetime_escape_worldtube_sample(&source, route.end_id, c->t0,
|
|
&cb0) == 0 &&
|
|
cb0.valid)
|
|
cb0ok = 1;
|
|
|
|
/* Reproduce the first-segment public kernel classification at the actual
|
|
* canonical inputs, so a reference ENTER / kernel MISS / route ESCAPED is
|
|
* directly visible and not confused with canonical normalisation. */
|
|
int kern_ok = 0;
|
|
int kern_status = -1;
|
|
double kern_sigma = -1.0;
|
|
EntryQuadratic kk = {0};
|
|
SpacetimeAsymptoticEnd end_desc;
|
|
if (canon_ok && cb0ok &&
|
|
spacetime_asymptotic_end(&source, 0, &end_desc) == 0) {
|
|
double c_frame[3], v_frame[3];
|
|
backend_position_to_frame(&end_desc, cb0.center, c_frame);
|
|
backend_vector_to_frame(&end_desc, cb0.velocity, v_frame);
|
|
kk = entry_quadratic_coeffs(canon.x, c_frame, canon.w, v_frame,
|
|
cb0.radius, cb0.radius_rate);
|
|
kern_status = (int)entry_solve(&kk, &kern_sigma);
|
|
kern_ok = 1;
|
|
}
|
|
|
|
fprintf(rf, "%d,%s,%d,%d,%d,%u,%d,%d,%a,%a,%d", c->id, c->category,
|
|
(int)status, (int)route.kind, (int)route.failure_reason,
|
|
route.entry_fallback_evaluations, pi_match, lcam_match, F, tol,
|
|
(int)why);
|
|
for (int j = 0; j < 3; ++j)
|
|
fprintf(rf, ",%a", route.x[j]);
|
|
for (int j = 0; j < 3; ++j)
|
|
fprintf(rf, ",%a", route.Pi[j]);
|
|
for (int j = 0; j < 3; ++j)
|
|
fprintf(rf, ",%a", route.n_infinity[j]);
|
|
fprintf(rf, ",%a", canon.t);
|
|
for (int j = 0; j < 3; ++j)
|
|
fprintf(rf, ",%a", canon.x[j]);
|
|
for (int j = 0; j < 3; ++j)
|
|
fprintf(rf, ",%a", canon.w[j]);
|
|
for (int j = 0; j < 3; ++j)
|
|
fprintf(rf, ",%a", cb.center[j]);
|
|
fprintf(rf, ",%a,%d", cb.radius, cbok);
|
|
for (int j = 0; j < 3; ++j)
|
|
fprintf(rf, ",%a", cb0.center[j]);
|
|
fprintf(rf, ",%a,%d", cb0.radius, cb0ok);
|
|
fprintf(rf, ",%d,%d,%a,", kern_ok, kern_status, kern_sigma);
|
|
PRINT_COEFF(rf, kk.a);
|
|
fputc(',', rf);
|
|
PRINT_COEFF(rf, kk.b);
|
|
fputc(',', rf);
|
|
PRINT_COEFF(rf, kk.c);
|
|
fprintf(rf, ",%a,%a", route.log_alpha_p0_camera, route.log_alpha_p0);
|
|
fprintf(rf, ",%a", c->t0);
|
|
for (int j = 0; j < 3; ++j)
|
|
fprintf(rf, ",%a", c->obs[j]);
|
|
for (int j = 0; j < 3; ++j)
|
|
fprintf(rf, ",%a", c->dir[j]);
|
|
for (int j = 0; j < 3; ++j)
|
|
fprintf(rf, ",%a", c->c0[j]);
|
|
for (int j = 0; j < 3; ++j)
|
|
fprintf(rf, ",%a", c->v[j]);
|
|
fprintf(rf, ",%a,%a,%d,%s\n", c->R0, c->rr, c->model,
|
|
ray_reason_name(route.failure_reason));
|
|
}
|
|
fclose(rf);
|
|
}
|
|
|
|
/* ------------------------------------------------------------------ */
|
|
/* microbench mode: coefficient assembly + entry_solve, serial. */
|
|
static volatile double g_kernel_sink;
|
|
|
|
static __attribute__((noinline)) double kernel_batch(const QuadKernelCase *cs,
|
|
int n, long reps) {
|
|
double acc = 0.0;
|
|
for (long r = 0; r < reps; ++r) {
|
|
for (int i = 0; i < n; ++i) {
|
|
/* Force the index through an opaque register so the compiler cannot
|
|
* hoist the pure coefficient solve out of the repetition loop or prove
|
|
* the loaded fixture invariant. Identical for every variant. */
|
|
int idx = i;
|
|
__asm__ __volatile__("" : "+r"(idx) : : "memory");
|
|
const QuadKernelCase *c = cs + idx;
|
|
EntryQuadratic k = entry_quadratic_coeffs(c->x, c->c, c->w, c->v,
|
|
c->R0, c->rr);
|
|
double s = 0.0;
|
|
acc += (double)entry_solve(&k, &s) + s * 1e-300;
|
|
}
|
|
}
|
|
return acc;
|
|
}
|
|
|
|
static void mode_microbench(long target_calls, const char *out) {
|
|
const int n = quad_kernel_case_count;
|
|
long reps = target_calls / n;
|
|
if (reps < 1)
|
|
reps = 1;
|
|
/* Warm-up outside the timed region. */
|
|
g_kernel_sink += kernel_batch(quad_kernel_cases, n, 1);
|
|
const double t0 = omp_get_wtime();
|
|
const double acc = kernel_batch(quad_kernel_cases, n, reps);
|
|
const double t1 = omp_get_wtime();
|
|
g_kernel_sink += acc;
|
|
const long calls = (long)n * reps;
|
|
const double seconds = t1 - t0;
|
|
FILE *f = fopen(out, "w");
|
|
if (f == NULL) {
|
|
fprintf(stderr, "cannot open %s\n", out);
|
|
exit(2);
|
|
}
|
|
fprintf(f,
|
|
"{\"variant\":\"%s\",\"mode\":\"microbench\",\"cases\":%d,"
|
|
"\"reps\":%ld,\"calls\":%ld,\"seconds\":%.9f,\"ns_per_call\":%.6f,"
|
|
"\"sink\":%.17g}\n",
|
|
PROBE_VARIANT_NAME, n, reps, calls, seconds,
|
|
seconds * 1e9 / (double)calls, g_kernel_sink);
|
|
fclose(f);
|
|
fprintf(stdout, "microbench %s: %ld calls in %.6f s (%.2f ns/call)\n",
|
|
PROBE_VARIANT_NAME, calls, seconds, seconds * 1e9 / (double)calls);
|
|
}
|
|
|
|
/* ------------------------------------------------------------------ */
|
|
/* routebench mode: public asymptotic_route_camera, OpenMP static. */
|
|
typedef struct {
|
|
FixtureContext *ctxs;
|
|
SpacetimeSource *srcs;
|
|
ObserverState *obss;
|
|
const QuadRouteCase *rcs;
|
|
int n;
|
|
} Preloaded;
|
|
|
|
typedef struct {
|
|
long entry, escaped, inside, time_exhausted, invalid, unsupported, other;
|
|
long fallback_sum;
|
|
} RouteCounts;
|
|
|
|
static volatile double g_route_sink;
|
|
|
|
static __attribute__((noinline)) void route_batch(const Preloaded *p, long reps,
|
|
int threads, double *seconds,
|
|
RouteCounts *counts) {
|
|
const int n = p->n;
|
|
long entry = 0, escaped = 0, inside = 0, texh = 0, inv = 0, unsup = 0,
|
|
other = 0, fallback = 0;
|
|
const long total = (long)n * reps;
|
|
const double t0 = omp_get_wtime();
|
|
#pragma omp parallel num_threads(threads) reduction(+ : entry, escaped, inside, texh, inv, unsup, other, fallback)
|
|
{
|
|
#pragma omp for schedule(static)
|
|
for (long k = 0; k < total; ++k) {
|
|
long kk = k;
|
|
__asm__ __volatile__("" : "+r"(kk) : : "memory");
|
|
const int i = (int)(kk % n);
|
|
AsymptoticRoute route;
|
|
const AsymptoticStatus st = asymptotic_route_camera(
|
|
&p->srcs[i], &p->obss[i], p->rcs[i].dir, &route);
|
|
fallback += (long)route.entry_fallback_evaluations;
|
|
if (st == ASYMPTOTIC_OK) {
|
|
switch (route.kind) {
|
|
case ASYMPTOTIC_ROUTE_ENTRY:
|
|
++entry;
|
|
break;
|
|
case ASYMPTOTIC_ROUTE_ESCAPED:
|
|
++escaped;
|
|
break;
|
|
case ASYMPTOTIC_ROUTE_INSIDE:
|
|
++inside;
|
|
break;
|
|
case ASYMPTOTIC_ROUTE_TIME_RANGE_EXHAUSTED:
|
|
++texh;
|
|
break;
|
|
default:
|
|
++other;
|
|
break;
|
|
}
|
|
} else if (st == ASYMPTOTIC_UNSUPPORTED) {
|
|
++unsup;
|
|
} else {
|
|
++inv;
|
|
}
|
|
}
|
|
}
|
|
*seconds = omp_get_wtime() - t0;
|
|
counts->entry = entry;
|
|
counts->escaped = escaped;
|
|
counts->inside = inside;
|
|
counts->time_exhausted = texh;
|
|
counts->invalid = inv;
|
|
counts->unsupported = unsup;
|
|
counts->other = other;
|
|
counts->fallback_sum = fallback;
|
|
}
|
|
|
|
static void mode_routebench(long target_calls, int threads, double max_seconds,
|
|
const char *out) {
|
|
const int n = quad_route_case_count;
|
|
Preloaded p;
|
|
p.n = n;
|
|
p.rcs = quad_route_cases;
|
|
p.ctxs = malloc(sizeof(FixtureContext) * (size_t)n);
|
|
p.srcs = malloc(sizeof(SpacetimeSource) * (size_t)n);
|
|
p.obss = malloc(sizeof(ObserverState) * (size_t)n);
|
|
if (!p.ctxs || !p.srcs || !p.obss) {
|
|
fprintf(stderr, "allocation failure\n");
|
|
exit(2);
|
|
}
|
|
for (int i = 0; i < n; ++i) {
|
|
const QuadRouteCase *c = &quad_route_cases[i];
|
|
p.ctxs[i] = (FixtureContext){.c0 = {c->c0[0], c->c0[1], c->c0[2]},
|
|
.v = {c->v[0], c->v[1], c->v[2]},
|
|
.R0 = c->R0,
|
|
.rr = c->rr,
|
|
.t0 = c->t0,
|
|
.valid_t_min = c->valid_t_min,
|
|
.model = c->model};
|
|
p.srcs[i] = (SpacetimeSource){.ops = &fixture_ops, .context = &p.ctxs[i]};
|
|
p.obss[i] = (ObserverState){0};
|
|
p.obss[i].coordinate_time = c->t0;
|
|
for (int j = 0; j < 3; ++j)
|
|
p.obss[i].coordinate_position[j] = c->obs[j];
|
|
p.obss[i].tetrad[0][0] = 1.0;
|
|
p.obss[i].tetrad[1][1] = 1.0;
|
|
p.obss[i].tetrad[2][2] = 1.0;
|
|
p.obss[i].tetrad[3][3] = 1.0;
|
|
}
|
|
|
|
/* Calibrate with one pass, then size reps to the call/time budgets. */
|
|
double cal_seconds = 0.0;
|
|
RouteCounts cal_counts;
|
|
route_batch(&p, 1, threads, &cal_seconds, &cal_counts);
|
|
const double per_call = cal_seconds / (double)n;
|
|
long reps = target_calls / n;
|
|
if (reps < 1)
|
|
reps = 1;
|
|
if (per_call > 0.0) {
|
|
const long by_time = (long)(max_seconds / (per_call * (double)n));
|
|
if (by_time < 1)
|
|
reps = 1;
|
|
else if (reps > by_time)
|
|
reps = by_time;
|
|
}
|
|
|
|
double seconds = 0.0;
|
|
RouteCounts counts;
|
|
route_batch(&p, reps, threads, &seconds, &counts);
|
|
g_route_sink += (double)counts.entry + (double)counts.escaped;
|
|
|
|
FILE *f = fopen(out, "w");
|
|
if (f == NULL) {
|
|
fprintf(stderr, "cannot open %s\n", out);
|
|
exit(2);
|
|
}
|
|
fprintf(f,
|
|
"{\"variant\":\"%s\",\"mode\":\"routebench\",\"cases\":%d,"
|
|
"\"reps\":%ld,\"calls\":%ld,\"threads\":%d,\"cal_seconds\":%.9f,"
|
|
"\"seconds\":%.9f,\"ns_per_call\":%.6f,\"entry\":%ld,\"escaped\":%ld,"
|
|
"\"inside\":%ld,\"time_exhausted\":%ld,\"invalid\":%ld,"
|
|
"\"unsupported\":%ld,\"other\":%ld,\"fallback_sum\":%ld}\n",
|
|
PROBE_VARIANT_NAME, n, reps, (long)n * reps, threads, cal_seconds,
|
|
seconds, seconds * 1e9 / (double)((long)n * reps), counts.entry,
|
|
counts.escaped, counts.inside, counts.time_exhausted, counts.invalid,
|
|
counts.unsupported, counts.other, counts.fallback_sum);
|
|
fclose(f);
|
|
fprintf(stdout,
|
|
"routebench %s: %ld calls in %.6f s on %d threads (%.2f ns/call)\n",
|
|
PROBE_VARIANT_NAME, (long)n * reps, seconds, threads,
|
|
seconds * 1e9 / (double)((long)n * reps));
|
|
free(p.ctxs);
|
|
free(p.srcs);
|
|
free(p.obss);
|
|
}
|
|
|
|
/* ------------------------------------------------------------------ */
|
|
static void usage(const char *argv0) {
|
|
fprintf(stderr,
|
|
"usage:\n"
|
|
" %s accuracy --kernel-out K.csv --route-out R.csv\n"
|
|
" %s microbench --out F.json [--target-calls N]\n"
|
|
" %s routebench --out F.json [--target-calls N] [--threads T]"
|
|
" [--max-seconds S]\n",
|
|
argv0, argv0, argv0);
|
|
}
|
|
|
|
static const char *arg_value(int argc, char **argv, const char *flag) {
|
|
for (int i = 1; i + 1 < argc; ++i)
|
|
if (strcmp(argv[i], flag) == 0)
|
|
return argv[i + 1];
|
|
return NULL;
|
|
}
|
|
|
|
int main(int argc, char **argv) {
|
|
print_environment();
|
|
if (argc < 2) {
|
|
usage(argv[0]);
|
|
return 1;
|
|
}
|
|
if (strcmp(argv[1], "accuracy") == 0) {
|
|
const char *k = arg_value(argc, argv, "--kernel-out");
|
|
const char *r = arg_value(argc, argv, "--route-out");
|
|
if (!k || !r) {
|
|
usage(argv[0]);
|
|
return 1;
|
|
}
|
|
mode_accuracy(k, r);
|
|
return 0;
|
|
}
|
|
if (strcmp(argv[1], "microbench") == 0) {
|
|
const char *out = arg_value(argc, argv, "--out");
|
|
const char *tc = arg_value(argc, argv, "--target-calls");
|
|
if (!out) {
|
|
usage(argv[0]);
|
|
return 1;
|
|
}
|
|
mode_microbench(tc ? atol(tc) : 10000000L, out);
|
|
return 0;
|
|
}
|
|
if (strcmp(argv[1], "routebench") == 0) {
|
|
const char *out = arg_value(argc, argv, "--out");
|
|
const char *tc = arg_value(argc, argv, "--target-calls");
|
|
const char *th = arg_value(argc, argv, "--threads");
|
|
const char *ms = arg_value(argc, argv, "--max-seconds");
|
|
if (!out) {
|
|
usage(argv[0]);
|
|
return 1;
|
|
}
|
|
mode_routebench(tc ? atol(tc) : 1000000L, th ? atoi(th) : 4,
|
|
ms ? atof(ms) : 20.0, out);
|
|
return 0;
|
|
}
|
|
usage(argv[0]);
|
|
return 1;
|
|
}
|