Ray: Balance movie tracing with dynamic chunks

Schedule ray-pool advancement in dynamic chunks of 32 slots to distribute frame-grouped active rays across workers while preserving stable endpoint destinations.

Add 1/4/16-thread lens-map equivalence coverage and document schedule comparisons with complete commands, input data and original logs. The 393-frame 64x36 production-source benchmark improves from 62.13 to 38.69 seconds with byte-identical PNGs and lens maps. Release and Debug test suites pass.
This commit is contained in:
wyj committed 2026-09-06 15:56:37 -04:00
1 parent 8cf6106127
commit 8741597d9d
48 files changed
+895 -4

No files matched your search

+18 -3
View File
@@ -12,8 +12,8 @@ BUILD = Path(sys.argv[1] if len(sys.argv) > 1 else 'build/Release').resolve()
ENV = dict(os.environ, OMP_NUM_THREADS='4')
def run(binary, *args, ok=True):
result = subprocess.run([str(binary), *map(str, args)], env=ENV,
def run(binary, *args, ok=True, env=ENV):
result = subprocess.run([str(binary), *map(str, args)], env=env,
capture_output=True, text=True)
if (result.returncode == 0) != ok:
raise AssertionError(f'{binary.name} {args}: {result.returncode}\n{result.stderr}')
@@ -144,13 +144,28 @@ with tempfile.TemporaryDirectory(prefix='gr-camera-cli-') as directory:
# former still needs refinement. Finishing the empty batch used
# to abort the whole movie in generation 1.
mixed = Path(__file__).parent / 'fixtures/schwarzschild_mixed_refinement.csv'
parallel_map = tmp / 'mixed-parallel.grlens'
result = run(binary, *common, '--height', 36, '--fov-deg', 60,
'--refine-max-level', 3, '--observer-track', mixed,
'--movie-track-samples', '--frames-dir', tmp,
'--frames-prefix', 'mixed', '--verbose')
'--frames-prefix', 'mixed', '--verbose',
'--lens-map-output', parallel_map)
assert 'Ray trace generation 1: frame 0 added' in result.stderr
assert 'Ray trace generation 0: frame 1 added' not in result.stderr
for frame in range(2):
image_payload(tmp / f'mixed_{frame:06d}.png',
dimensions=(64, 36), allow_black=True)
# Thread scheduling must preserve endpoints, frame/sample IDs and
# the resulting adaptive mesh across the entire slab sweep.
for threads in (1, 16):
comparison_map = tmp / f'mixed-{threads}-threads.grlens'
run(binary, *common, '--height', 36, '--fov-deg', 60,
'--refine-max-level', 3, '--observer-track', mixed,
'--movie-track-samples', '--frames-dir', tmp,
'--frames-prefix', f'mixed-{threads}-threads',
'--lens-map-output', comparison_map,
env=dict(ENV, OMP_NUM_THREADS=str(threads)))
assert parallel_map.read_bytes() == comparison_map.read_bytes(), \
f'movie lens map changed with {threads} threads'
print('schwarzschild: movie lens map identical with 1, 4 and 16 threads', flush=True)
print(f'{backend}: CLI checks passed; single/movie PNG identical, map max error {max_error:.3g}', flush=True)