diff --git a/tests/replay_psf.hip b/tests/replay_psf.hip index 88aef0e..53e029a 100644 --- a/tests/replay_psf.hip +++ b/tests/replay_psf.hip @@ -6,6 +6,8 @@ #include #include #include +#include +#include #include struct TileTask { unsigned tile, first, count, partial; }; @@ -102,13 +104,26 @@ static size_t contributions(const std::vector& events,int width, return n; } int main(int argc,char **argv) { - if(argc<4 || argc>5) {puts("Usage: replay_psf INPUT.events EVENTS(1..65536) atomic|16|32|adaptive [disperse|mixed]\nExternal timeout <=45s; sequential processes only.");return 2;} + if(argc<4 || argc>6) {puts("Usage: replay_psf INPUT.events EVENTS(1..65536) atomic|16|32|adaptive|production|production-prepared|production-parallel|production-busy [disperse|mixed] [rounds=1..512]\nExternal timeout <=45s; sequential processes only. production-busy adds 15 CPU spin workers to test host contention. Repeated rounds compare against a scaled CPU double reference.");return 2;} char *end=nullptr;const long requested=strtol(argv[2],&end,10); if(!*argv[2] || *end || requested<1 || requested>65536) return 2; const bool adaptive=!strcmp(argv[3],"adaptive"); - const int requested_tile=!strcmp(argv[3],"atomic") ? 0 : !strcmp(argv[3],"16") ? 16 : + const bool parallel_production=!strcmp(argv[3],"production-parallel"); + const bool busy_production=!strcmp(argv[3],"production-busy"); + const bool prepared_production=!strcmp(argv[3],"production-prepared"); + const bool production=!strcmp(argv[3],"production") || prepared_production || parallel_production || busy_production; + const int requested_tile=(!strcmp(argv[3],"atomic") || production) ? 0 : !strcmp(argv[3],"16") ? 16 : !strcmp(argv[3],"32") ? 32 : adaptive ? 16 : -1; - if(requested_tile<0 || (argc==5 && strcmp(argv[4],"disperse") && strcmp(argv[4],"mixed")))return 2; + const bool distribution_arg=argc>=5 && (!strcmp(argv[4],"disperse") || !strcmp(argv[4],"mixed")); + if(requested_tile<0 || (argc==6 && !distribution_arg))return 2; + const char *rounds_arg=argc==6 ? argv[5] : argc==5 && !distribution_arg ? argv[4] : nullptr; + long rounds=1; + if(rounds_arg) { + char *rounds_end=nullptr; + rounds=strtol(rounds_arg,&rounds_end,10); + if(!*rounds_arg || *rounds_end || rounds<1 || rounds>512)return 2; + } + if(parallel_production && (rounds<16 || rounds%16 || distribution_arg)) return 2; FILE *f=fopen(argv[1],"r");if(!f){perror(argv[1]);return 2;} int width,height;size_t count;PointSpreadFunction psf;double tail; if(fscanf(f,"PSFEVENTS1 %d %d %zu %lf %lf %lf",&width,&height,&count,&psf.fwhm_pixels,&psf.moffat_beta,&tail)!=6 || @@ -127,8 +142,8 @@ int main(int argc,char **argv) { fclose(f); // Wing-clipped events retain their physical radius; traversal alone is cache-limited. const size_t original_contributions=contributions(events,width,height,cache); - const bool mixed=argc==5 && !strcmp(argv[4],"mixed"); - if(argc==5 && !mixed) { + const bool mixed=distribution_arg && !strcmp(argv[4],"mixed"); + if(distribution_arg && !mixed) { unsigned state=12345; for(auto&e:events) { // Preserve fractional phase and clipping: move only fully interior events. @@ -154,7 +169,7 @@ int main(int argc,char **argv) { start=omp_get_wtime();if(hip_psf_sink_create(&s,width,height,&cache,16384,message,sizeof message)){fprintf(stderr,"%s\n",message);return 1;} const double create=omp_get_wtime()-start; hipDeviceProp_t prop;check(hipGetDeviceProperties(&prop,0)); - printf("device=%s arch=%s events=%zu frame=%dx%d mode=%s distribution=%s contributions=%zu CPU_reference=%.9f create=%.9f\n",prop.name,prop.gcnArchName,events.size(),width,height,argv[3],mixed?"fixture-mixed-direct":argc==5?"synthetic-phase-preserving-dispersal":"captured",work,cpu_time,create);fflush(stdout); + printf("device=%s arch=%s events=%zu rounds=%ld frame=%dx%d mode=%s distribution=%s contributions_per_round=%zu CPU_reference=%.9f create=%.9f\n",prop.name,prop.gcnArchName,events.size(),rounds,width,height,argv[3],mixed?"fixture-mixed-direct":distribution_arg?"synthetic-phase-preserving-dispersal":"captured",work,cpu_time,create);fflush(stdout); double select=0,bin=0,upload=0,prepare=0,kernel=0,merge=0,clear=0; size_t peak_scratch=0,ref_total=0,task_total=0,center_tiles=0,tile_chunks=0,atomic_chunks=0; Prepared *prepared=nullptr;int *bounds=nullptr; @@ -166,6 +181,17 @@ int main(int argc,char **argv) { if(hip_psf_sink_load_hdr(s,gpu.data(),message,sizeof message)) {fprintf(stderr,"%s\n",message);exit(1);} }; const double replay_start=omp_get_wtime(); + std::atomic stop_busy(false); + std::vector busy_workers; + if(busy_production) for(unsigned worker=0;worker<15;++worker) + busy_workers.emplace_back([&,worker] { + volatile uint64_t state=worker+1; + while(!stop_busy.load(std::memory_order_relaxed)) + for(int i=0;i<4096;++i)state=state*1664525u+1013904223u; + }); + HipPsfPreparedChunk *serial_chunk=nullptr; + if(prepared_production && hip_psf_prepared_chunk_create( + &serial_chunk,s,message,sizeof message)) {fprintf(stderr,"%s\n",message);return 1;} if(requested_tile) { // Fixed hard bounds; no full-frame event list or per-chunk allocation on GPU. check(hipMalloc(&tile_events,16384*sizeof(PsfCachedEvent))); @@ -174,7 +200,34 @@ int main(int argc,char **argv) { check(hipMalloc(&drefs,32*1024*1024));check(hipMalloc(&dtasks,2*1024*1024)); check(hipMalloc(&dstarts,131072));check(hipMalloc(&partial,128*1024*1024)); } - for(size_t offset=0;offset1e-10L*fmaxl(fabsl(sums[c]),1e-30L))pass=false; if(fabsl(ydiff)>1e-10L*fmaxl(fabsl(ysum),1e-30L))pass=false; hip_psf_sink_destroy(s);psf_kernel_cache_destroy(&cache); - return pass && max_abs<1e-10 && max_rel<1e-10 ? 0 : 1; + return pass && max_rel<1e-10 && (rounds>1 || max_abs<1e-10) ? 0 : 1; }