Test: Add adaptive HIP tile replay selection
Add a bounded per-chunk center-tile selector to the real-event replay prototype and retain the paired timing, mixed-boundary, CPU, HIP, and sandbox-failure evidence.
This commit is contained in:
1 parent
11e703e33d
commit
776f9b7247
31 files changed
+347
-11
No files matched your search
+41
-11
@@ -102,11 +102,13 @@ static size_t contributions(const std::vector<PsfCachedEvent>& events,int width,
|
||||
return n;
|
||||
}
|
||||
int main(int argc,char **argv) {
|
||||
if(argc<4 || argc>5) {puts("Usage: replay_psf INPUT.events EVENTS(1..65536) atomic|16|32 [disperse|mixed]\nExternal timeout <=45s; sequential processes only.");return 2;}
|
||||
if(argc<4 || argc>5) {puts("Usage: replay_psf INPUT.events EVENTS(1..65536) atomic|16|32|adaptive [disperse|mixed]\nExternal timeout <=45s; sequential processes only.");return 2;}
|
||||
char *end=nullptr;const long requested=strtol(argv[2],&end,10);
|
||||
if(!*argv[2] || *end || requested<1 || requested>65536) return 2;
|
||||
const int tile=!strcmp(argv[3],"atomic") ? 0 : !strcmp(argv[3],"16") ? 16 : !strcmp(argv[3],"32") ? 32 : -1;
|
||||
if(tile<0 || (argc==5 && strcmp(argv[4],"disperse") && strcmp(argv[4],"mixed")))return 2;
|
||||
const bool adaptive=!strcmp(argv[3],"adaptive");
|
||||
const int requested_tile=!strcmp(argv[3],"atomic") ? 0 : !strcmp(argv[3],"16") ? 16 :
|
||||
!strcmp(argv[3],"32") ? 32 : adaptive ? 16 : -1;
|
||||
if(requested_tile<0 || (argc==5 && strcmp(argv[4],"disperse") && strcmp(argv[4],"mixed")))return 2;
|
||||
FILE *f=fopen(argv[1],"r");if(!f){perror(argv[1]);return 2;}
|
||||
int width,height;size_t count;PointSpreadFunction psf;double tail;
|
||||
if(fscanf(f,"PSFEVENTS1 %d %d %zu %lf %lf %lf",&width,&height,&count,&psf.fwhm_pixels,&psf.moffat_beta,&tail)!=6 ||
|
||||
@@ -153,8 +155,10 @@ int main(int argc,char **argv) {
|
||||
const double create=omp_get_wtime()-start;
|
||||
hipDeviceProp_t prop;check(hipGetDeviceProperties(&prop,0));
|
||||
printf("device=%s arch=%s events=%zu frame=%dx%d mode=%s distribution=%s contributions=%zu CPU_reference=%.9f create=%.9f\n",prop.name,prop.gcnArchName,events.size(),width,height,argv[3],mixed?"fixture-mixed-direct":argc==5?"synthetic-phase-preserving-dispersal":"captured",work,cpu_time,create);fflush(stdout);
|
||||
double bin=0,upload=0,prepare=0,kernel=0,merge=0,clear=0;size_t peak_scratch=0,ref_total=0,task_total=0;
|
||||
double select=0,bin=0,upload=0,prepare=0,kernel=0,merge=0,clear=0;
|
||||
size_t peak_scratch=0,ref_total=0,task_total=0,center_tiles=0,tile_chunks=0,atomic_chunks=0;
|
||||
Prepared *prepared=nullptr;int *bounds=nullptr;
|
||||
PsfCachedEvent *tile_events=nullptr;
|
||||
unsigned *drefs=nullptr,*dstarts=nullptr;TileTask *dtasks=nullptr;double *partial=nullptr;
|
||||
auto direct_boundary=[&]() {
|
||||
if(hip_psf_sink_finish(s,gpu.data(),message,sizeof message)) {fprintf(stderr,"%s\n",message);exit(1);}
|
||||
@@ -162,8 +166,9 @@ int main(int argc,char **argv) {
|
||||
if(hip_psf_sink_load_hdr(s,gpu.data(),message,sizeof message)) {fprintf(stderr,"%s\n",message);exit(1);}
|
||||
};
|
||||
const double replay_start=omp_get_wtime();
|
||||
if(tile) {
|
||||
if(requested_tile) {
|
||||
// Fixed hard bounds; no full-frame event list or per-chunk allocation on GPU.
|
||||
check(hipMalloc(&tile_events,16384*sizeof(PsfCachedEvent)));
|
||||
check(hipMalloc(&prepared,16384*sizeof(Prepared)));
|
||||
check(hipMalloc(&bounds,16384*(size_t)(2*cache.radius_pixels+1)*2*sizeof(int)));
|
||||
check(hipMalloc(&drefs,32*1024*1024));check(hipMalloc(&dtasks,2*1024*1024));
|
||||
@@ -171,7 +176,32 @@ int main(int argc,char **argv) {
|
||||
}
|
||||
for(size_t offset=0;offset<events.size();offset+=16384) {
|
||||
const size_t n=std::min((size_t)16384,events.size()-offset);
|
||||
if(!tile) {if(hip_psf_sink_submit(s,events.data()+offset,n,message,sizeof message)) {fprintf(stderr,"%s\n",message);return 1;}if(mixed && offset==0)direct_boundary();continue;}
|
||||
int tile=requested_tile;
|
||||
if(adaptive) {
|
||||
start=omp_get_wtime();
|
||||
const int selector_tile=32,nx=(width+selector_tile-1)/selector_tile;
|
||||
const int ny=(height+selector_tile-1)/selector_tile;
|
||||
std::vector<unsigned char> occupied((size_t)nx*ny,0);
|
||||
size_t count_occupied=0;
|
||||
for(size_t i=0;i<n;++i) {
|
||||
const int x=(int)floor(events[offset+i].x),y=(int)floor(events[offset+i].y);
|
||||
if(x<0 || x>=width || y<0 || y>=height)continue;
|
||||
unsigned char &mark=occupied[(size_t)(y/selector_tile)*nx+x/selector_tile];
|
||||
if(!mark) {mark=1;++count_occupied;}
|
||||
}
|
||||
/* This threshold is deliberately experimental. The captured dense and
|
||||
* lensed chunks have >=45 events per occupied center tile, while the
|
||||
* equal-work dispersed control has about 2.4. Small chunks retain the
|
||||
* lower-overhead production atomic path. */
|
||||
tile=n>=8192 && count_occupied && n/count_occupied>=32 ? 16 : 0;
|
||||
center_tiles+=count_occupied;select+=omp_get_wtime()-start;
|
||||
}
|
||||
if(!tile) {
|
||||
++atomic_chunks;
|
||||
if(hip_psf_sink_submit(s,events.data()+offset,n,message,sizeof message)) {fprintf(stderr,"%s\n",message);return 1;}
|
||||
if(mixed && offset==0)direct_boundary();continue;
|
||||
}
|
||||
++tile_chunks;
|
||||
start=omp_get_wtime();
|
||||
const int nx=(width+tile-1)/tile,ny=(height+tile-1)/tile;
|
||||
std::vector<std::vector<unsigned>> lists(nx*ny);
|
||||
@@ -208,13 +238,13 @@ int main(int argc,char **argv) {
|
||||
peak_scratch=std::max(peak_scratch,bytes+refs.size()*sizeof(unsigned)+tasks.size()*sizeof(TileTask)+starts.size()*sizeof(unsigned)+n*(sizeof(Prepared)+(size_t)(2*cache.radius_pixels+1)*2*sizeof(int)));
|
||||
ref_total+=refs.size();task_total+=tasks.size();bin+=omp_get_wtime()-start;
|
||||
start=omp_get_wtime();
|
||||
check(hipMemcpyAsync(s->slots[0].device,events.data()+offset,n*sizeof(PsfCachedEvent),hipMemcpyHostToDevice,s->stream));
|
||||
check(hipMemcpyAsync(tile_events,events.data()+offset,n*sizeof(PsfCachedEvent),hipMemcpyHostToDevice,s->stream));
|
||||
if(!refs.empty())check(hipMemcpyAsync(drefs,refs.data(),refs.size()*sizeof(unsigned),hipMemcpyHostToDevice,s->stream));
|
||||
if(!tasks.empty())check(hipMemcpyAsync(dtasks,tasks.data(),tasks.size()*sizeof(TileTask),hipMemcpyHostToDevice,s->stream));
|
||||
check(hipMemcpyAsync(dstarts,starts.data(),starts.size()*sizeof(unsigned),hipMemcpyHostToDevice,s->stream));
|
||||
upload+=sync_time(s,start);
|
||||
start=omp_get_wtime();
|
||||
hipLaunchKernelGGL(prepare_tiles,dim3((n*32+127)/128),dim3(128),0,s->stream,s->slots[0].device,n,s->phase_resolution,s->radius_pixels,s->max_radius_pixels,prepared,bounds);
|
||||
hipLaunchKernelGGL(prepare_tiles,dim3((n*32+127)/128),dim3(128),0,s->stream,tile_events,n,s->phase_resolution,s->radius_pixels,s->max_radius_pixels,prepared,bounds);
|
||||
check(hipGetLastError());prepare+=sync_time(s,start);
|
||||
start=omp_get_wtime();
|
||||
if(!tasks.empty())hipLaunchKernelGGL(tile_partial,dim3(tasks.size()),dim3(tile*tile),0,s->stream,prepared,bounds,drefs,dtasks,tile,nx,width,height,s->weights,s->phase_resolution,s->radius_pixels,partial,s->hdr);
|
||||
@@ -225,10 +255,10 @@ int main(int argc,char **argv) {
|
||||
// Every partial pixel is written by exactly one thread: no memset required.
|
||||
}
|
||||
if(hip_psf_sink_finish(s,gpu.data(),message,sizeof message)){fprintf(stderr,"%s\n",message);return 1;}
|
||||
if(tile) {start=omp_get_wtime();check(hipFree(prepared));check(hipFree(bounds));check(hipFree(drefs));check(hipFree(dtasks));check(hipFree(dstarts));check(hipFree(partial));clear=omp_get_wtime()-start;}
|
||||
if(requested_tile) {start=omp_get_wtime();check(hipFree(tile_events));check(hipFree(prepared));check(hipFree(bounds));check(hipFree(drefs));check(hipFree(dtasks));check(hipFree(dstarts));check(hipFree(partial));clear=omp_get_wtime()-start;}
|
||||
const double wall=omp_get_wtime()-replay_start;
|
||||
HipPsfTiming timing={};hip_psf_sink_get_timing(s,&timing);
|
||||
if(!tile){upload=timing.upload_seconds;kernel=timing.kernel_seconds;}
|
||||
upload+=timing.upload_seconds;kernel+=timing.kernel_seconds;
|
||||
double max_abs=0,max_rel=0;long double sums[3]={},diffs[3]={};bool pass=true;
|
||||
for(size_t i=0;i<values;++i) {
|
||||
if(!std::isfinite(cpu[i]) || !std::isfinite(gpu[i]))pass=false;
|
||||
@@ -238,7 +268,7 @@ int main(int argc,char **argv) {
|
||||
}
|
||||
long double ysum=.2126L*sums[0]+.7152L*sums[1]+.0722L*sums[2],ydiff=.2126L*diffs[0]+.7152L*diffs[1]+.0722L*diffs[2];
|
||||
struct rusage usage;getrusage(RUSAGE_SELF,&usage);
|
||||
printf("complete_replay=%.9f replay_wall=%.9f bin=%.9f upload=%.9f prepare=%.9f accumulation=%.9f merge=%.9f download=%.9f cleanup=%.9f events_s=%.3f contributions_s=%.3f refs=%zu tasks=%zu scratch_used_peak=%zu scratch_device_reserved=%zu RSS_KiB=%ld max_abs=%.17g max_rel=%.17g flux_R=%.17Lg flux_G=%.17Lg flux_B=%.17Lg flux_Y=%.17Lg\n",create+wall,wall,bin,upload,prepare,kernel,merge,timing.download_seconds,clear,events.size()/wall,work/wall,ref_total,task_total,peak_scratch,tile?162*1024*1024+131072+16384*(sizeof(Prepared)+(size_t)(2*cache.radius_pixels+1)*2*sizeof(int)):0,usage.ru_maxrss,max_abs,max_rel,sums[0]?diffs[0]/sums[0]:0,sums[1]?diffs[1]/sums[1]:0,sums[2]?diffs[2]/sums[2]:0,ysum?ydiff/ysum:0);
|
||||
printf("complete_replay=%.9f replay_wall=%.9f select=%.9f bin=%.9f upload=%.9f prepare=%.9f accumulation=%.9f merge=%.9f download=%.9f cleanup=%.9f events_s=%.3f contributions_s=%.3f center_tiles=%zu tile_chunks=%zu atomic_chunks=%zu refs=%zu tasks=%zu scratch_used_peak=%zu scratch_device_reserved=%zu RSS_KiB=%ld max_abs=%.17g max_rel=%.17g flux_R=%.17Lg flux_G=%.17Lg flux_B=%.17Lg flux_Y=%.17Lg\n",create+wall,wall,select,bin,upload,prepare,kernel,merge,timing.download_seconds,clear,events.size()/wall,work/wall,center_tiles,tile_chunks,atomic_chunks,ref_total,task_total,peak_scratch,requested_tile?163*1024*1024+131072+16384*(sizeof(Prepared)+(size_t)(2*cache.radius_pixels+1)*2*sizeof(int)):0,usage.ru_maxrss,max_abs,max_rel,sums[0]?diffs[0]/sums[0]:0,sums[1]?diffs[1]/sums[1]:0,sums[2]?diffs[2]/sums[2]:0,ysum?ydiff/ysum:0);
|
||||
for(int c=0;c<3;++c)if(fabsl(diffs[c])>1e-10L*fmaxl(fabsl(sums[c]),1e-30L))pass=false;
|
||||
if(fabsl(ydiff)>1e-10L*fmaxl(fabsl(ysum),1e-30L))pass=false;
|
||||
hip_psf_sink_destroy(s);psf_kernel_cache_destroy(&cache);
|
||||
|
||||
Reference in new issue
Block a user