import re,json,pathlib,statistics,csv
root=pathlib.Path('artifacts/exchange-cost-20260913')
allrows=[]
for name in ['base','gelu']:
 text=(root/f'{name}-run.log').read_text()
 costs={int(p):(src,int(b),int(f),int(ctl),int(c)) for p,src,b,f,ctl,c in re.findall(r'estimated logical exchange phase=(\d+) source=(.*?) bytes=(\d+) fragments=(\d+) controls=(\d+) cycles=(\d+)',text)}
 actual={int(p):int(h)+600 for p,h in re.findall(r'finished large physical exchange phase phase=(\d+) tile=\d+ row_words=\d+ horizon=(\d+)',text)}
 rows=[]
 for p,(src,b,f,ctl,c) in costs.items():
  if p in actual:
   a=actual[p];old=max((b+3)//4+600,f*160)
   rows.append(dict(variant=name,phase=p,source=src,bytes=b,fragments=f,controls=ctl,estimate=c,scheduled=a,old_fragment_model=old,ratio=c/a))
 allrows+=rows
 print(name,'matched',len(rows),'median new/actual',statistics.median(r['ratio'] for r in rows),'median old/actual',statistics.median(r['old_fragment_model']/r['scheduled'] for r in rows))
 print('worst underestimates',sorted(rows,key=lambda r:r['ratio'])[:5])
 for r in rows:
  if r['source'] in ['Some(OperationId(19))','Some(OperationId(20))','Some(OperationId(21))']:print(r)
with (root/'calibration.csv').open('w') as f:
 w=csv.DictWriter(f,fieldnames=allrows[0]);w.writeheader();w.writerows(allrows)
