from pathlib import Path
import subprocess,json
root=Path(__file__).resolve().parent
binary=root/'target/release/ipu-refactor-source-size'
records={}
for ref in ['1501698','HEAD','worktree']:
    listing = ['git', 'ls-files', '--cached', '--others', '--exclude-standard', '--', 'crates', 'device'] if ref == 'worktree' else ['git', 'ls-tree', '-r', '--name-only', ref]
    files=subprocess.check_output(listing,text=True).splitlines()
    inputs=[]
    for name in files:
        path=Path(name)
        if not name.startswith(('crates/','device/')) or path.suffix not in {'.rs','.cpp','.S','.inc','.h','.def'}: continue
        if name.startswith('crates/ipu-tests/') or any(part in {'tests','benches'} for part in path.parts): continue
        if path.name in {'tests.rs','test_support.rs'} or path.stem.startswith('test_') or path.stem.endswith(('_tests','_test','_bench')): continue
        if ref == 'worktree' and not path.exists(): continue
        source=path.read_text() if ref=='worktree' else subprocess.check_output(['git','show',ref+':'+name],text=True)
        inputs.append(json.dumps({'path':name,'source':source}))
    p=subprocess.run([str(binary)],input='\n'.join(inputs)+'\n',text=True,capture_output=True,check=True)
    rows=[json.loads(x) for x in p.stdout.splitlines()]
    records[ref]={x['path']:x for x in rows}
    (root/(ref+'.json')).write_text(json.dumps(rows,indent=2)+'\n')
    print(ref,{field:sum(x[field] for x in rows) for field in ['nonblank_production_lines','production_code_lines']},flush=True)
base=records['1501698'];current=records['worktree']
changes=[]
for path in set(base)|set(current):
    delta=current.get(path,{}).get('production_code_lines',0)-base.get(path,{}).get('production_code_lines',0)
    if delta:changes.append((delta,path))
(root/'changes.json').write_text(json.dumps(sorted(changes,reverse=True),indent=2)+'\n')
print('Largest growth:',sorted(changes,reverse=True)[:12])
print('Largest removals:',sorted(changes)[:8])
