Buckets:
| """Sequential 20–40-entry QA pipeline. Credentials only via HF_TOKEN. | |
| Kept separate from capture; never run simultaneously with another shared-stage verifier. | |
| """ | |
| import os,sys,json,hashlib,shutil,subprocess,tempfile,zlib,struct,ast,re,collections,threading,importlib.util,time | |
| from pathlib import Path | |
| from urllib.parse import urljoin,urlparse,unquote | |
| from huggingface_hub import HfApi | |
| from playwright.sync_api import sync_playwright | |
| H=Path('/home/user/game-archive');R=H/'reports';T=H/'tools';B='smodusermc/offline-games';api=HfApi(token=os.environ['HF_TOKEN']) | |
| plan=json.loads((R/'large-batch-plan.json').read_text());IDS=plan['ids'];results=[];started=time.time() | |
| # Import only pure discovery functions, not the collector's main program. | |
| nodes=[] | |
| for n in ast.parse((T/'collect_next.py').read_text()).body: | |
| if isinstance(n,(ast.Import,ast.ImportFrom)):nodes.append(n) | |
| elif isinstance(n,ast.FunctionDef) and n.name in ['normalized','dest','extract']:nodes.append(n) | |
| elif isinstance(n,ast.Assign) and any(isinstance(x,ast.Name) and x.id in ['EXT','PATH_RE','LITERAL','IGNORE_HOSTS'] for x in n.targets):nodes.append(n) | |
| ns={'BASE':'https://garbsoftball.com'};exec(compile(ast.Module(body=nodes,type_ignores=[]),'discovery','exec'),ns) | |
| def swf_info(data): | |
| result={'signature':data[:3].decode('ascii',errors='replace')} | |
| if len(data)<8:raise ValueError('SWF header truncated') | |
| expected=int.from_bytes(data[4:8],'little');result['declared_uncompressed_bytes']=expected | |
| if expected>192*1024**2:return {**result,'container_check':'deferred: expansion exceeds safety limit'} | |
| if data[:3]==b'CWS': | |
| dec=zlib.decompressobj();body=dec.decompress(data[8:],expected+1) | |
| if not dec.eof:raise ValueError('Compressed SWF incomplete or larger than declared') | |
| elif data[:3]==b'FWS':body=data[8:] | |
| elif data[:3]==b'ZWS':return {**result,'container_check':'LZMA SWF present; deep parse not implemented'} | |
| else:raise ValueError('Not a SWF container') | |
| if len(body)+8!=expected:raise ValueError(f'SWF length mismatch: {len(body)+8} != {expected}') | |
| if not body:raise ValueError('Empty SWF') | |
| pos=(5+4*(body[0]>>3)+7)//8+4;count=0;imports=[];end=False | |
| while pos+2<=len(body): | |
| head=int.from_bytes(body[pos:pos+2],'little');pos+=2;tag=head>>6;length=head&63 | |
| if length==63: | |
| if pos+4>len(body):raise ValueError('Truncated long tag header') | |
| length=int.from_bytes(body[pos:pos+4],'little');pos+=4 | |
| if pos+length>len(body):raise ValueError('SWF tag extends beyond file') | |
| if tag in [57,71]:imports.append(body[pos:pos+length].split(b'\0',1)[0].decode('utf-8',errors='replace')) | |
| pos+=length;count+=1 | |
| if tag==0:end=True;break | |
| return {**result,'container_check':'passed','tags':count,'end_tag':end,'import_urls':imports,'scope':'Container structure and explicit ImportAssets tags only; not an exhaustive ActionScript URL analysis.'} | |
| def publish_progress(phase): | |
| state={'phase':phase,'target_count':len(IDS),'completed_checks':len(results),'seconds_elapsed':round(time.time()-started),'results':results} | |
| (R/'large-batch-validation.json').write_text(json.dumps(state,indent=2)) | |
| text=f'# Larger batch: {len(IDS)} selected games\n\n**Phase: {phase}. Checked {len(results)}/{len(IDS)} entries.**\n\n' | |
| text+='Files are captured in chunks; games are checked one at a time. Recorded-file integrity, declared assets and automated runtime checks are separate from full gameplay. No full-playthrough pass is claimed.\n\n| Game | Files / SHA comparisons | Declared missing | Media failures | Automated runtime result |\n|---|---:|---:|---:|---|\n' | |
| for x in results:text+=f"| {x['name']} | {x.get('files',0)} / {x.get('sha_comparisons',0)} | {len(x.get('declared_unresolved',[]))} | {len(x.get('media_failures',[]))} | {x.get('runtime_status',x.get('error','not run'))} |\n" | |
| text+='\nSee `_verification/large-batch-validation.json` and individual game reports for exact failures and scope. Browser startup is not promoted to playability.\n' | |
| (H/'LARGE-BATCH.md').write_text(text) | |
| api.batch_bucket_files(B,add=[(R/'large-batch-validation.json','_verification/large-batch-validation.json'),(H/'LARGE-BATCH.md','_verification/large-batch-update.md')]) | |
| publish_progress('loading captured manifests') | |
| paths={f.path for f in api.list_bucket_tree(B,prefix='_reports/games',recursive=True) if hasattr(f,'size')} | |
| for ident in IDS: | |
| if f'_reports/games/{ident:04d}.json' not in paths: | |
| results.append({'id':ident,'name':f'Catalog {ident}','error':'No captured manifest: capture deferred or incomplete'});continue | |
| api.download_bucket_files(B,[(f'_reports/games/{ident:04d}.json',R/f'game-{ident}.json')],raise_on_missing_files=True) | |
| # Existing isolated browser verifier handles observed dynamic resources and blocks browser external traffic. | |
| subprocess.run([sys.executable,str(T/'run_isolated.py'),str(ident)],check=True) | |
| sys.argv=[sys.argv[0]] | |
| spec=importlib.util.spec_from_file_location('batch_verifier',T/'verify_games.py');v=importlib.util.module_from_spec(spec);spec.loader.exec_module(v) | |
| mp=R/f'game-{ident}.json';m=json.loads(mp.read_text());j={'id':ident,'name':m['name'],'entry':m['entry'],'declared_unresolved':[],'declaration_inputs':[],'independent_repairs':[],'container_checks':[]} | |
| server=None;shutil.rmtree(v.STAGE,ignore_errors=True);v.STAGE.mkdir() | |
| try: | |
| records={x['path']:x for x in m['files']};v.api.download_bucket_files(B,[(key,v.STAGE/key) for key in records],raise_on_missing_files=True) | |
| state=v.State(m);seen=set();expected={};parse_queue=list(records) | |
| for key in parse_queue: | |
| if key in seen:continue | |
| seen.add(key);path=v.STAGE/key | |
| if not path.exists():continue | |
| refs=[];url=records.get(key,{}).get('url') or 'https://garbsoftball.com/'+key.removeprefix('site/') | |
| if key.endswith('.swf'): | |
| try: | |
| info=swf_info(path.read_bytes());j['container_checks'].append({'path':key,**info}) | |
| refs.extend((urljoin(url,u),'SWF ImportAssets') for u in info.get('import_urls',[])) | |
| except Exception as e:j['container_checks'].append({'path':key,'container_check':'failed','error':str(e)}) | |
| name=path.name | |
| if name in ['offline.json','offline.js','data.json','data.js','game.json','temple.json']: | |
| try:doc=json.loads(path.read_text(encoding='utf-8-sig')) | |
| except (ValueError,UnicodeError):doc=None | |
| if doc is not None: | |
| j['declaration_inputs'].append(key) | |
| if isinstance(doc,dict) and isinstance(doc.get('fileList'),list):refs.extend((urljoin(url,f),'offline fileList') for f in doc['fileList'] if isinstance(f,str)) | |
| for u,kind in ns['extract'](path.read_text(encoding='utf-8-sig'),url,m['entry'])[0]: | |
| if kind.startswith(('construct','temple-','phaser-')) or ('/images/' in u and name in ['data.js','data.json']):refs.append((u,kind)) | |
| for u,reason in refs: | |
| normalized=ns['normalized'](u) | |
| if not normalized: | |
| if reason=='SWF ImportAssets':j['declared_unresolved'].append({'url':u,'reason':'External SWF import not automatically mirrored'}) | |
| continue | |
| dep=ns['dest'](normalized);expected[dep]={'url':normalized,'reason':reason} | |
| if dep not in records: | |
| found=state.fetch(normalized) | |
| if found: | |
| records[found]=state.added[found];parse_queue.append(found) | |
| else:j['declared_unresolved'].append({'path':dep,'url':normalized,'reason':state.failed.get(dep,'Fetch failed')}) | |
| adds=[(v.STAGE/key,key) for key in state.added] | |
| if adds: | |
| for start in range(0,len(adds),24):api.batch_bucket_files(B,add=adds[start:start+24]) | |
| # Read restored bytes back, rather than checking only the upload's local source. | |
| api.download_bucket_files(B,[(key,v.STAGE/key) for key in state.added],raise_on_missing_files=True) | |
| m['files']=list(records.values());mp.write_text(json.dumps(m,indent=2));api.batch_bucket_files(B,add=[(mp,f'_reports/games/{ident:04d}.json')]);j['independent_repairs']=list(state.added.values()) | |
| j['declared_paths']=expected;j['files']=len(records);j['sha_comparisons']=0;j['unbaselined_files']=[];j['sha_mismatches']=[];j['size_mismatches']=[];j['json_errors']=[] | |
| for key,f in records.items(): | |
| path=v.STAGE/key;sha=hashlib.file_digest(path.open('rb'),'sha256').hexdigest() | |
| if path.stat().st_size!=f['bytes']:j['size_mismatches'].append(key) | |
| if f.get('sha256'): | |
| j['sha_comparisons']+=1 | |
| if sha!=f['sha256']:j['sha_mismatches'].append(key) | |
| else:j['unbaselined_files'].append({'path':key,'stored_sha256':sha,'note':'No recorded original-source hash; current bytes only.'}) | |
| if key.endswith('.json'): | |
| try:json.loads(path.read_text(encoding='utf-8-sig')) | |
| except Exception as e:j['json_errors'].append({'path':key,'error':str(e)[:150]}) | |
| # Validate actual audio/image files, not whether they happened to be used on the menu. | |
| (v.STAGE/'site/__batch_fixture__.html').write_text('<!doctype html><title>Media validation</title>') | |
| state.repair=False;server=v.ThreadingHTTPServer(('127.0.0.1',0),v.Handler);server.state=state;server.daemon_threads=False;threading.Thread(target=server.serve_forever,daemon=True).start() | |
| media=[{'path':'/'+key.removeprefix('site/'),'kind':'audio' if re.search(r'\.(wav|mp3|ogg|m4a|webm|opus|aac|flac)$',key,re.I) else 'image'} for key in records if key.startswith('site/') and re.search(r'\.(wav|mp3|ogg|m4a|webm|opus|aac|flac|png|jpe?g|webp|gif|svg|avif)$',key,re.I)] | |
| j['media']=[] | |
| if media: | |
| with sync_playwright() as p: | |
| b=p.chromium.launch(headless=True,args=['--no-sandbox','--disable-dev-shm-usage','--js-flags=--expose-gc']);c=b.new_context(service_workers='block');origin=f'http://127.0.0.1:{server.server_port}' | |
| c.route('**/*',lambda r:r.continue_() if r.request.url.startswith(origin+'/') else r.abort());page=c.new_page();page.goto(origin+'/__batch_fixture__.html') | |
| j['media']=page.evaluate('''async(fs)=>{const ctx=new AudioContext();const out=[];async function one(f){try{if(f.kind==='audio'){const r=await fetch(f.path);if(!r.ok)throw Error(r.status);const a=await ctx.decodeAudioData(await r.arrayBuffer());return {...f,decoded:true,frames:a.length,duration:a.duration};}const img=new Image();img.src=f.path;await img.decode();return {...f,decoded:true,width:img.naturalWidth,height:img.naturalHeight};}catch(e){return {...f,decoded:false,error:String(e)}}}for(const f of fs){out.push(await one(f));window.gc?.();}await ctx.close();return out;}''',media);b.close() | |
| j['media_failures']=[x for x in j['media'] if not x['decoded']] | |
| runtime=json.loads((R/f'verification-{ident:04d}.json').read_text());j['runtime_status']=runtime.get('status');j['runtime_missing']=runtime.get('offline_missing_requests',{});j['page_errors']=runtime.get('offline_pass',{}).get('page_errors',[]) | |
| j['known_checks_passed']=not any(j[k] for k in ['declared_unresolved','sha_mismatches','size_mismatches','json_errors','media_failures']) and not any(x.get('container_check')=='failed' for x in j['container_checks']) | |
| j['full_playthrough_verified']=False | |
| except Exception as e:j['error']=f'{type(e).__name__}: {e}'[:700];j['known_checks_passed']=False | |
| finally: | |
| if server:server.shutdown();server.server_close() | |
| shutil.rmtree(v.STAGE,ignore_errors=True) | |
| results.append(j);publish_progress('checking games');print('BATCH CHECK',len(results),'/',len(IDS),ident,j.get('known_checks_passed'),flush=True) | |
| # Preserve per-game evidence as soon as each test finishes. | |
| vp=R/f'verification-{ident:04d}.json';evidence=[(p,'_verification/evidence/'+p.name) for p in (R/'evidence').glob(f'{ident:04d}-*.jpg')] | |
| if vp.exists():evidence.append((vp,f'_verification/games/{ident:04d}.json')) | |
| if evidence:api.batch_bucket_files(B,add=evidence) | |
| subprocess.run([sys.executable,str(T/'reconcile_manifests.py')],check=True) | |
| api.download_bucket_files(B,[('_reports/inventory.json',R/'inventory.json')],raise_on_missing_files=True);inventory=json.loads((R/'inventory.json').read_text());checked={x['id']:x for x in results} | |
| for row in inventory: | |
| i=row['id'] | |
| if i in checked: | |
| row['large_batch_validation']='_verification/large-batch-validation.json';row['verification_outcome']=checked[i].get('runtime_status','Not tested');row['verification_report']=f'_verification/games/{i:04d}.json' | |
| mp=R/f'game-{i}.json' | |
| if mp.exists(): | |
| m=json.loads(mp.read_text());row['captured_files']=len(m['files']);row['captured_bytes']=sum(x['bytes'] for x in m['files']) | |
| (R/'inventory.json').write_text(json.dumps(inventory,indent=2));api.batch_bucket_files(B,add=[(R/'inventory.json','_reports/inventory.json')]);publish_progress('automated batch complete; inspect caveats') | |
| # Keep a small visual contact sheet for review; full screenshots already uploaded. | |
| from PIL import Image,ImageDraw | |
| for start in range(0,len(IDS),10): | |
| sheet=Image.new('RGB',(1280,420),'#172030');draw=ImageDraw.Draw(sheet) | |
| for n,i in enumerate(IDS[start:start+10]): | |
| x=n%5*256;y=n//5*210;entry=checked.get(i,{});draw.text((x+4,y+3),str(i)+' '+entry.get('name','')[:28],fill='white');p=R/'evidence'/f'{i:04d}-offline.jpg' | |
| if p.exists(): | |
| image=Image.open(p);image.thumbnail((256,180));sheet.paste(image,(x,y+25)) | |
| else:draw.text((x+8,y+60),entry.get('runtime_status','No screenshot')[:29],fill='#dddddd') | |
| name=f'large-batch-overview-{start//10+1}.jpg';sheet.save('/tmp/'+name,quality=75);api.batch_bucket_files(B,add=[(Path('/tmp')/name,'_verification/evidence/'+name)]) | |
| summary={'batch_size':len(IDS),'batch_reports':len(results),'passed_known_asset_checks':sum(x.get('known_checks_passed',False) for x in results),'pending_candidates':sum(x['status']=='pending_not_captured' for x in inventory),'runtime_status_counts':dict(collections.Counter(x.get('runtime_status','not tested') for x in results)),'bucket_bytes':api.bucket_info(B).size,'note':'Automated checks, not full playthroughs. See detailed reports before marking individual games playable.'} | |
| (R/'large-batch-summary.json').write_text(json.dumps(summary,indent=2));api.batch_bucket_files(B,add=[(R/'large-batch-summary.json','_verification/large-batch-summary.json')]);print('BATCH FINISHED',json.dumps(summary),flush=True) | |
Xet Storage Details
- Size:
- 13.9 kB
- Xet hash:
- 3fdf7daf43ba75569c13c70909944ac28a700faedede00dde428b6a8d46c3c3f
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.