Buckets:
| """Finalize a completed larger batch without conflating container/media checks with gameplay.""" | |
| import os,json,hashlib,collections | |
| from pathlib import Path | |
| from huggingface_hub import HfApi | |
| H=Path('/home/user/game-archive');R=H/'reports';B='smodusermc/offline-games';api=HfApi(token=os.environ['HF_TOKEN']) | |
| plan=json.loads((R/'large-batch-plan.json').read_text());state=json.loads((R/'large-batch-validation.json').read_text());s=json.loads((R/'large-batch-summary.json').read_text());ids=plan['ids'];rows=state['results'];assert len(rows)==len(ids) and state['phase'].startswith('automated batch complete') | |
| batch_id='batch-'+hashlib.sha256(','.join(map(str,ids)).encode()).hexdigest()[:10];prefix='_verification/batches/'+batch_id | |
| visual=json.loads((R/'large-batch-visual-review.json').read_text()) if (R/'large-batch-visual-review.json').exists() else {} | |
| for x in rows: | |
| p=R/f"verification-{x['id']:04d}.json" | |
| if not p.exists():continue | |
| j=json.loads(p.read_text());v=visual.get(str(x['id'])) | |
| if v:j.update(reviewed_outcome=v['outcome'],review_note=v['note']) | |
| elif j.get('status')=='runtime_test_inconclusive_resource_limit':j.update(reviewed_outcome='Inconclusive: resource limit',review_note=j.get('reason','Host resource limit. Not evidence that the archive is broken.')) | |
| else:j.update(reviewed_outcome='Automated startup observed; review pending',review_note='No full playthrough or game-specific input assertion in this batch.') | |
| j['asset_validation_report']=prefix+'/validation.json';j['full_playthrough_verified']=False;p.write_text(json.dumps(j,indent=2));x['reviewed_outcome']=j['reviewed_outcome'];x['review_note']=j['review_note'] | |
| state['results']=rows;state['batch_id']=batch_id;(R/'large-batch-validation.json').write_text(json.dumps(state,indent=2)) | |
| sha=sum(x.get('sha_comparisons',0) for x in rows);unbaselined=sum(len(x.get('unbaselined_files',[])) for x in rows);audio=[f for x in rows for f in x.get('media',[]) if f['kind']=='audio'];images=[f for x in rows for f in x.get('media',[]) if f['kind']=='image'];containers=[f for x in rows for f in x.get('container_checks',[])];missing=[{'game':x['name'],**f} for x in rows for f in x.get('declared_unresolved',[])];repairs=sum(len(x.get('independent_repairs',[])) for x in rows) | |
| info=api.bucket_info(B);text=f'''# Larger batch: {len(ids)} game captures | |
| [Public bucket](https://huggingface.co/buckets/{B}) | |
| **Default batch size is now 30, normally 20–40 depending on size.** Payloads are still transferred in small chunks and checked one game at a time. This batch's static pass uploaded approximately 219 MB; runtime/declaration repairs and evidence are additional. | |
| ## Completed checks | |
| - **{len(rows)}/{len(ids)} entries** received automated archive and runtime checks. | |
| - **{sha:,} comparisons against recorded SHA-256 hashes**; {unbaselined} files had no prior hash baseline and are distinguished in the report. | |
| - **{len(audio)} separate audio-file decode checks**, {sum(f['decoded'] for f in audio)} successful. | |
| - **{len(images)} image decode checks**, {sum(f['decoded'] for f in images)} successful. | |
| - **{len(containers)} SWF container checks**, {sum(f.get('container_check')=='passed' for f in containers)} passing decompression/length/tag-bound checks. These also inspect explicit ImportAssets records, **not every possible ActionScript-generated URL or embedded sound codec**. | |
| - **{repairs} additional files** recovered by the independent declaration pass, beyond the initial traversal/browser repairs. | |
| - **{s['passed_known_asset_checks']}/{len(rows)} entries passed the scoped asset checks.** This is **not** a count of fully playable games. | |
| ## Declared assets still missing | |
| ''' | |
| if missing: | |
| for game in sorted({x['game'] for x in missing}): | |
| group=[x for x in missing if x['game']==game] | |
| text+=f"- **{game}**: {len(group)} declared files remain missing at the original source. Types: {', '.join(sorted({x.get('path','').split('.')[-1] for x in group}))}. Full filenames and responses are in the validation JSON.\n" | |
| else:text+='No missing objects in the explicitly enumerated declarations checked.\n' | |
| if any(x['id']==301 and x.get('declared_unresolved') for x in rows):text+='\nKnife Hit’s four missing M4A variants return HTML at the original source. Its OGG counterparts are stored and decode successfully. No renamed substitute or fabricated audio was used; alternate-codec coverage remains incomplete.\n' | |
| text+='\n**There Is No Game repair:** 30 missing OGG clips were recovered from a secondary public mirror, after three existing control clips matched recorded SHA-256 exactly. All 30 were downloaded back, hash-checked and decoded. This is not an author-upstream claim or proof of equivalence with the unavailable original files. Its M4A variants remain source-missing; the OGG counterparts are present. Details: `_verification/large-batch-audio-repair.json`.\n' | |
| text+='\n## Per-game result\n\n| Game | Known asset checks | Runtime evidence / caveat |\n|---|---|---|\n' | |
| for x in rows:text+=f"| {x['name']} | {'Passed scoped checks' if x.get('known_checks_passed') else 'Needs follow-up; see JSON'} | {x.get('reviewed_outcome',x.get('runtime_status','Not tested'))}: {x.get('review_note','See individual report.').replace('|','/')} |\n" | |
| text+=f''' | |
| ## What is and is not established | |
| The tests use a local HTTP mirror with external browser traffic blocked. Title/menu screenshots, parsed movies, and valid media do not prove full game progression. Resource-limited or splash-only runs are labeled accordingly. Static-crawler failures also include guessed paths and optional wrappers; the JSON reports retain them rather than silently erasing them. | |
| Original folders remain intact. These are not automatically self-contained HTML files or untouched `file://` releases. No full playthroughs are claimed. Geometry Dash's separate dataset is untouched. | |
| **{s['pending_candidates']} candidates remain queued**, plus two oversized entry pages deferred. Bucket size is approximately **{info.size/1e9:.2f} GB** before final report upload. | |
| ## Evidence | |
| Stable batch archive: `{prefix}/` contains the plan, validation details, summary and this report. Individual `_verification/games/` records link to that stable archive. Screenshots are under `_verification/evidence/`. | |
| Scripts, small inventories and notes remain locally. Temporary game payloads, screenshots and bulky caches are removed after publication. | |
| ''' | |
| (H/'LARGE-BATCH.md').write_text(text) | |
| api.download_bucket_files(B,[('_reports/inventory.json',R/'inventory.json')],raise_on_missing_files=True);inventory=json.loads((R/'inventory.json').read_text());by={x['id']:x for x in rows} | |
| for r in inventory: | |
| if r['id'] in by: | |
| r['large_batch_validation']=prefix+'/validation.json';r['verification_outcome']=by[r['id']].get('reviewed_outcome',by[r['id']].get('runtime_status')) | |
| (R/'inventory.json').write_text(json.dumps(inventory,indent=2)) | |
| allreports=[json.loads(p.read_text()) for p in R.glob('verification-*.json') if p.name!='verification-summary.json'];overall={'entries_with_verification_reports':len(allreports),'latest_batch_ids':ids,'latest_batch_archive':prefix,'pending_catalog_candidates':s['pending_candidates'],'deferred_large_entry_pages':2,'bucket_bytes_before_report_upload':info.size,'full_playthroughs_verified':0,'outcomes':dict(collections.Counter(x.get('reviewed_outcome',x.get('status')) for x in allreports))} | |
| (R/'verification-summary.json').write_text(json.dumps(overall,indent=2)) | |
| old=H/'VERIFICATION.md';history=R/'verification-before-large-batch.md' | |
| if not history.exists():history.write_text(old.read_text()) | |
| full=text+'\n## Cumulative outcome index\n\n| Game | Outcome |\n|---|---|\n' | |
| for j in sorted(allreports,key=lambda x:x['id']):full+=f"| {j.get('name',j['id'])} | {j.get('reviewed_outcome',j.get('status'))} |\n" | |
| (H/'VERIFICATION.md').write_text(full) | |
| readme=f'''# Public original-file game archive | |
| Source catalog: https://garbsoftball.com/g. Public bucket created at the account owner's request; original rights remain with their owners. | |
| ## Batch policy | |
| Target **30 games per batch**, normally **20–40 depending on size**. Transfers and tests remain sequential/small-chunk to control local storage and memory. See `_reports/batch-policy.md`. | |
| ## Latest: 30-game batch | |
| All 30 selected entries have capture and automated QA reports. **{s['passed_known_asset_checks']}/30 passed the scoped recorded/declared-asset checks**, not full gameplay tests. {len(missing)} declared-file gaps remain explicitly listed. Runtime resource limits and splash/menu-only evidence are not promoted to playability passes. | |
| Report: `_verification/large-batch-update.md`. Stable archive: `{prefix}/`. Added games include Gun Mayhem 1–3; Learn to Fly 1–3 and Idle; six Papa's games; Raft Wars 1–2; both Impossible Quizzes; Vex; Worlds Hardest Game 2–4; and others listed in the report. | |
| ## Earlier repairs retained | |
| - Thirty Dollar Website: 191 selectable sounds + 3 intros stored/decoded; preview and sampled sequence produced nonzero audio. | |
| - Cookie Clicker: 46 effects added, 75 declared sounds decoded. Basket Random: 16 WebM sounds decoded. | |
| - Fireboy/Watergirl 1–4: all 149 declared map paths stored, including documented upstream recoveries. | |
| - OvO: 94 audio files restored; full runtime remains memory-limited on the test host. | |
| Prior reports: `_verification/lazy-assets-update.md` and `_verification/next-batch-update.md`. | |
| ## Progress / layout | |
| **{len(allreports)} entries have verification reports; {s['pending_candidates']} candidates remain queued**, plus two oversized entry pages deferred. Historical external-build exclusions remain labeled. No full playthroughs are claimed. | |
| `site/` retains server-relative game files; `_vendor/` contains support libraries; `_originals/` holds backups of deliberate patches; `_reports/`, `_verification/`, and `_tools/` contain manifests, evidence, caveats, and credential-free scripts. | |
| To download site files: `hf buckets sync hf://buckets/smodusermc/offline-games/site ./game-files`. | |
| Some games also need `_vendor/` and HTTP hosting/routing. **These are not automatically single-file HTML or untouched file:// releases.** Online services, ads, telemetry and some source-broken features remain external/unresolved. Geometry Dash's separate dataset is untouched. Revoke shared access tokens when finished authorizing uploads. | |
| ''' | |
| (H/'BUCKET-README.md').write_text(readme) | |
| with (H/'WORKLOG.md').open('a') as f:f.write(f'\n\n## {batch_id}: larger batch\n\nDefault changed to 30 games (20–40 by size). Captured and checked all 30 selected entries. {sha} recorded-hash comparisons, {len(audio)} audio checks, {len(images)} image checks, {len(containers)} SWF checks. {len(missing)} declared gaps remain; runtime evidence kept separate from asset checks. {s["pending_candidates"]} candidates remain queued. Stable evidence: {prefix}/. See LARGE-BATCH.md for caveats.\n') | |
| archive=R/'batches'/batch_id;archive.mkdir(parents=True,exist_ok=True) | |
| for src,name in [(R/'large-batch-plan.json','plan.json'),(R/'large-batch-validation.json','validation.json'),(R/'large-batch-summary.json','summary.json'),(H/'LARGE-BATCH.md','report.md')]: | |
| (archive/name).write_bytes(src.read_bytes()) | |
| adds=[(H/'BUCKET-README.md','README.md'),(H/'LARGE-BATCH.md','_verification/large-batch-update.md'),(R/'large-batch-validation.json','_verification/large-batch-validation.json'),(H/'VERIFICATION.md','_verification/verification-summary.md'),(H/'VERIFICATION.md','_reports/status.md'),(H/'WORKLOG.md','_reports/worklog.md'),(R/'inventory.json','_reports/inventory.json'),(R/'verification-summary.json','_verification/verification-summary.json'),(R/'verification-summary.json','_reports/summary.json')] | |
| for p in archive.iterdir():adds.append((p,prefix+'/'+p.name)) | |
| repair=R/'large-batch-audio-repair.json' | |
| if repair.exists():adds.append((repair,prefix+'/audio-repair.json')) | |
| for name in ['large-batch-audio-repair.json','large-batch-named-audio-discovery.json']: | |
| p=R/name | |
| if p.exists():adds.append((p,'_verification/'+name)) | |
| for p in (R/'evidence').glob('0543-*.jpg'):adds.append((p,'_verification/evidence/'+p.name)) | |
| for i in ids: | |
| for p,dest in [(R/f'game-{i}.json',f'_reports/games/{i:04d}.json'),(R/f'verification-{i:04d}.json',f'_verification/games/{i:04d}.json')]: | |
| if p.exists():adds.append((p,dest)) | |
| for p in (H/'tools').glob('*.py'):adds.append((p,'_tools/'+p.name)) | |
| adds.append((H/'tools/collect_next.py','_tools/resume-collector.py')) | |
| for start in range(0,len(adds),25):api.batch_bucket_files(B,add=adds[start:start+25]) | |
| remote={f.path:f for f in api.get_bucket_paths_info(B,[dest for _,dest in adds])} | |
| for p,dest in adds:assert remote[dest].size==p.stat().st_size | |
| print(json.dumps({'batch':batch_id,'published_objects':len(adds),'sha_checks':sha,'unbaselined':unbaselined,'audio':len(audio),'images':len(images),'SWF_containers':len(containers),'declared_gaps':len(missing),'total_reports':len(allreports),'pending':s['pending_candidates']},indent=2)) | |
Xet Storage Details
- Size:
- 13 kB
- Xet hash:
- 0e59d614648e125c373b6a76603c44563e8b6b5291672600d3b260514f338646
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.