smodusermc/offline-games / _tools /publish_large_batch.py
smodusermc's picture
download
raw
13 kB
"""Finalize a completed larger batch without conflating container/media checks with gameplay."""
import os,json,hashlib,collections
from pathlib import Path
from huggingface_hub import HfApi
H=Path('/home/user/game-archive');R=H/'reports';B='smodusermc/offline-games';api=HfApi(token=os.environ['HF_TOKEN'])
plan=json.loads((R/'large-batch-plan.json').read_text());state=json.loads((R/'large-batch-validation.json').read_text());s=json.loads((R/'large-batch-summary.json').read_text());ids=plan['ids'];rows=state['results'];assert len(rows)==len(ids) and state['phase'].startswith('automated batch complete')
batch_id='batch-'+hashlib.sha256(','.join(map(str,ids)).encode()).hexdigest()[:10];prefix='_verification/batches/'+batch_id
visual=json.loads((R/'large-batch-visual-review.json').read_text()) if (R/'large-batch-visual-review.json').exists() else {}
for x in rows:
p=R/f"verification-{x['id']:04d}.json"
if not p.exists():continue
j=json.loads(p.read_text());v=visual.get(str(x['id']))
if v:j.update(reviewed_outcome=v['outcome'],review_note=v['note'])
elif j.get('status')=='runtime_test_inconclusive_resource_limit':j.update(reviewed_outcome='Inconclusive: resource limit',review_note=j.get('reason','Host resource limit. Not evidence that the archive is broken.'))
else:j.update(reviewed_outcome='Automated startup observed; review pending',review_note='No full playthrough or game-specific input assertion in this batch.')
j['asset_validation_report']=prefix+'/validation.json';j['full_playthrough_verified']=False;p.write_text(json.dumps(j,indent=2));x['reviewed_outcome']=j['reviewed_outcome'];x['review_note']=j['review_note']
state['results']=rows;state['batch_id']=batch_id;(R/'large-batch-validation.json').write_text(json.dumps(state,indent=2))
sha=sum(x.get('sha_comparisons',0) for x in rows);unbaselined=sum(len(x.get('unbaselined_files',[])) for x in rows);audio=[f for x in rows for f in x.get('media',[]) if f['kind']=='audio'];images=[f for x in rows for f in x.get('media',[]) if f['kind']=='image'];containers=[f for x in rows for f in x.get('container_checks',[])];missing=[{'game':x['name'],**f} for x in rows for f in x.get('declared_unresolved',[])];repairs=sum(len(x.get('independent_repairs',[])) for x in rows)
info=api.bucket_info(B);text=f'''# Larger batch: {len(ids)} game captures
[Public bucket](https://huggingface.co/buckets/{B})
**Default batch size is now 30, normally 20–40 depending on size.** Payloads are still transferred in small chunks and checked one game at a time. This batch's static pass uploaded approximately 219 MB; runtime/declaration repairs and evidence are additional.
## Completed checks
- **{len(rows)}/{len(ids)} entries** received automated archive and runtime checks.
- **{sha:,} comparisons against recorded SHA-256 hashes**; {unbaselined} files had no prior hash baseline and are distinguished in the report.
- **{len(audio)} separate audio-file decode checks**, {sum(f['decoded'] for f in audio)} successful.
- **{len(images)} image decode checks**, {sum(f['decoded'] for f in images)} successful.
- **{len(containers)} SWF container checks**, {sum(f.get('container_check')=='passed' for f in containers)} passing decompression/length/tag-bound checks. These also inspect explicit ImportAssets records, **not every possible ActionScript-generated URL or embedded sound codec**.
- **{repairs} additional files** recovered by the independent declaration pass, beyond the initial traversal/browser repairs.
- **{s['passed_known_asset_checks']}/{len(rows)} entries passed the scoped asset checks.** This is **not** a count of fully playable games.
## Declared assets still missing
'''
if missing:
for game in sorted({x['game'] for x in missing}):
group=[x for x in missing if x['game']==game]
text+=f"- **{game}**: {len(group)} declared files remain missing at the original source. Types: {', '.join(sorted({x.get('path','').split('.')[-1] for x in group}))}. Full filenames and responses are in the validation JSON.\n"
else:text+='No missing objects in the explicitly enumerated declarations checked.\n'
if any(x['id']==301 and x.get('declared_unresolved') for x in rows):text+='\nKnife Hit’s four missing M4A variants return HTML at the original source. Its OGG counterparts are stored and decode successfully. No renamed substitute or fabricated audio was used; alternate-codec coverage remains incomplete.\n'
text+='\n**There Is No Game repair:** 30 missing OGG clips were recovered from a secondary public mirror, after three existing control clips matched recorded SHA-256 exactly. All 30 were downloaded back, hash-checked and decoded. This is not an author-upstream claim or proof of equivalence with the unavailable original files. Its M4A variants remain source-missing; the OGG counterparts are present. Details: `_verification/large-batch-audio-repair.json`.\n'
text+='\n## Per-game result\n\n| Game | Known asset checks | Runtime evidence / caveat |\n|---|---|---|\n'
for x in rows:text+=f"| {x['name']} | {'Passed scoped checks' if x.get('known_checks_passed') else 'Needs follow-up; see JSON'} | {x.get('reviewed_outcome',x.get('runtime_status','Not tested'))}: {x.get('review_note','See individual report.').replace('|','/')} |\n"
text+=f'''
## What is and is not established
The tests use a local HTTP mirror with external browser traffic blocked. Title/menu screenshots, parsed movies, and valid media do not prove full game progression. Resource-limited or splash-only runs are labeled accordingly. Static-crawler failures also include guessed paths and optional wrappers; the JSON reports retain them rather than silently erasing them.
Original folders remain intact. These are not automatically self-contained HTML files or untouched `file://` releases. No full playthroughs are claimed. Geometry Dash's separate dataset is untouched.
**{s['pending_candidates']} candidates remain queued**, plus two oversized entry pages deferred. Bucket size is approximately **{info.size/1e9:.2f} GB** before final report upload.
## Evidence
Stable batch archive: `{prefix}/` contains the plan, validation details, summary and this report. Individual `_verification/games/` records link to that stable archive. Screenshots are under `_verification/evidence/`.
Scripts, small inventories and notes remain locally. Temporary game payloads, screenshots and bulky caches are removed after publication.
'''
(H/'LARGE-BATCH.md').write_text(text)
api.download_bucket_files(B,[('_reports/inventory.json',R/'inventory.json')],raise_on_missing_files=True);inventory=json.loads((R/'inventory.json').read_text());by={x['id']:x for x in rows}
for r in inventory:
if r['id'] in by:
r['large_batch_validation']=prefix+'/validation.json';r['verification_outcome']=by[r['id']].get('reviewed_outcome',by[r['id']].get('runtime_status'))
(R/'inventory.json').write_text(json.dumps(inventory,indent=2))
allreports=[json.loads(p.read_text()) for p in R.glob('verification-*.json') if p.name!='verification-summary.json'];overall={'entries_with_verification_reports':len(allreports),'latest_batch_ids':ids,'latest_batch_archive':prefix,'pending_catalog_candidates':s['pending_candidates'],'deferred_large_entry_pages':2,'bucket_bytes_before_report_upload':info.size,'full_playthroughs_verified':0,'outcomes':dict(collections.Counter(x.get('reviewed_outcome',x.get('status')) for x in allreports))}
(R/'verification-summary.json').write_text(json.dumps(overall,indent=2))
old=H/'VERIFICATION.md';history=R/'verification-before-large-batch.md'
if not history.exists():history.write_text(old.read_text())
full=text+'\n## Cumulative outcome index\n\n| Game | Outcome |\n|---|---|\n'
for j in sorted(allreports,key=lambda x:x['id']):full+=f"| {j.get('name',j['id'])} | {j.get('reviewed_outcome',j.get('status'))} |\n"
(H/'VERIFICATION.md').write_text(full)
readme=f'''# Public original-file game archive
Source catalog: https://garbsoftball.com/g. Public bucket created at the account owner's request; original rights remain with their owners.
## Batch policy
Target **30 games per batch**, normally **20–40 depending on size**. Transfers and tests remain sequential/small-chunk to control local storage and memory. See `_reports/batch-policy.md`.
## Latest: 30-game batch
All 30 selected entries have capture and automated QA reports. **{s['passed_known_asset_checks']}/30 passed the scoped recorded/declared-asset checks**, not full gameplay tests. {len(missing)} declared-file gaps remain explicitly listed. Runtime resource limits and splash/menu-only evidence are not promoted to playability passes.
Report: `_verification/large-batch-update.md`. Stable archive: `{prefix}/`. Added games include Gun Mayhem 1–3; Learn to Fly 1–3 and Idle; six Papa's games; Raft Wars 1–2; both Impossible Quizzes; Vex; Worlds Hardest Game 2–4; and others listed in the report.
## Earlier repairs retained
- Thirty Dollar Website: 191 selectable sounds + 3 intros stored/decoded; preview and sampled sequence produced nonzero audio.
- Cookie Clicker: 46 effects added, 75 declared sounds decoded. Basket Random: 16 WebM sounds decoded.
- Fireboy/Watergirl 1–4: all 149 declared map paths stored, including documented upstream recoveries.
- OvO: 94 audio files restored; full runtime remains memory-limited on the test host.
Prior reports: `_verification/lazy-assets-update.md` and `_verification/next-batch-update.md`.
## Progress / layout
**{len(allreports)} entries have verification reports; {s['pending_candidates']} candidates remain queued**, plus two oversized entry pages deferred. Historical external-build exclusions remain labeled. No full playthroughs are claimed.
`site/` retains server-relative game files; `_vendor/` contains support libraries; `_originals/` holds backups of deliberate patches; `_reports/`, `_verification/`, and `_tools/` contain manifests, evidence, caveats, and credential-free scripts.
To download site files: `hf buckets sync hf://buckets/smodusermc/offline-games/site ./game-files`.
Some games also need `_vendor/` and HTTP hosting/routing. **These are not automatically single-file HTML or untouched file:// releases.** Online services, ads, telemetry and some source-broken features remain external/unresolved. Geometry Dash's separate dataset is untouched. Revoke shared access tokens when finished authorizing uploads.
'''
(H/'BUCKET-README.md').write_text(readme)
with (H/'WORKLOG.md').open('a') as f:f.write(f'\n\n## {batch_id}: larger batch\n\nDefault changed to 30 games (20–40 by size). Captured and checked all 30 selected entries. {sha} recorded-hash comparisons, {len(audio)} audio checks, {len(images)} image checks, {len(containers)} SWF checks. {len(missing)} declared gaps remain; runtime evidence kept separate from asset checks. {s["pending_candidates"]} candidates remain queued. Stable evidence: {prefix}/. See LARGE-BATCH.md for caveats.\n')
archive=R/'batches'/batch_id;archive.mkdir(parents=True,exist_ok=True)
for src,name in [(R/'large-batch-plan.json','plan.json'),(R/'large-batch-validation.json','validation.json'),(R/'large-batch-summary.json','summary.json'),(H/'LARGE-BATCH.md','report.md')]:
(archive/name).write_bytes(src.read_bytes())
adds=[(H/'BUCKET-README.md','README.md'),(H/'LARGE-BATCH.md','_verification/large-batch-update.md'),(R/'large-batch-validation.json','_verification/large-batch-validation.json'),(H/'VERIFICATION.md','_verification/verification-summary.md'),(H/'VERIFICATION.md','_reports/status.md'),(H/'WORKLOG.md','_reports/worklog.md'),(R/'inventory.json','_reports/inventory.json'),(R/'verification-summary.json','_verification/verification-summary.json'),(R/'verification-summary.json','_reports/summary.json')]
for p in archive.iterdir():adds.append((p,prefix+'/'+p.name))
repair=R/'large-batch-audio-repair.json'
if repair.exists():adds.append((repair,prefix+'/audio-repair.json'))
for name in ['large-batch-audio-repair.json','large-batch-named-audio-discovery.json']:
p=R/name
if p.exists():adds.append((p,'_verification/'+name))
for p in (R/'evidence').glob('0543-*.jpg'):adds.append((p,'_verification/evidence/'+p.name))
for i in ids:
for p,dest in [(R/f'game-{i}.json',f'_reports/games/{i:04d}.json'),(R/f'verification-{i:04d}.json',f'_verification/games/{i:04d}.json')]:
if p.exists():adds.append((p,dest))
for p in (H/'tools').glob('*.py'):adds.append((p,'_tools/'+p.name))
adds.append((H/'tools/collect_next.py','_tools/resume-collector.py'))
for start in range(0,len(adds),25):api.batch_bucket_files(B,add=adds[start:start+25])
remote={f.path:f for f in api.get_bucket_paths_info(B,[dest for _,dest in adds])}
for p,dest in adds:assert remote[dest].size==p.stat().st_size
print(json.dumps({'batch':batch_id,'published_objects':len(adds),'sha_checks':sha,'unbaselined':unbaselined,'audio':len(audio),'images':len(images),'SWF_containers':len(containers),'declared_gaps':len(missing),'total_reports':len(allreports),'pending':s['pending_candidates']},indent=2))

Xet Storage Details

Size:
13 kB
·
Xet hash:
0e59d614648e125c373b6a76603c44563e8b6b5291672600d3b260514f338646

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.