smodusermc/offline-games / _tools /restore_next_batch_upstream.py
smodusermc's picture
download
raw
3.71 kB
"""Restore source-missing declared assets from identified upstream releases/builds."""
import os,json,hashlib,tempfile,concurrent.futures,re
from pathlib import Path
import requests
from huggingface_hub import HfApi
H=Path('/home/user/game-archive');R=H/'reports';B='smodusermc/offline-games';api=HfApi(token=os.environ['HF_TOKEN']);audits=json.loads((R/'next-batch-declared-assets.json').read_text());output=json.loads((R/'next-batch-upstream-restorations.json').read_text()) if (R/'next-batch-upstream-restorations.json').exists() else []
origins={181:('site/games/watergirl-2/','https://html5.gamedistribution.com/383ad09b92c7446b9113cccc29630517/'),182:('site/games/watergirl-3/play/','https://html5.gamedistribution.com/f3a6e1ac0a77412289cbac47658b2b68/'),183:('site/games/watergirl-4/','https://html5.gamedistribution.com/3790681b69584409b7f681a8e400102d/'),373:('site/games/ovo/','https://dedragames.com/games/ovo/1.4.4/')}
for a in audits:
i=a['game_id']
if i not in origins:continue
prefix,upstream=origins[i];m=json.loads((R/f'game-{i}.json').read_text());records={f['path']:f for f in m['files']};jobs=[]
for f in a['unresolved']:
if f['path'].startswith(prefix) and f['path'] not in records:jobs.append({**f,'upstream_url':upstream+f['path'].removeprefix(prefix)})
if not jobs:continue
with tempfile.TemporaryDirectory(prefix='upstream-batch-assets-') as temp:
def fetch(f):
try:
r=requests.get(f['upstream_url'],timeout=30);r.raise_for_status();body=r.content
assert len(body)<32*1024**2 and not re.match(br'\s*(?:<!doctype html|<html)',body,re.I),'Unexpected HTML or size'
if f['path'].endswith('.json'):
j=json.loads(body);assert 'layers' in j and 'tilesets' in j,'Not a Tiled map'
elif f['path'].endswith('.ogg'):assert body.startswith(b'OggS'),'Not Ogg'
elif f['path'].endswith('.m4a'):assert body[4:8]==b'ftyp','Not MP4 audio'
local=Path(temp)/hashlib.sha256(f['path'].encode()).hexdigest();local.write_bytes(body)
return f,local,{'path':f['path'],'url':f['upstream_url'],'bytes':len(body),'sha256':hashlib.sha256(body).hexdigest(),'content_type':r.headers.get('content-type'),'provenance':{'original_url':f['url'],'original_failure':f['error'],'source':'Named upstream release/build identified by the stored game loader','byte_equivalence_to_missing_mirror':'not established'}}
except Exception as e:return f,None,{'error':str(e)[:180]}
for start in range(0,len(jobs),16):
with concurrent.futures.ThreadPoolExecutor(max_workers=4) as ex:items=list(ex.map(fetch,jobs[start:start+16]))
adds=[(p,f['path']) for f,p,rec in items if p]
if adds:api.batch_bucket_files(B,add=adds)
for f,p,rec in items:
if p:
records[f['path']]=rec;f['resolved_by_upstream']=rec['url'];output.append({'game_id':i,**rec});p.unlink()
else:output.append({'game_id':i,'path':f['path'],'upstream_url':f['upstream_url'],**rec})
m['files']=list(records.values());(R/f'game-{i}.json').write_text(json.dumps(m,indent=2));api.batch_bucket_files(B,add=[(R/f'game-{i}.json',f'_reports/games/{i:04d}.json')])
resolved={f['path']:f['url'] for f in output if f['game_id']==i and 'sha256' in f}
for f in a['unresolved']:
if f['path'] in resolved:f['resolved_by_upstream']=resolved[f['path']]
print(i,'upstream restored',len(resolved),'of',len(jobs),flush=True)
(R/'next-batch-upstream-restorations.json').write_text(json.dumps(output,indent=2));(R/'next-batch-declared-assets.json').write_text(json.dumps(audits,indent=2))
api.batch_bucket_files(B,add=[(R/'next-batch-upstream-restorations.json','_verification/next-batch-upstream-restorations.json'),(R/'next-batch-declared-assets.json','_verification/next-batch-declared-assets.json')])

Xet Storage Details

Size:
3.71 kB
·
Xet hash:
a5aa75415586575418f0633d49c1d79024fc211496af4894cbf7bf3ea5e713df

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.