smodusermc/offline-games / _tools /repair_split_archives.py
smodusermc's picture
download
raw
3.11 kB
"""Repair source-declared archive chunks without loading the game into browser RAM."""
import os,json,requests,hashlib,shutil,time
from pathlib import Path
from huggingface_hub import HfApi
R=Path('/home/user/game-archive/reports');TMP=Path('/tmp/split-game-repair');TMP.mkdir(exist_ok=True)
api=HfApi(token=os.environ['HF_TOKEN']);B='smodusermc/offline-games';BASE='https://garbsoftball.com'
existing={f.path:f for f in api.list_bucket_tree(B,recursive=True) if hasattr(f,'size')}
allnotes=[]
try:
for ident,base,mode in [(33,'/games/Apotheon/Content.tar','count'),(33,'/games/Apotheon/Dialog.tar','count'),(59,'/games/bendy/Build/BATIM.data','sequential')]:
mp=R/f'game-{ident}.json';m=json.loads(mp.read_text());files={x['path']:x for x in m['files']};note={'id':ident,'base':base,'mode':mode,'added':[],'source_terminal':None}
count=int(requests.get(BASE+base+'.count',timeout=20).text.strip()) if mode=='count' else 100
assert 0<count<=100
for i in range(count):
path=base+(f'{i:02d}' if mode=='count' else f'.{i+1:03d}');key='site'+path
if key in files and key in existing:continue
if key in existing:
files[key]={'url':BASE+path,'path':key,'bytes':existing[key].size,'sha256':None,'content_type':'application/octet-stream','note':'Recovered prior interrupted batch object; size verified; checksum not yet re-verified'}
continue
with requests.get(BASE+path,stream=True,timeout=(8,35)) as response:
if response.status_code!=200:
note['source_terminal']={'path':path,'status':response.status_code}
if mode=='count':note['incomplete']=True
break
local=TMP/'chunk';h=hashlib.sha256();size=0;head=b''
with local.open('wb') as f:
for c in response.iter_content(1024*1024):
if not head:
head=c[:100]
if head.lstrip().lower().startswith((b'<!doctype html',b'<html')):break
size+=len(c)
if size>80*1024**2:raise RuntimeError('chunk larger than safe 80 MiB limit')
f.write(c);h.update(c)
if not size:
local.unlink(missing_ok=True);note['source_terminal']={'path':path,'status':'HTML fallback, not binary'};break
entry={'url':BASE+path,'path':key,'bytes':size,'sha256':h.hexdigest(),'content_type':response.headers.get('Content-Type','application/octet-stream')}
api.batch_bucket_files(B,add=[(local,key)])
rf=list(api.get_bucket_paths_info(B,[key]));assert len(rf)==1 and rf[0].size==size
existing[key]=rf[0];local.unlink();files[key]=entry;note['added'].append(entry)
m['files']=list(files.values());m['split_archive_repairs']=m.get('split_archive_repairs',[])+[{k:v for k,v in note.items() if k!='added'}]
mp.write_text(json.dumps(m,indent=2));api.batch_bucket_files(B,add=[(mp,f'_reports/games/{ident:04d}.json')])
allnotes.append(note);print(ident,base,'added',len(note['added']),'bytes',sum(x['bytes'] for x in note['added']),'terminal',note['source_terminal'],flush=True)
out=R/'split-archive-repairs.json';out.write_text(json.dumps(allnotes,indent=2));api.batch_bucket_files(B,add=[(out,'_verification/split-archive-repairs.json')])
finally:shutil.rmtree(TMP,ignore_errors=True)

Xet Storage Details

Size:
3.11 kB
·
Xet hash:
26d07cfa2beca13eee494828ca17ab055066b0cdaf318528a7d55e47802b4b49

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.