Buckets:
| """Repair source-declared archive chunks without loading the game into browser RAM.""" | |
| import os,json,requests,hashlib,shutil,time | |
| from pathlib import Path | |
| from huggingface_hub import HfApi | |
| R=Path('/home/user/game-archive/reports');TMP=Path('/tmp/split-game-repair');TMP.mkdir(exist_ok=True) | |
| api=HfApi(token=os.environ['HF_TOKEN']);B='smodusermc/offline-games';BASE='https://garbsoftball.com' | |
| existing={f.path:f for f in api.list_bucket_tree(B,recursive=True) if hasattr(f,'size')} | |
| allnotes=[] | |
| try: | |
| for ident,base,mode in [(33,'/games/Apotheon/Content.tar','count'),(33,'/games/Apotheon/Dialog.tar','count'),(59,'/games/bendy/Build/BATIM.data','sequential')]: | |
| mp=R/f'game-{ident}.json';m=json.loads(mp.read_text());files={x['path']:x for x in m['files']};note={'id':ident,'base':base,'mode':mode,'added':[],'source_terminal':None} | |
| count=int(requests.get(BASE+base+'.count',timeout=20).text.strip()) if mode=='count' else 100 | |
| assert 0<count<=100 | |
| for i in range(count): | |
| path=base+(f'{i:02d}' if mode=='count' else f'.{i+1:03d}');key='site'+path | |
| if key in files and key in existing:continue | |
| if key in existing: | |
| files[key]={'url':BASE+path,'path':key,'bytes':existing[key].size,'sha256':None,'content_type':'application/octet-stream','note':'Recovered prior interrupted batch object; size verified; checksum not yet re-verified'} | |
| continue | |
| with requests.get(BASE+path,stream=True,timeout=(8,35)) as response: | |
| if response.status_code!=200: | |
| note['source_terminal']={'path':path,'status':response.status_code} | |
| if mode=='count':note['incomplete']=True | |
| break | |
| local=TMP/'chunk';h=hashlib.sha256();size=0;head=b'' | |
| with local.open('wb') as f: | |
| for c in response.iter_content(1024*1024): | |
| if not head: | |
| head=c[:100] | |
| if head.lstrip().lower().startswith((b'<!doctype html',b'<html')):break | |
| size+=len(c) | |
| if size>80*1024**2:raise RuntimeError('chunk larger than safe 80 MiB limit') | |
| f.write(c);h.update(c) | |
| if not size: | |
| local.unlink(missing_ok=True);note['source_terminal']={'path':path,'status':'HTML fallback, not binary'};break | |
| entry={'url':BASE+path,'path':key,'bytes':size,'sha256':h.hexdigest(),'content_type':response.headers.get('Content-Type','application/octet-stream')} | |
| api.batch_bucket_files(B,add=[(local,key)]) | |
| rf=list(api.get_bucket_paths_info(B,[key]));assert len(rf)==1 and rf[0].size==size | |
| existing[key]=rf[0];local.unlink();files[key]=entry;note['added'].append(entry) | |
| m['files']=list(files.values());m['split_archive_repairs']=m.get('split_archive_repairs',[])+[{k:v for k,v in note.items() if k!='added'}] | |
| mp.write_text(json.dumps(m,indent=2));api.batch_bucket_files(B,add=[(mp,f'_reports/games/{ident:04d}.json')]) | |
| allnotes.append(note);print(ident,base,'added',len(note['added']),'bytes',sum(x['bytes'] for x in note['added']),'terminal',note['source_terminal'],flush=True) | |
| out=R/'split-archive-repairs.json';out.write_text(json.dumps(allnotes,indent=2));api.batch_bucket_files(B,add=[(out,'_verification/split-archive-repairs.json')]) | |
| finally:shutil.rmtree(TMP,ignore_errors=True) | |
Xet Storage Details
- Size:
- 3.11 kB
- Xet hash:
- 26d07cfa2beca13eee494828ca17ab055066b0cdaf318528a7d55e47802b4b49
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.