Buckets:
| """Recover missing chapter audio ONLY when an existing same-title file matches the loader's declared MD5. | |
| Candidate bytes are also SHA256 checked. Cross-chapter copies retain exact provenance; no guessing/transcoding. | |
| """ | |
| import os,json,re,hashlib,shutil | |
| from pathlib import Path | |
| from huggingface_hub import HfApi | |
| R=Path('/home/user/game-archive/reports');B='smodusermc/offline-games';api=HfApi(token=os.environ['HF_TOKEN']);mp=R/'game-131.json';m=json.loads(mp.read_text());records={f['path']:f for f in m['files']};stage=Path('/home/user/.cache/deltarune-declared');stage.mkdir(parents=True,exist_ok=True);declared={};conflicts=[];report={'method':'Match source-loader manifestFiles / manifestFilesMD5 pairs to already archived same-title audio. MD5 is the loader declaration; SHA256 separately verifies candidate and destination bytes.','restored':[],'unresolved':[],'declaration_conflicts':conflicts} | |
| texts=[] | |
| for k in records: | |
| if '/deltarune/chapter' in k and k.endswith(('index.html','runner.js')):texts.append(k) | |
| api.download_bucket_files(B,[(k,stage/k) for k in texts],raise_on_missing_files=True) | |
| for key in texts: | |
| text=(stage/key).read_text(errors='replace');arrays=[] | |
| for fn in ['manifestFiles','manifestFilesMD5']: | |
| match=re.search(r'function\s+'+fn+r'\s*\(\s*\)\s*\{\s*return\s*(\[.*?\])',text,re.S) | |
| try:arrays.append(json.loads(match[1]) if match else None) | |
| except ValueError:arrays.append(None) | |
| if not all(isinstance(a,list) for a in arrays) or len(arrays[0])!=len(arrays[1]):continue | |
| base=key.rsplit('/',1)[0]+'/' | |
| for name,md5 in zip(*arrays): | |
| if not isinstance(name,str) or not re.search(r'\.(ogg|mp3|wav)$',name,re.I) or not re.fullmatch(r'[a-fA-F0-9]{32}',md5):continue | |
| dst=base+name | |
| if dst in declared and declared[dst]!=md5.lower():conflicts.append(dst) | |
| declared[dst]=md5.lower() | |
| missing={k:h for k,h in declared.items() if k not in records and k not in conflicts};names={Path(k).name.casefold() for k in missing};candidates=[f for k,f in records.items() if Path(k).name.casefold() in names and re.search(r'\.(ogg|mp3|wav)$',k,re.I)];cache={};source_hashes={};existing_media=json.loads((R/'heavy-batch-media.json').read_text());decoded={f['stored_sha256'] for x in existing_media['results'] for f in x['files'] if f.get('kind')=='audio' and f.get('decoded')} | |
| for start in range(0,len(candidates),16): | |
| batch=candidates[start:start+16];api.download_bucket_files(B,[(f['path'],stage/f['path']) for f in batch],raise_on_missing_files=True) | |
| for f in batch: | |
| p=stage/f['path'] | |
| with p.open('rb') as stream:md5=hashlib.file_digest(stream,'md5').hexdigest() | |
| with p.open('rb') as stream:sha=hashlib.file_digest(stream,'sha256').hexdigest() | |
| if f.get('sha256'):assert sha==f['sha256'],'Candidate SHA mismatch' | |
| cache.setdefault((p.name.casefold(),md5),f);source_hashes[f['path']]=sha;p.unlink() | |
| chosen=[] | |
| for dst,md5 in missing.items(): | |
| f=cache.get((Path(dst).name.casefold(),md5)) | |
| if f:chosen.append((dst,md5,f)) | |
| else:report['unresolved'].append({'path':dst,'declared_md5':md5,'reason':'No archived same-title candidate with matching declared checksum'}) | |
| source_info={f.path:f for f in api.get_bucket_paths_info(B,sorted({f['path'] for _,_,f in chosen}))} if chosen else {} | |
| for start in range(0,len(chosen),24): | |
| batch=chosen[start:start+24];api.batch_bucket_files(B,copy=[('bucket',B,source_info[f['path']].xet_hash,dst) for dst,md5,f in batch]);api.download_bucket_files(B,[(dst,stage/dst) for dst,_,_ in batch],raise_on_missing_files=True) | |
| for dst,md5,f in batch: | |
| p=stage/dst | |
| with p.open('rb') as stream:digest=hashlib.file_digest(stream,'sha256').hexdigest() | |
| assert digest==source_hashes[f['path']];assert p.stat().st_size==f['bytes'];rec={'path':dst,'url':'https://garbsoftball.com/'+dst.removeprefix('site/'),'bytes':f['bytes'],'sha256':digest,'restored_from_bucket_path':f['path'],'declared_source_md5_match':md5,'note':'Direct target unavailable at capture; exact loader-declared checksum matched an archived file from another chapter of the same title.'};records[dst]=rec;report['restored'].append({**rec,'readback_sha256_match':True,'identical_bytes_audio_decoded':digest in decoded});p.unlink() | |
| report['declared_audio_paths']=len(declared);report['previously_missing_declared_paths']=len(missing);report['restored_count']=len(report['restored']);report['still_unresolved_count']=len(report['unresolved']);m['files']=list(records.values());mp.write_text(json.dumps(m,indent=2));q=R/'deltarune-declared-audio-recovery.json';q.write_text(json.dumps(report,indent=2));api.batch_bucket_files(B,add=[(mp,'_reports/games/0131.json'),(q,'_verification/deltarune-declared-audio-recovery.json')]) | |
| p=R/'heavy-batch-validation.json';state=json.loads(p.read_text());x=next(x for x in state['results'] if x['id']==131);x['declared_audio_manifest_recovery']='_verification/deltarune-declared-audio-recovery.json';x['declared_audio_manifest_unresolved']=report['unresolved'];x['declared_audio_manifest_restored']=len(report['restored']);p.write_text(json.dumps(state,indent=2));shutil.rmtree(stage,ignore_errors=True);print('DECLARED AUDIO',report['declared_audio_paths'],'restored',report['restored_count'],'unresolved',report['still_unresolved_count'],flush=True) | |
Xet Storage Details
- Size:
- 5.21 kB
- Xet hash:
- 28114d6efa0c49d4f096890294a69c4ef89cc024d96abad0e88a998aa601c2a0
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.