smodusermc/offline-games / _tools /collect_next.py
smodusermc's picture
download
raw
21.4 kB
import os,json,re,hashlib,time,shutil,collections,concurrent.futures,threading
from pathlib import Path
from urllib.parse import urljoin,urlparse,unquote,urlunparse
import requests
from bs4 import BeautifulSoup
from huggingface_hub import HfApi
BASE='https://garbsoftball.com';BUCKET='smodusermc/offline-games';T0=time.time()
DEADLINE=int(os.environ.get('CAPTURE_SECONDS','900'));TOTAL_CAP=int(os.environ.get('CAPTURE_TOTAL_GIB','16'))*1024**3;FILE_CAP=int(os.environ.get('CAPTURE_FILE_MIB','2048'))*1024**2
ROOT=Path('/home/user/.cache/game-transfer');ROOT.mkdir(exist_ok=True)
api=HfApi(token=os.environ['HF_TOKEN']);api.whoami()
resume_dir=Path('/tmp/game-survey');resume_dir.mkdir(exist_ok=True)
if os.environ.get('RESUME_CAPTURE','1')=='1':
api.download_bucket_files(BUCKET,[('_reports/inventory.json',resume_dir/'inventory.json')],raise_on_missing_files=True)
rows=json.load(open(resume_dir/'inventory.json'))
known={}
for f in api.list_bucket_tree(BUCKET,prefix='site',recursive=True):
if hasattr(f,'size'):
known[f.path]={'path':f.path,'bytes':f.size,'url':BASE+'/'+f.path.removeprefix('site/'),'sha256':None,'note':'Previously uploaded; see original game manifest for SHA-256'}
for prior in Path('/home/user/game-archive/reports').glob('game-*.json'):
for f in json.loads(prior.read_text()).get('files',[]):
if f['path'] in known and f['bytes']==known[f['path']]['bytes'] and f.get('sha256'):known[f['path']]=f
prior_files=len(known);prior_bytes=sum(x['bytes'] for x in known.values());total=0;uploaded=0
EXT=r'(?:js|mjs|css|json|wasm|data|pck|unityweb|unity3d|swf|png|jpe?g|webp|gif|svg|ico|bmp|ogg|mp3|wav|m4a|mp4|webm|ttf|woff2?|otf|fnt|xml|atlas|bin|dat|pak|zip|gz|br|mem|bundle|bytes|rpgmvp|rpgmvo|rpgmvm|nds|gba|gbc|gb|nes|n64|z64|v64|iso|cue|chd|rom|wad|pk4|jsdos|dcd|sdl|cfg|ags|vox|txt|csv|tsv|count|rpa|rpyc|unx|dll|exe|tar[0-9]*|webmanifest|manifest)'
PATH_RE=re.compile(r'^[^<>\n\r\t{}|*]{1,500}\.'+EXT+r'(?:\.(?:part)?[0-9]+)?(?:[?#][^\s]*)?$',re.I)
LITERAL=re.compile(r'''(["'`])([^"'`\n\r]{1,500}\.'''+EXT+r'''(?:\.(?:part)?[0-9]+)?(?:[?#][^"'`\n\r]*)?)\1''',re.I)
IGNORE_HOSTS=('googletagmanager.com','google-analytics.com','cloudflareinsights.com','fonts.googleapis.com','fonts.gstatic.com')
# Original-byte mirror: no HTML/JS rewrites. No off-site fetches, no proxy calls.
def normalized(u):
p=urlparse(u)
if p.scheme not in ('http','https') or p.hostname!='garbsoftball.com':return None
if '/games/geo/' in p.path or p.path.startswith(('/__proxy','/scramjet','/bare','/api/')):return None
path=unquote(p.path)
if '..' in Path(path).parts or '\x00' in path:return None
return BASE+p.path+('?' + p.query if p.query else '')
def dest(u):
p=unquote(urlparse(u).path).lstrip('/')
if not p or p.endswith('/'):p+='index.html'
return 'site/'+p
def extract(text,url,entry):
found=[];externals=set();optional=[]
ext=urlparse(url).path.lower();ishtml=ext.endswith(('.html','.htm','/'))
base=entry if ext.endswith(('.js','.mjs')) else url
if '/_framework/' in url:base=url
if '/deltarune/chapter' in url and ext.endswith(('.js','.mjs')):
base=re.match(r'(.*?/deltarune/chapter[0-9]+/)',url)[1]+'index.html'
if ishtml:
soup=BeautifulSoup(text,'html.parser')
b=soup.find('base',href=True)
if b:base=urljoin(url,b['href'])
for e in soup.select('script[src],link[href],img[src],audio[src],video[src],source[src],embed[src],object[data]'):
if e.name=='link' and not any(x in (e.get('rel') or []) for x in ['stylesheet','icon','shortcut','preload','manifest','apple-touch-icon']):continue
v=e.get('src') or e.get('href') or e.get('data');found.append((v,base,'explicit'))
for m in re.finditer(r'''url\(\s*["']?([^\s)'";]+)''',text):found.append((m[1],url if ext.endswith('.css') else base,'css'))
# Evaluate simple string-variable concatenations used by Unity and web-port loaders.
variables={m[1]:m[3] for m in re.finditer(r'''(?:var|let|const)\s+(\w+)\s*=\s*(["'])([^"'\n]{0,400})\2''',text)}
combined_positions=[]
for m in re.finditer(r'''\b(\w+)\s*\+\s*(["'])([^"'\n]{1,450})\2''',text):
if m[1] in variables:
v=variables[m[1]]+m[3]
if PATH_RE.match(v):found.append((v,base,'concatenated'));combined_positions.append((m.start(),m.end()))
# String literals with recognizable file extensions; data URIs and virtual-FS paths excluded below.
for m in LITERAL.finditer(text):
v=m[2].replace('\\/','/')
if not PATH_RE.match(v) or len(v)>500:continue
if any(a<=m.start()<=b for a,b in combined_positions):continue
if v.startswith(('/Resources/','/persistent/','/home/','/proc/','/dev/','/tmp/','C:', 'data:', 'blob:')):continue
if ext.endswith('.js') and v.startswith('.') and '/' not in v:continue
if v.endswith('.html'):continue
# Unity JSON file paths are relative to the manifest, JS fetch paths to the document.
found.append((v,base,'literal'))
# Proven lazy-manifest rule: Thirty Dollar Website constructs WAV/icon URLs from IDs.
if urlparse(url).path.endswith('/30dolar/sounds.json'):
try:
for sound in json.loads(text):
sid=sound['id']
if sid!='_pause':found.append(('sounds/'+sid+'.wav',url,'sound-catalog'))
if not sound.get('emoji') and re.search('[a-z0-9]',sid,re.I):icon='icons/'+sound.get('img',sid)+'.png'
else:icon='icons/'+format(ord((sound.get('emoji') or sid)[0]),'x')+'.svg'
found.append((icon,url,'sound-catalog-icon'))
except (ValueError,KeyError,TypeError):pass
# Construct exports store sound names and extensions separately in data.json.
if urlparse(url).path.endswith('/data.json'):
try:
def find_audio(node):
if isinstance(node,list):
if len(node)>=2 and isinstance(node[0],str) and isinstance(node[1],list):
for fmt in node[1]:
if isinstance(fmt,list) and len(fmt)>=3 and isinstance(fmt[0],str) and fmt[0].startswith('audio/') and isinstance(fmt[1],str) and fmt[1].startswith('.'):
found.append(('media/'+node[0]+fmt[1],url,'construct-audio-manifest'))
for child in node:find_audio(child)
elif isinstance(node,dict):
for child in node.values():find_audio(child)
find_audio(json.loads(text))
except (ValueError,TypeError,RecursionError):pass
# Cookie Clicker's jukebox is an explicit inventory of dynamically named sound effects.
if '/cookie-clicker/' in urlparse(url).path and urlparse(url).path.endswith('/main.js'):
match=re.search(r'Game\.jukebox\s*=\s*\{\s*sounds\s*:\s*\[(.*?)\]',text,re.S)
if match:
clean=re.sub(r'//[^\n]*','',match[1])
for sound in re.findall(r'"([^"\n]+)"',clean):found.append(('snd/'+sound+'.mp3',base,'cookie-jukebox-manifest'))
# Fireboy/Watergirl ports keep later maps in per-temple inventories.
if urlparse(url).path.endswith('/temple.json') and '/data/' in url:
try:
gamebase=url.split('/data/',1)[0]+'/'
for level in json.loads(text).get('levels',[]):
if level.get('filename'):found.append(('data/'+level['filename'],gamebase,'temple-level-manifest'))
except (ValueError,TypeError):pass
if urlparse(url).path.endswith('/game.json') and '/watergirl-' in url:
try:
config=json.loads(text)
for temple in ([config['temples']] if isinstance(config.get('temples'),str) else config.get('temples',[])):
found.append(('data/'+temple+'/temple.json',url,'temple-inventory'))
for asset in ([config['assets']] if isinstance(config.get('assets'),str) else config.get('assets',[])):
for ending in ['TempleAssets.json','TempleAssets.png']:found.append(('assets/atlasses/Temples/'+asset+'/'+ending,url,'temple-atlas'))
for prefix,suffix in [('TempleHall','.jpg'),('GameName','.png')]:found.append(('assets/images/'+prefix+asset[0].upper()+asset[1:]+suffix,url,'temple-images'))
except (ValueError,KeyError,TypeError):pass
if '/watergirl-' in url and 'assets/audio/' in text:
for a in re.finditer(r'\[((?:\s*"[^"\n]+"\s*,?){10,})\s*\]',text):
names=re.findall(r'"([^"\n]+)"',a[1])
if 'levelMusicFinish' in names and 'levelMusic' in names:
names+=re.findall(r'"(levelMusic[A-Za-z_]+)"',text)
for name in set(names):found.append(('assets/audio/'+name[0].upper()+name[1:]+'.mp3',url,'phaser-audio-name-list'))
# Older Construct 2 exports pair complete audio filenames with sizes.
if urlparse(url).path.endswith(('/data.js','/data.json')) and '"media/"' in text:
for name in re.findall(r'"([^"/]+\.(?:ogg|m4a|mp3|wav|webm))"',text):found.append(('media/'+name,url,'construct2-audio-list'))
# A local landing page can link to its actual game under play/.
if ishtml:
for a in soup.select('a[href]'):
href=a['href']
if href in ('play/','./play/','play/index.html','./play/index.html'):found.append((href,base,'local-play-entry'))
# ES module imports resolve relative to the module, not the document.
if ext.endswith(('.js','.mjs')):
for match in re.finditer(r'(?:from\s*|import\s*\(?)\s*["\'](\.[^"\']+\.m?js)["\']',text):found.append((match[1],url,'module-import'))
# This Construct export also plays music by name, outside its sound-file inventory.
if urlparse(url).path.endswith('/there-is-no-game/data.js'):
try:
def named_audio(node):
if isinstance(node,list):
if len(node)>=6 and node[0]==45 and isinstance(node[5],list):
params=node[5]
if len(params)>1 and isinstance(params[0],list) and len(params[0])==2 and params[0][0]==3 and isinstance(params[1],list) and len(params[1])==2 and params[1][0]==1 and isinstance(params[1][1],list) and len(params[1][1])==2 and params[1][1][0]==2 and isinstance(params[1][1][1],str):
for suffix in ['.ogg','.m4a']:found.append(('media/'+params[1][1][1].lower()+suffix,url,'construct-play-by-name'))
for child in node:named_audio(child)
elif isinstance(node,dict):
for child in node.values():named_audio(child)
named_audio(json.loads(text))
except (ValueError,TypeError,RecursionError):pass
# Loader-defined split binaries, including game.unx and Godot packages.
for m in re.finditer(r'getParts\(\s*["\']([^"\']+)["\']\s*,\s*(\d+)\s*,\s*(\d+)\s*\)',text):
start,end=int(m[2]),int(m[3])
if 0<=start<=end<=512:
for i in range(start,end+1):found.append((m[1]+'.part'+str(i),base,'declared-split-part'))
# Chapter launchers must not be archived as just a small menu.
if '/deltarune/' in url and ishtml and 'function loadChapter(' in text:
for chapter in sorted(set(re.findall(r'loadChapter\(([0-9]+)\)',text))):found.append(('chapter'+chapter+'/index.html',url,'chapter-entry'))
# AGS export explicitly declares counts and the directory for split game/audio archives.
if '/endacopia/' in url and ishtml:
split=re.search(r'(?:var|let|const)\s+splitFiles\s*=\s*(\{[^}]+\})',text)
if split:
for name,count in json.loads(split[1]).items():
if not isinstance(count,int) or not 0<count<=256:continue
for i in range(1,count+1):found.append(('Actual%20Game/'+name+'.part'+str(i).zfill(3),url,'ags-split-file'))
for name in re.findall(r'loadNormalFile\(\s*"([^"\n]+)"',text):found.append(('Actual%20Game/'+name,url,'ags-support-file'))
# Fallout's disc/OS/game image paths are relative to its explicit toolBase.
if '/fo2/' in url and ishtml:
for path in re.findall(r'["\'](/assets/[^"\']+\.(?:DCD|jsdos))["\']',text,re.I):found.append((path.lstrip('/'),url,'emulator-image'))
# Count sidecars used by split tar packages (e.g. the .NET ports).
if urlparse(url).path.endswith('.tar.count'):
try:
count=int(text.strip())
if 0<count<=256:
for i in range(count):found.append((url[:-6]+str(i).zfill(2),url,'counted-tar-chunk'))
except ValueError:pass
# Godot ENGINE_CONFIG executable often omits the binary extensions.
for m in re.finditer(r'''["']executable["']\s*:\s*["']([^"']+)["']''',text):
found.extend([(m[1]+'.wasm',base,'godot'),(m[1]+'.pck',base,'godot')])
# Construct 2/3 exports use data.json or data.js referenced by loaders; handled by literals.
# AMD/ESM imports resolve relative to the module, unlike fetch().
if ext.endswith(('.js','.mjs')):
for m in re.finditer(r'''(?:from\s*|import\s*\(|importScripts\s*\()\s*["']([^"']+)["']''',text):found.append((m[1],url,'module'))
result=[];seen=set()
for value,b,kind in found:
if not value or value.startswith(('data:','blob:','#','javascript:')):continue
absolute=urljoin(b,value);p=urlparse(absolute)
if p.hostname and p.hostname!='garbsoftball.com':
if not any(x in p.hostname for x in IGNORE_HOSTS):externals.add(absolute)
continue
u=normalized(absolute)
if u and kind=='local-play-entry' and u.endswith('/'):u+='index.html'
if not u:continue
if re.search(r'https?:',urlparse(u).path):continue
if not (kind in ('chapter-entry','local-play-entry') and u.endswith('.html')) and not re.search(r'\.'+EXT+r'(?:\.(?:part)?[0-9]+)?$',urlparse(u).path,re.I):continue
# Discard extension-only templates and invalid root filenames from minified loaders.
if re.match(r'^\.[a-z0-9]+$',Path(urlparse(u).path).name):continue
if u not in seen:seen.add(u);result.append((u,kind))
return result,sorted(externals)
sessions=threading.local()
def fetch(task):
u,kind=task
try:
if not hasattr(sessions,'s'):sessions.s=requests.Session()
with sessions.s.get(u,timeout=(10,40),stream=True) as r:
if r.status_code!=200:return {'url':u,'kind':kind,'error':'HTTP '+str(r.status_code)}
if urlparse(r.url).hostname!='garbsoftball.com':return {'url':u,'kind':kind,'error':'external redirect skipped'}
length=int(r.headers.get('Content-Length',0))
if length>FILE_CAP:return {'url':u,'kind':kind,'error':f'deferred: file exceeds {FILE_CAP//1048576} MiB batch-file limit','bytes':length}
ct=r.headers.get('Content-Type','');path=ROOT/hashlib.sha256(u.encode()).hexdigest();n=0;h=hashlib.sha256();head=b''
with path.open('wb') as f:
for c in r.iter_content(1024*1024):
n+=len(c)
if n>FILE_CAP:raise ValueError('file size limit')
if len(head)<256:head+=c[:256-len(head)]
f.write(c);h.update(c)
# Avoid accepting the site's HTML 404/fallback page as a binary or script.
if head.strip().startswith(b'PS 404 Not Found!'):
path.unlink();return {'url':u,'kind':kind,'error':'Source returned a plain-text 404 body with HTTP 200'}
is_html=bool(re.match(br'\s*(?:<!doctype html|<html)',head,re.I))
if is_html and not urlparse(u).path.lower().endswith(('.html','.htm','/')):
path.unlink();return {'url':u,'kind':kind,'error':'unexpected HTML response'}
return {'url':u,'kind':kind,'local':str(path),'path':dest(u),'bytes':n,'sha256':h.hexdigest(),'content_type':ct}
except Exception as e:
if 'path' in locals():path.unlink(missing_ok=True)
return {'url':u,'kind':kind,'error':str(e)[:180]}
readme='''# Locally hosted game-file collection\n\nSource catalog: https://garbsoftball.com/g\n\nThis is a Hugging Face **public storage bucket**, not a dataset repository.\nFiles under `site/` preserve their original server-relative paths and bytes.\nGeometry Dash, external frame embeds, proxied entries, and identified externally hosted game launchers are excluded.\n\n**This is a static-dependency capture, not a verified complete/offline game release.**\nSee `_reports/inventory.json` for every catalog entry and each `_reports/games/` manifest for captured files, failed requests, and external dependencies. Dynamically generated filenames, streaming assets, runtime-only downloads, servers, and online features may be missing. No game is marked gameplay-tested.\n\nDo not assume that downloading only an index.html retrieves a game. Download its recorded assets as well. Original scripts may still reference external services or absolute paths and may require hosting from the `site/` root; these original-file captures are NOT self-contained HTML conversions or a promise of file:// compatibility. Bucket storage is not a configured game web server.\n\nUploads are processed in small batches. Local copies are deleted after the uploaded objects are listed and their sizes verified. SHA-256 hashes are recorded for source-byte verification. A source 404/error is recorded, not silently treated as a game asset.\n\nOriginal games/assets retain their owners' rights. Source-site availability is not proof of redistribution permission. This bucket was made public at the account owner's explicit request. No new license is granted.\n'''
api.batch_bucket_files(BUCKET,add=[(readme.encode(),'_reports/collector-method.md'),(json.dumps(rows,indent=2).encode(),'_reports/inventory.json')])
def flush(batch):
global uploaded
if not batch:return
api.batch_bucket_files(BUCKET,add=[(x['local'],x['path']) for x in batch])
remote={f.path:f for f in api.get_bucket_paths_info(BUCKET,[x['path'] for x in batch])}
for x in batch:
assert x['path'] in remote and remote[x['path']].size==x['bytes'],f"Upload verification failed for {x['path']}"
known[x['path']]={k:v for k,v in x.items() if k not in ('local','kind')};uploaded+=x['bytes']
Path(x['local']).unlink(missing_ok=True)
batch.clear()
pending=[r for r in rows if r['status'] in ('local_candidate','pending_not_captured')]
selected={int(x) for x in os.environ.get('SELECT_GAME_IDS','').split(',') if x.strip()}
if selected:pending=[r for r in rows if r['id'] in selected]
# Work in catalog order, completing small file batches rather than accumulating the site locally.
try:
with concurrent.futures.ThreadPoolExecutor(max_workers=3) as pool:
for gi,row in enumerate(pending):
if time.time()-T0>DEADLINE or uploaded>TOTAL_CAP:break
manifest={'name':row['name'],'entry':row['entry'],'method':'static dependency traversal; original bytes','gameplay_tested':False,'files':[],'failures':[],'external_dependencies':list(row.get('external',[])),'limits':[]}
entry=normalized(row['entry']);queue=collections.deque([(entry,'entry')]);seen=set();batch=[];batchbytes=0;count=0;gametotal=0
while queue and count<1200:
if time.time()-T0>DEADLINE or uploaded+batchbytes>TOTAL_CAP:
manifest['limits'].append('run time or total transfer safety limit reached');break
tasks=[]
while queue and len(tasks)<3:
u,kind=queue.popleft()
if u in seen:continue
seen.add(u)
if dest(u) in known and kind!='entry' and not re.search(r'\.(?:html?|js|mjs|css|json|xml|count|manifest|webmanifest)$',urlparse(u).path,re.I):
x=known[dest(u)];manifest['files'].append(x);continue
tasks.append((u,kind))
if not tasks:continue
for x in pool.map(fetch,tasks):
count+=1
if 'error' in x:manifest['failures'].append(x);continue
gametotal+=x['bytes'];p=Path(x['local']);text=None
if x['bytes']<24*1024*1024 and (re.search(r'\.(?:html?|js|mjs|css|json|xml|count|manifest|webmanifest)$',urlparse(x['url']).path,re.I) or x['kind']=='entry'):
text=p.read_text(errors='replace')
if text is not None:
refs,external=extract(text,x['url'],entry);queue.extend(refs);manifest['external_dependencies'].extend(external)
record={k:v for k,v in x.items() if k not in ('local','kind')};manifest['files'].append(record);batch.append(x);batchbytes+=x['bytes']
if len(batch)>=32 or batchbytes>=64*1024*1024:flush(batch);batchbytes=0
if gametotal>int(os.environ.get('CAPTURE_GAME_GIB','6'))*1024**3:
manifest['limits'].append('per-game capture safety limit reached');break
if queue:manifest['limits'].append(f'{len(queue)} discovered URLs remain unprocessed')
flush(batch)
manifest['external_dependencies']=sorted(set(manifest['external_dependencies']))
row['captured_files']=len(manifest['files']);row['captured_bytes']=sum(x['bytes'] for x in manifest['files']);row['failed_requests']=len(manifest['failures']);row['external_dependency_count']=len(manifest['external_dependencies'])
row['status']='partial_static_capture' if manifest['failures'] or manifest['limits'] or manifest['external_dependencies'] else 'static_capture_unverified'
row['manifest']=f"_reports/games/{row['id']:04d}.json"
api.batch_bucket_files(BUCKET,add=[(json.dumps(manifest,indent=2).encode(),row['manifest'])])
# Entry survey files aren't retained once their game has been processed.
(Path('/tmp/game-survey')/f"{row['id']}.html").unlink(missing_ok=True)
print(f"{gi+1}/{len(pending)} {row['name']}: {row['captured_files']} files, {row['captured_bytes']/1e6:.1f} MB; {row['status']}; uploaded total {uploaded/1e9:.2f} GB",flush=True)
if gi%5==0:api.batch_bucket_files(BUCKET,add=[(json.dumps(rows,indent=2).encode(),'_reports/inventory.json')])
except Exception as e:
print('STOPPED:',type(e).__name__,str(e)[:350],flush=True)
finally:
for r in rows:
if r['status']=='local_candidate':r['status']='pending_not_captured'
summary={'catalog_entries':len(rows),'status_counts':dict(collections.Counter(r['status'] for r in rows)),'unique_verified_uploaded_files':len(known),'unique_verified_uploaded_bytes':prior_bytes+uploaded,'uploaded_bytes_this_run':uploaded,'seconds':round(time.time()-T0),'note':'This collection pass does not run browser tests. See _verification for separate test results. Pending entries were not captured.'}
api.batch_bucket_files(BUCKET,add=[(json.dumps(rows,indent=2).encode(),'_reports/inventory.json'),(json.dumps(summary,indent=2).encode(),'_reports/summary.json')])
Path('/home/user/game-archive/reports/collection-latest-summary.json').write_text(json.dumps(summary,indent=2));Path('/home/user/game-archive/reports/inventory.json').write_text(json.dumps(rows,indent=2))
print('SUMMARY',json.dumps(summary),flush=True)
shutil.rmtree(ROOT,ignore_errors=True)

Xet Storage Details

Size:
21.4 kB
·
Xet hash:
6131bbe2e55a11ba8600862d689ad55435867c03c1743051f58e40ffb48fa0d4

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.