smodusermc/offline-games / _tools /previous-collector.py
smodusermc's picture
download
raw
13.2 kB
import os,json,re,hashlib,time,shutil,collections,concurrent.futures,threading
from pathlib import Path
from urllib.parse import urljoin,urlparse,unquote,urlunparse
import requests
from bs4 import BeautifulSoup
from huggingface_hub import HfApi
BASE='https://garbsoftball.com';BUCKET='smodusermc/offline-games';T0=time.time()
DEADLINE=int(os.environ.get('CAPTURE_SECONDS','900'));TOTAL_CAP=16*1024**3;FILE_CAP=768*1024**2
ROOT=Path('/tmp/game-transfer');ROOT.mkdir(exist_ok=True)
api=HfApi(token=os.environ['HF_TOKEN']);api.whoami()
resume_dir=Path('/tmp/game-survey');resume_dir.mkdir(exist_ok=True)
if os.environ.get('RESUME_CAPTURE','1')=='1':
api.download_bucket_files(BUCKET,[('_reports/inventory.json',resume_dir/'inventory.json')],raise_on_missing_files=True)
rows=json.load(open(resume_dir/'inventory.json'))
known={}
for f in api.list_bucket_tree(BUCKET,prefix='site',recursive=True):
if hasattr(f,'size'):
known[f.path]={'path':f.path,'bytes':f.size,'url':BASE+'/'+f.path.removeprefix('site/'),'sha256':None,'note':'Previously uploaded; see original game manifest for SHA-256'}
prior_files=len(known);prior_bytes=sum(x['bytes'] for x in known.values());total=0;uploaded=0
EXT=r'(?:js|mjs|css|json|wasm|data|pck|unityweb|unity3d|swf|png|jpe?g|webp|gif|svg|ico|bmp|ogg|mp3|wav|m4a|mp4|webm|ttf|woff2?|otf|fnt|xml|atlas|bin|dat|pak|zip|gz|br|mem|bundle|bytes|rpgmvp|rpgmvo|rpgmvm|nds|gba|gbc|gb|nes|n64|z64|v64|iso|cue|chd|rom|wad|webmanifest|manifest)'
PATH_RE=re.compile(r'^[^<>\n\r\t{}|*]{1,500}\.'+EXT+r'(?:\.[0-9]+)?(?:[?#][^\s]*)?$',re.I)
LITERAL=re.compile(r'''(["'`])([^"'`\n\r]{1,550})\1''')
IGNORE_HOSTS=('googletagmanager.com','google-analytics.com','cloudflareinsights.com','fonts.googleapis.com','fonts.gstatic.com')
# Original-byte mirror: no HTML/JS rewrites. No off-site fetches, no proxy calls.
def normalized(u):
p=urlparse(u)
if p.scheme not in ('http','https') or p.hostname!='garbsoftball.com':return None
if '/games/geo/' in p.path or p.path.startswith(('/__proxy','/scramjet','/bare','/api/')):return None
path=unquote(p.path)
if '..' in Path(path).parts or '\x00' in path:return None
return BASE+p.path+('?' + p.query if p.query else '')
def dest(u):
p=unquote(urlparse(u).path).lstrip('/')
if not p or p.endswith('/'):p+='index.html'
return 'site/'+p
def extract(text,url,entry):
found=[];externals=set();optional=[]
ext=urlparse(url).path.lower();ishtml=ext.endswith(('.html','.htm','/'))
base=entry if ext.endswith(('.js','.mjs')) else url
if ishtml:
soup=BeautifulSoup(text,'html.parser')
b=soup.find('base',href=True)
if b:base=urljoin(url,b['href'])
for e in soup.select('script[src],link[href],img[src],audio[src],video[src],source[src],embed[src],object[data]'):
if e.name=='link' and not any(x in (e.get('rel') or []) for x in ['stylesheet','icon','shortcut','preload','manifest','apple-touch-icon']):continue
v=e.get('src') or e.get('href') or e.get('data');found.append((v,base,'explicit'))
for m in re.finditer(r'''url\(\s*["']?([^\s)'";]+)''',text):found.append((m[1],url if ext.endswith('.css') else base,'css'))
# Evaluate simple string-variable concatenations used by Unity and web-port loaders.
variables={m[1]:m[3] for m in re.finditer(r'''(?:var|let|const)\s+(\w+)\s*=\s*(["'])([^"'\n]{0,400})\2''',text)}
combined_positions=[]
for m in re.finditer(r'''\b(\w+)\s*\+\s*(["'])([^"'\n]{1,450})\2''',text):
if m[1] in variables:
v=variables[m[1]]+m[3]
if PATH_RE.match(v):found.append((v,base,'concatenated'));combined_positions.append((m.start(),m.end()))
# String literals with recognizable file extensions; data URIs and virtual-FS paths excluded below.
for m in LITERAL.finditer(text):
v=m[2].replace('\\/','/')
if not PATH_RE.match(v) or len(v)>500:continue
if any(a<=m.start()<=b for a,b in combined_positions):continue
if v.startswith(('/Resources/','/persistent/','/home/','/proc/','/dev/','/tmp/','C:', 'data:', 'blob:')):continue
if ext.endswith('.js') and v.startswith('.') and '/' not in v:continue
if v.endswith('.html'):continue
# Unity JSON file paths are relative to the manifest, JS fetch paths to the document.
found.append((v,base,'literal'))
# Godot ENGINE_CONFIG executable often omits the binary extensions.
for m in re.finditer(r'''["']executable["']\s*:\s*["']([^"']+)["']''',text):
found.extend([(m[1]+'.wasm',base,'godot'),(m[1]+'.pck',base,'godot')])
# Construct 2/3 exports use data.json or data.js referenced by loaders; handled by literals.
# AMD/ESM imports resolve relative to the module, unlike fetch().
if ext.endswith(('.js','.mjs')):
for m in re.finditer(r'''(?:from\s*|import\s*\(|importScripts\s*\()\s*["']([^"']+)["']''',text):found.append((m[1],url,'module'))
result=[];seen=set()
for value,b,kind in found:
if not value or value.startswith(('data:','blob:','#','javascript:')):continue
absolute=urljoin(b,value);p=urlparse(absolute)
if p.hostname and p.hostname!='garbsoftball.com':
if not any(x in p.hostname for x in IGNORE_HOSTS):externals.add(absolute)
continue
u=normalized(absolute)
if not u or not re.search(r'\.'+EXT+r'(?:\.[0-9]+)?$',urlparse(u).path,re.I):continue
# Discard extension-only templates and invalid root filenames from minified loaders.
if re.match(r'^\.[a-z0-9]+$',Path(urlparse(u).path).name):continue
if u not in seen:seen.add(u);result.append((u,kind))
return result,sorted(externals)
sessions=threading.local()
def fetch(task):
u,kind=task
try:
if not hasattr(sessions,'s'):sessions.s=requests.Session()
with sessions.s.get(u,timeout=(10,40),stream=True) as r:
if r.status_code!=200:return {'url':u,'kind':kind,'error':'HTTP '+str(r.status_code)}
if urlparse(r.url).hostname!='garbsoftball.com':return {'url':u,'kind':kind,'error':'external redirect skipped'}
length=int(r.headers.get('Content-Length',0))
if length>FILE_CAP:return {'url':u,'kind':kind,'error':'deferred: file exceeds 768 MiB batch-file limit','bytes':length}
ct=r.headers.get('Content-Type','');path=ROOT/hashlib.sha256(u.encode()).hexdigest();n=0;h=hashlib.sha256();head=b''
with path.open('wb') as f:
for c in r.iter_content(1024*1024):
n+=len(c)
if n>FILE_CAP:raise ValueError('file size limit')
if len(head)<256:head+=c[:256-len(head)]
f.write(c);h.update(c)
# Avoid accepting the site's HTML 404/fallback page as a binary or script.
is_html=bool(re.match(br'\s*(?:<!doctype html|<html)',head,re.I))
if is_html and not urlparse(u).path.lower().endswith(('.html','.htm','/')):
path.unlink();return {'url':u,'kind':kind,'error':'unexpected HTML response'}
return {'url':u,'kind':kind,'local':str(path),'path':dest(u),'bytes':n,'sha256':h.hexdigest(),'content_type':ct}
except Exception as e:return {'url':u,'kind':kind,'error':str(e)[:180]}
readme='''# Locally hosted game-file collection\n\nSource catalog: https://garbsoftball.com/g\n\nThis is a Hugging Face **public storage bucket**, not a dataset repository.\nFiles under `site/` preserve their original server-relative paths and bytes.\nGeometry Dash, external frame embeds, proxied entries, and identified externally hosted game launchers are excluded.\n\n**This is a static-dependency capture, not a verified complete/offline game release.**\nSee `_reports/inventory.json` for every catalog entry and each `_reports/games/` manifest for captured files, failed requests, and external dependencies. Dynamically generated filenames, streaming assets, runtime-only downloads, servers, and online features may be missing. No game is marked gameplay-tested.\n\nDo not assume that downloading only an index.html retrieves a game. Download its recorded assets as well. Original scripts may still reference external services or absolute paths and may require hosting from the `site/` root; these original-file captures are NOT self-contained HTML conversions or a promise of file:// compatibility. Bucket storage is not a configured game web server.\n\nUploads are processed in small batches. Local copies are deleted after the uploaded objects are listed and their sizes verified. SHA-256 hashes are recorded for source-byte verification. A source 404/error is recorded, not silently treated as a game asset.\n\nOriginal games/assets retain their owners' rights. Source-site availability is not proof of redistribution permission. This bucket was made public at the account owner's explicit request. No new license is granted.\n'''
api.batch_bucket_files(BUCKET,add=[(readme.encode(),'_reports/collector-method.md'),(json.dumps(rows,indent=2).encode(),'_reports/inventory.json')])
def flush(batch):
global uploaded
if not batch:return
api.batch_bucket_files(BUCKET,add=[(x['local'],x['path']) for x in batch])
remote={f.path:f for f in api.get_bucket_paths_info(BUCKET,[x['path'] for x in batch])}
for x in batch:
assert x['path'] in remote and remote[x['path']].size==x['bytes'],f"Upload verification failed for {x['path']}"
known[x['path']]={k:v for k,v in x.items() if k not in ('local','kind')};uploaded+=x['bytes']
Path(x['local']).unlink(missing_ok=True)
batch.clear()
pending=[r for r in rows if r['status'] in ('local_candidate','pending_not_captured')]
# Work in catalog order, completing small file batches rather than accumulating the site locally.
try:
with concurrent.futures.ThreadPoolExecutor(max_workers=3) as pool:
for gi,row in enumerate(pending):
if time.time()-T0>DEADLINE or uploaded>TOTAL_CAP:break
manifest={'name':row['name'],'entry':row['entry'],'method':'static dependency traversal; original bytes','gameplay_tested':False,'files':[],'failures':[],'external_dependencies':list(row.get('external',[])),'limits':[]}
entry=normalized(row['entry']);queue=collections.deque([(entry,'entry')]);seen=set();batch=[];batchbytes=0;count=0;gametotal=0
while queue and count<1200:
if time.time()-T0>DEADLINE or uploaded+batchbytes>TOTAL_CAP:
manifest['limits'].append('run time or total transfer safety limit reached');break
tasks=[]
while queue and len(tasks)<3:
u,kind=queue.popleft()
if u in seen:continue
seen.add(u)
if dest(u) in known:
x=known[dest(u)];manifest['files'].append(x);continue
tasks.append((u,kind))
if not tasks:continue
for x in pool.map(fetch,tasks):
count+=1
if 'error' in x:manifest['failures'].append(x);continue
gametotal+=x['bytes'];p=Path(x['local']);text=None
if x['bytes']<24*1024*1024 and (re.search(r'\.(?:html?|js|mjs|css|json|xml|manifest|webmanifest)$',urlparse(x['url']).path,re.I) or x['kind']=='entry'):
text=p.read_text(errors='replace')
if text is not None:
refs,external=extract(text,x['url'],entry);queue.extend(refs);manifest['external_dependencies'].extend(external)
record={k:v for k,v in x.items() if k not in ('local','kind')};manifest['files'].append(record);batch.append(x);batchbytes+=x['bytes']
if len(batch)>=32 or batchbytes>=64*1024*1024:flush(batch);batchbytes=0
if gametotal>2*1024**3:
manifest['limits'].append('per-game 2 GiB capture safety limit reached');break
if queue:manifest['limits'].append(f'{len(queue)} discovered URLs remain unprocessed')
flush(batch)
manifest['external_dependencies']=sorted(set(manifest['external_dependencies']))
row['captured_files']=len(manifest['files']);row['captured_bytes']=sum(x['bytes'] for x in manifest['files']);row['failed_requests']=len(manifest['failures']);row['external_dependency_count']=len(manifest['external_dependencies'])
row['status']='partial_static_capture' if manifest['failures'] or manifest['limits'] or manifest['external_dependencies'] else 'static_capture_unverified'
row['manifest']=f"_reports/games/{row['id']:04d}.json"
api.batch_bucket_files(BUCKET,add=[(json.dumps(manifest,indent=2).encode(),row['manifest'])])
# Entry survey files aren't retained once their game has been processed.
(Path('/tmp/game-survey')/f"{row['id']}.html").unlink(missing_ok=True)
print(f"{gi+1}/{len(pending)} {row['name']}: {row['captured_files']} files, {row['captured_bytes']/1e6:.1f} MB; {row['status']}; uploaded total {uploaded/1e9:.2f} GB",flush=True)
if gi%5==0:api.batch_bucket_files(BUCKET,add=[(json.dumps(rows,indent=2).encode(),'_reports/inventory.json')])
except Exception as e:
print('STOPPED:',type(e).__name__,str(e)[:350],flush=True)
finally:
for r in rows:
if r['status']=='local_candidate':r['status']='pending_not_captured'
summary={'catalog_entries':len(rows),'status_counts':dict(collections.Counter(r['status'] for r in rows)),'unique_verified_uploaded_files':len(known),'unique_verified_uploaded_bytes':prior_bytes+uploaded,'uploaded_bytes_this_run':uploaded,'seconds':round(time.time()-T0),'note':'This collection pass does not run browser tests. See _verification for separate test results. Pending entries were not captured.'}
api.batch_bucket_files(BUCKET,add=[(json.dumps(rows,indent=2).encode(),'_reports/inventory.json'),(json.dumps(summary,indent=2).encode(),'_reports/summary.json')])
Path('/tmp/game-summary.json').write_text(json.dumps(summary,indent=2))
print('SUMMARY',json.dumps(summary),flush=True)
shutil.rmtree(ROOT,ignore_errors=True)

Xet Storage Details

Size:
13.2 kB
·
Xet hash:
a0e7dacd701605c92673bf1942b2934636c1ab496878a2697f830b61e0762d75

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.