"""Precomputed example presets shown in the app so a visitor can try the tool WITHOUT signing in and WITHOUT running any real HF job. Each example's output WARC(s) were produced once by ``scripts/build_examples.py`` and uploaded to a public HF bucket, so the "Download" link is a real, working file URL. The app streams *simulated* (but realistic) job logs and renders the precomputed estimate and result — no ``JobRunner`` call, no OAuth token needed. The record counts / byte sizes / filenames below come from the actual build; update them whenever the datasets are rebuilt. `n_records` for the capped examples equals `max_records`. """ from __future__ import annotations from dataclasses import dataclass, field from typing import List from . import config from .estimate import Estimate, format_estimate from .jobs import bucket_file_url, bucket_web_url # Public bucket that holds the precomputed example outputs (must be public-readable). EXAMPLES_BUCKET = "malteos/cc-repackage-examples" @dataclass(frozen=True) class Example: key: str title: str description: str # Form values the button fills in (mirrors the live filter fields). sql_where: str = "" domains: List[str] = field(default_factory=list) hostnames: List[str] = field(default_factory=list) languages: List[str] = field(default_factory=list) crawls: List[str] = field(default_factory=list) name: str = "repackaged" max_records: int = 0 # Precomputed results (from the build) — n_records/total_bytes drive the estimate. n_records: int = 0 total_bytes: int = 0 # Hosting on the public examples bucket. bucket: str = EXAMPLES_BUCKET path: str = "" files: List[str] = field(default_factory=list) # e.g. ["commoncrawl-org-001.warc.gz"] # Representative count for the "N parquet file(s) to scan" log line (from the build). scan_files: int = 300 def download_urls(self) -> List[tuple]: """(filename, resolve-url) for each produced WARC — real, direct downloads.""" return [(f, bucket_file_url(self.bucket, self.path, f)) for f in self.files] def folder_url(self) -> str: return bucket_web_url(self.bucket, self.path) # The "from build" values below were produced by scripts/build_examples.py and uploaded to # EXAMPLES_BUCKET; re-run that script and refresh them whenever the datasets are rebuilt. _CRAWL = "CC-MAIN-2026-21" EXAMPLES: List[Example] = [ Example( key="commoncrawl-org", title="All commoncrawl.org pages", description="Every capture of commoncrawl.org in one crawl — uses just the " "Registered domains field, no record cap.", domains=["commoncrawl.org"], crawls=[_CRAWL], name="commoncrawl-org", max_records=0, n_records=307, # from build total_bytes=2_494_825, # from build path="commoncrawl-org", files=["commoncrawl-org-001.warc.gz"], # from build scan_files=300, # from build ), Example( key="wikipedia-fr", title="10 French Wikipedia pages", description="French-language pages under wikipedia.org — uses the Registered domains " "+ Content languages fields with a Max records cap (no SQL).", domains=["wikipedia.org"], languages=["fra"], crawls=[_CRAWL], name="wikipedia-fr", max_records=10, n_records=10, total_bytes=334_225, # from build path="wikipedia-fr", files=["wikipedia-fr-001.warc.gz"], # from build scan_files=300, # from build ), Example( key="pdfs", title="10 PDF documents from one crawl", description="The first 10 records the index detected as application/pdf in one crawl.", sql_where="content_mime_detected = 'application/pdf'", crawls=[_CRAWL], name="pdfs", max_records=10, n_records=10, total_bytes=19_857_520, # from build path="pdfs", files=["pdfs-001.warc.gz"], # from build scan_files=8, # from build (capped via --max-files) ), Example( key="homepages", title="500 homepages from one crawl", description="The first 500 records whose URL path is '/' (site homepages) in one crawl.", sql_where="url_path = '/'", crawls=[_CRAWL], name="homepages", max_records=500, n_records=500, total_bytes=12_263_615, # from build path="homepages", files=["homepages-001.warc.gz"], # from build scan_files=4, # from build (capped via --max-files) ), ] EXAMPLES_BY_KEY = {ex.key: ex for ex in EXAMPLES} # --- Human-readable helpers -------------------------------------------------- def _fmt_bytes(n: int) -> str: step = 1024.0 val = float(n) for unit in ("B", "KiB", "MiB", "GiB", "TiB"): if val < step or unit == "TiB": return f"{val:.0f} {unit}" if unit == "B" else f"{val:.1f} {unit}" val /= step return f"{n} B" # --- Rendered markdown (reuses the real estimate/link builders) -------------- def estimate_markdown(ex: Example) -> str: """The same cost-estimate block a real estimate job would produce.""" return format_estimate(Estimate(ex.n_records, ex.total_bytes, config.DEFAULT_FLAVOR)) def result_markdown(ex: Example) -> str: """The 'repackaging complete' panel — with REAL download links to the hosted WARC(s).""" downloads = "\n".join( f"- **[⬇️ Download `{fname}`]({url})**" for fname, url in ex.download_urls() ) return ( "### ✅ Repackaging complete\n" "🧪 _This is a **precomputed example** — no HF Job ran and no sign-in was needed. " "Sign in above to run your own repackaging._\n\n" f"**{ex.n_records:,} records · {_fmt_bytes(ex.total_bytes)}** in the public " "examples bucket:\n\n" f"{downloads}\n\n" f"**[📂 Browse the example folder]({ex.folder_url()})**" ) # --- Simulated (but realistic) job logs, streamed line-by-line --------------- def estimate_log_lines(ex: Example) -> List[str]: """Mirror src/index_query.py's real estimate-job output (SQL vs structured filter).""" limit_note = f" limit={ex.max_records}" if ex.max_records > 0 else "" if ex.sql_where: filter_line = f"[index] mode=cdn crawls=['{ex.crawls[0]}'] sql_where={ex.sql_where!r}{limit_note}" else: filter_line = ( f"[index] mode=cdn crawls=['{ex.crawls[0]}'] hostnames={ex.hostnames} " f"domains={ex.domains} languages={ex.languages}{limit_note}" ) lines = [ "+ pip install --quiet duckdb huggingface_hub", "Successfully installed duckdb huggingface_hub", filter_line, "[index] enumerating index parquet files via the bucket API…", f"[index] {ex.scan_files} parquet file(s) to scan", ] if ex.sql_where: lines.append("[index] validating SQL filter…") lines += [ "[index] running query + writing range-jobs CSV…", f"ESTIMATE n_records={ex.n_records} total_bytes={ex.total_bytes}", ] return lines def fetch_log_lines(ex: Example) -> List[str]: """Mirror `cdxt -v repackage` fetch-job output.""" out = ex.files[0] if ex.files else f"{ex.name}-001.warc.gz" return [ f"+ cdxt -v repackage --target-source csv --csv-path ranges.csv " f"--warc-download-prefix hf://buckets/{config.CC_BUCKET} --hf-reader cdn " f"--prefix hf://buckets/{ex.bucket}/{ex.path}/{ex.name}", "INFO:cdx_toolkit:reading range jobs from ranges.csv", f"INFO:cdx_toolkit:{ex.n_records} records to fetch", "INFO:cdx_toolkit:fetching WARC ranges via hf-cdn (parallel_readers=48)", f"INFO:cdx_toolkit:{ex.n_records} records extracted", f"INFO:cdx_toolkit:wrote {out} ({_fmt_bytes(ex.total_bytes)})", "INFO:cdx_toolkit:execution time: 6s", ]