cc-repackage / src /examples.py
malteos
Add advanced SQL filter, precomputed Examples, and Max-records limit
c380737 unverified
Raw History Blame Contribute Delete
8.03 kB
"""Precomputed example presets shown in the app so a visitor can try the tool WITHOUT
signing in and WITHOUT running any real HF job.
Each example's output WARC(s) were produced once by ``scripts/build_examples.py`` and
uploaded to a public HF bucket, so the "Download" link is a real, working file URL. The
app streams *simulated* (but realistic) job logs and renders the precomputed estimate and
result — no ``JobRunner`` call, no OAuth token needed.
The record counts / byte sizes / filenames below come from the actual build; update them
whenever the datasets are rebuilt. `n_records` for the capped examples equals `max_records`.
"""
from __future__ import annotations
from dataclasses import dataclass, field
from typing import List
from . import config
from .estimate import Estimate, format_estimate
from .jobs import bucket_file_url, bucket_web_url
# Public bucket that holds the precomputed example outputs (must be public-readable).
EXAMPLES_BUCKET = "malteos/cc-repackage-examples"
@dataclass(frozen=True)
class Example:
key: str
title: str
description: str
# Form values the button fills in (mirrors the live filter fields).
sql_where: str = ""
domains: List[str] = field(default_factory=list)
hostnames: List[str] = field(default_factory=list)
languages: List[str] = field(default_factory=list)
crawls: List[str] = field(default_factory=list)
name: str = "repackaged"
max_records: int = 0
# Precomputed results (from the build) — n_records/total_bytes drive the estimate.
n_records: int = 0
total_bytes: int = 0
# Hosting on the public examples bucket.
bucket: str = EXAMPLES_BUCKET
path: str = ""
files: List[str] = field(default_factory=list) # e.g. ["commoncrawl-org-001.warc.gz"]
# Representative count for the "N parquet file(s) to scan" log line (from the build).
scan_files: int = 300
def download_urls(self) -> List[tuple]:
"""(filename, resolve-url) for each produced WARC — real, direct downloads."""
return [(f, bucket_file_url(self.bucket, self.path, f)) for f in self.files]
def folder_url(self) -> str:
return bucket_web_url(self.bucket, self.path)
# The "from build" values below were produced by scripts/build_examples.py and uploaded to
# EXAMPLES_BUCKET; re-run that script and refresh them whenever the datasets are rebuilt.
_CRAWL = "CC-MAIN-2026-21"
EXAMPLES: List[Example] = [
Example(
key="commoncrawl-org",
title="All commoncrawl.org pages",
description="Every capture of commoncrawl.org in one crawl — uses just the "
"Registered domains field, no record cap.",
domains=["commoncrawl.org"],
crawls=[_CRAWL],
name="commoncrawl-org",
max_records=0,
n_records=307, # from build
total_bytes=2_494_825, # from build
path="commoncrawl-org",
files=["commoncrawl-org-001.warc.gz"], # from build
scan_files=300, # from build
),
Example(
key="wikipedia-fr",
title="10 French Wikipedia pages",
description="French-language pages under wikipedia.org — uses the Registered domains "
"+ Content languages fields with a Max records cap (no SQL).",
domains=["wikipedia.org"],
languages=["fra"],
crawls=[_CRAWL],
name="wikipedia-fr",
max_records=10,
n_records=10,
total_bytes=334_225, # from build
path="wikipedia-fr",
files=["wikipedia-fr-001.warc.gz"], # from build
scan_files=300, # from build
),
Example(
key="pdfs",
title="10 PDF documents from one crawl",
description="The first 10 records the index detected as application/pdf in one crawl.",
sql_where="content_mime_detected = 'application/pdf'",
crawls=[_CRAWL],
name="pdfs",
max_records=10,
n_records=10,
total_bytes=19_857_520, # from build
path="pdfs",
files=["pdfs-001.warc.gz"], # from build
scan_files=8, # from build (capped via --max-files)
),
Example(
key="homepages",
title="500 homepages from one crawl",
description="The first 500 records whose URL path is '/' (site homepages) in one crawl.",
sql_where="url_path = '/'",
crawls=[_CRAWL],
name="homepages",
max_records=500,
n_records=500,
total_bytes=12_263_615, # from build
path="homepages",
files=["homepages-001.warc.gz"], # from build
scan_files=4, # from build (capped via --max-files)
),
]
EXAMPLES_BY_KEY = {ex.key: ex for ex in EXAMPLES}
# --- Human-readable helpers --------------------------------------------------
def _fmt_bytes(n: int) -> str:
step = 1024.0
val = float(n)
for unit in ("B", "KiB", "MiB", "GiB", "TiB"):
if val < step or unit == "TiB":
return f"{val:.0f} {unit}" if unit == "B" else f"{val:.1f} {unit}"
val /= step
return f"{n} B"
# --- Rendered markdown (reuses the real estimate/link builders) --------------
def estimate_markdown(ex: Example) -> str:
"""The same cost-estimate block a real estimate job would produce."""
return format_estimate(Estimate(ex.n_records, ex.total_bytes, config.DEFAULT_FLAVOR))
def result_markdown(ex: Example) -> str:
"""The 'repackaging complete' panel — with REAL download links to the hosted WARC(s)."""
downloads = "\n".join(
f"- **[⬇️ Download `{fname}`]({url})**" for fname, url in ex.download_urls()
)
return (
"### ✅ Repackaging complete\n"
"🧪 _This is a **precomputed example** — no HF Job ran and no sign-in was needed. "
"Sign in above to run your own repackaging._\n\n"
f"**{ex.n_records:,} records · {_fmt_bytes(ex.total_bytes)}** in the public "
"examples bucket:\n\n"
f"{downloads}\n\n"
f"**[📂 Browse the example folder]({ex.folder_url()})**"
)
# --- Simulated (but realistic) job logs, streamed line-by-line ---------------
def estimate_log_lines(ex: Example) -> List[str]:
"""Mirror src/index_query.py's real estimate-job output (SQL vs structured filter)."""
limit_note = f" limit={ex.max_records}" if ex.max_records > 0 else ""
if ex.sql_where:
filter_line = f"[index] mode=cdn crawls=['{ex.crawls[0]}'] sql_where={ex.sql_where!r}{limit_note}"
else:
filter_line = (
f"[index] mode=cdn crawls=['{ex.crawls[0]}'] hostnames={ex.hostnames} "
f"domains={ex.domains} languages={ex.languages}{limit_note}"
)
lines = [
"+ pip install --quiet duckdb huggingface_hub",
"Successfully installed duckdb huggingface_hub",
filter_line,
"[index] enumerating index parquet files via the bucket API…",
f"[index] {ex.scan_files} parquet file(s) to scan",
]
if ex.sql_where:
lines.append("[index] validating SQL filter…")
lines += [
"[index] running query + writing range-jobs CSV…",
f"ESTIMATE n_records={ex.n_records} total_bytes={ex.total_bytes}",
]
return lines
def fetch_log_lines(ex: Example) -> List[str]:
"""Mirror `cdxt -v repackage` fetch-job output."""
out = ex.files[0] if ex.files else f"{ex.name}-001.warc.gz"
return [
f"+ cdxt -v repackage --target-source csv --csv-path ranges.csv "
f"--warc-download-prefix hf://buckets/{config.CC_BUCKET} --hf-reader cdn "
f"--prefix hf://buckets/{ex.bucket}/{ex.path}/{ex.name}",
"INFO:cdx_toolkit:reading range jobs from ranges.csv",
f"INFO:cdx_toolkit:{ex.n_records} records to fetch",
"INFO:cdx_toolkit:fetching WARC ranges via hf-cdn (parallel_readers=48)",
f"INFO:cdx_toolkit:{ex.n_records} records extracted",
f"INFO:cdx_toolkit:wrote {out} ({_fmt_bytes(ex.total_bytes)})",
"INFO:cdx_toolkit:execution time: 6s",
]