Spaces:
Sleeping
Sleeping
Download src/examples.py from malteos/cc-repackage: direct link, hf CLI and curl.
- Browser
- Download file 8.03 kB
-
https://huggingface.co/spaces/malteos/cc-repackage/resolve/main/src/examples.py
- Command line
-
hf download hf://spaces/malteos/cc-repackage/src/examples.py
-
curl -L -o examples.py https://huggingface.co/spaces/malteos/cc-repackage/resolve/main/src/examples.py
8.03 kB
| """Precomputed example presets shown in the app so a visitor can try the tool WITHOUT | |
| signing in and WITHOUT running any real HF job. | |
| Each example's output WARC(s) were produced once by ``scripts/build_examples.py`` and | |
| uploaded to a public HF bucket, so the "Download" link is a real, working file URL. The | |
| app streams *simulated* (but realistic) job logs and renders the precomputed estimate and | |
| result — no ``JobRunner`` call, no OAuth token needed. | |
| The record counts / byte sizes / filenames below come from the actual build; update them | |
| whenever the datasets are rebuilt. `n_records` for the capped examples equals `max_records`. | |
| """ | |
| from __future__ import annotations | |
| from dataclasses import dataclass, field | |
| from typing import List | |
| from . import config | |
| from .estimate import Estimate, format_estimate | |
| from .jobs import bucket_file_url, bucket_web_url | |
| # Public bucket that holds the precomputed example outputs (must be public-readable). | |
| EXAMPLES_BUCKET = "malteos/cc-repackage-examples" | |
| class Example: | |
| key: str | |
| title: str | |
| description: str | |
| # Form values the button fills in (mirrors the live filter fields). | |
| sql_where: str = "" | |
| domains: List[str] = field(default_factory=list) | |
| hostnames: List[str] = field(default_factory=list) | |
| languages: List[str] = field(default_factory=list) | |
| crawls: List[str] = field(default_factory=list) | |
| name: str = "repackaged" | |
| max_records: int = 0 | |
| # Precomputed results (from the build) — n_records/total_bytes drive the estimate. | |
| n_records: int = 0 | |
| total_bytes: int = 0 | |
| # Hosting on the public examples bucket. | |
| bucket: str = EXAMPLES_BUCKET | |
| path: str = "" | |
| files: List[str] = field(default_factory=list) # e.g. ["commoncrawl-org-001.warc.gz"] | |
| # Representative count for the "N parquet file(s) to scan" log line (from the build). | |
| scan_files: int = 300 | |
| def download_urls(self) -> List[tuple]: | |
| """(filename, resolve-url) for each produced WARC — real, direct downloads.""" | |
| return [(f, bucket_file_url(self.bucket, self.path, f)) for f in self.files] | |
| def folder_url(self) -> str: | |
| return bucket_web_url(self.bucket, self.path) | |
| # The "from build" values below were produced by scripts/build_examples.py and uploaded to | |
| # EXAMPLES_BUCKET; re-run that script and refresh them whenever the datasets are rebuilt. | |
| _CRAWL = "CC-MAIN-2026-21" | |
| EXAMPLES: List[Example] = [ | |
| Example( | |
| key="commoncrawl-org", | |
| title="All commoncrawl.org pages", | |
| description="Every capture of commoncrawl.org in one crawl — uses just the " | |
| "Registered domains field, no record cap.", | |
| domains=["commoncrawl.org"], | |
| crawls=[_CRAWL], | |
| name="commoncrawl-org", | |
| max_records=0, | |
| n_records=307, # from build | |
| total_bytes=2_494_825, # from build | |
| path="commoncrawl-org", | |
| files=["commoncrawl-org-001.warc.gz"], # from build | |
| scan_files=300, # from build | |
| ), | |
| Example( | |
| key="wikipedia-fr", | |
| title="10 French Wikipedia pages", | |
| description="French-language pages under wikipedia.org — uses the Registered domains " | |
| "+ Content languages fields with a Max records cap (no SQL).", | |
| domains=["wikipedia.org"], | |
| languages=["fra"], | |
| crawls=[_CRAWL], | |
| name="wikipedia-fr", | |
| max_records=10, | |
| n_records=10, | |
| total_bytes=334_225, # from build | |
| path="wikipedia-fr", | |
| files=["wikipedia-fr-001.warc.gz"], # from build | |
| scan_files=300, # from build | |
| ), | |
| Example( | |
| key="pdfs", | |
| title="10 PDF documents from one crawl", | |
| description="The first 10 records the index detected as application/pdf in one crawl.", | |
| sql_where="content_mime_detected = 'application/pdf'", | |
| crawls=[_CRAWL], | |
| name="pdfs", | |
| max_records=10, | |
| n_records=10, | |
| total_bytes=19_857_520, # from build | |
| path="pdfs", | |
| files=["pdfs-001.warc.gz"], # from build | |
| scan_files=8, # from build (capped via --max-files) | |
| ), | |
| Example( | |
| key="homepages", | |
| title="500 homepages from one crawl", | |
| description="The first 500 records whose URL path is '/' (site homepages) in one crawl.", | |
| sql_where="url_path = '/'", | |
| crawls=[_CRAWL], | |
| name="homepages", | |
| max_records=500, | |
| n_records=500, | |
| total_bytes=12_263_615, # from build | |
| path="homepages", | |
| files=["homepages-001.warc.gz"], # from build | |
| scan_files=4, # from build (capped via --max-files) | |
| ), | |
| ] | |
| EXAMPLES_BY_KEY = {ex.key: ex for ex in EXAMPLES} | |
| # --- Human-readable helpers -------------------------------------------------- | |
| def _fmt_bytes(n: int) -> str: | |
| step = 1024.0 | |
| val = float(n) | |
| for unit in ("B", "KiB", "MiB", "GiB", "TiB"): | |
| if val < step or unit == "TiB": | |
| return f"{val:.0f} {unit}" if unit == "B" else f"{val:.1f} {unit}" | |
| val /= step | |
| return f"{n} B" | |
| # --- Rendered markdown (reuses the real estimate/link builders) -------------- | |
| def estimate_markdown(ex: Example) -> str: | |
| """The same cost-estimate block a real estimate job would produce.""" | |
| return format_estimate(Estimate(ex.n_records, ex.total_bytes, config.DEFAULT_FLAVOR)) | |
| def result_markdown(ex: Example) -> str: | |
| """The 'repackaging complete' panel — with REAL download links to the hosted WARC(s).""" | |
| downloads = "\n".join( | |
| f"- **[⬇️ Download `{fname}`]({url})**" for fname, url in ex.download_urls() | |
| ) | |
| return ( | |
| "### ✅ Repackaging complete\n" | |
| "🧪 _This is a **precomputed example** — no HF Job ran and no sign-in was needed. " | |
| "Sign in above to run your own repackaging._\n\n" | |
| f"**{ex.n_records:,} records · {_fmt_bytes(ex.total_bytes)}** in the public " | |
| "examples bucket:\n\n" | |
| f"{downloads}\n\n" | |
| f"**[📂 Browse the example folder]({ex.folder_url()})**" | |
| ) | |
| # --- Simulated (but realistic) job logs, streamed line-by-line --------------- | |
| def estimate_log_lines(ex: Example) -> List[str]: | |
| """Mirror src/index_query.py's real estimate-job output (SQL vs structured filter).""" | |
| limit_note = f" limit={ex.max_records}" if ex.max_records > 0 else "" | |
| if ex.sql_where: | |
| filter_line = f"[index] mode=cdn crawls=['{ex.crawls[0]}'] sql_where={ex.sql_where!r}{limit_note}" | |
| else: | |
| filter_line = ( | |
| f"[index] mode=cdn crawls=['{ex.crawls[0]}'] hostnames={ex.hostnames} " | |
| f"domains={ex.domains} languages={ex.languages}{limit_note}" | |
| ) | |
| lines = [ | |
| "+ pip install --quiet duckdb huggingface_hub", | |
| "Successfully installed duckdb huggingface_hub", | |
| filter_line, | |
| "[index] enumerating index parquet files via the bucket API…", | |
| f"[index] {ex.scan_files} parquet file(s) to scan", | |
| ] | |
| if ex.sql_where: | |
| lines.append("[index] validating SQL filter…") | |
| lines += [ | |
| "[index] running query + writing range-jobs CSV…", | |
| f"ESTIMATE n_records={ex.n_records} total_bytes={ex.total_bytes}", | |
| ] | |
| return lines | |
| def fetch_log_lines(ex: Example) -> List[str]: | |
| """Mirror `cdxt -v repackage` fetch-job output.""" | |
| out = ex.files[0] if ex.files else f"{ex.name}-001.warc.gz" | |
| return [ | |
| f"+ cdxt -v repackage --target-source csv --csv-path ranges.csv " | |
| f"--warc-download-prefix hf://buckets/{config.CC_BUCKET} --hf-reader cdn " | |
| f"--prefix hf://buckets/{ex.bucket}/{ex.path}/{ex.name}", | |
| "INFO:cdx_toolkit:reading range jobs from ranges.csv", | |
| f"INFO:cdx_toolkit:{ex.n_records} records to fetch", | |
| "INFO:cdx_toolkit:fetching WARC ranges via hf-cdn (parallel_readers=48)", | |
| f"INFO:cdx_toolkit:{ex.n_records} records extracted", | |
| f"INFO:cdx_toolkit:wrote {out} ({_fmt_bytes(ex.total_bytes)})", | |
| "INFO:cdx_toolkit:execution time: 6s", | |
| ] | |