Spaces:
Sleeping
Sleeping
File size: 8,033 Bytes
c380737 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 | """Precomputed example presets shown in the app so a visitor can try the tool WITHOUT
signing in and WITHOUT running any real HF job.
Each example's output WARC(s) were produced once by ``scripts/build_examples.py`` and
uploaded to a public HF bucket, so the "Download" link is a real, working file URL. The
app streams *simulated* (but realistic) job logs and renders the precomputed estimate and
result — no ``JobRunner`` call, no OAuth token needed.
The record counts / byte sizes / filenames below come from the actual build; update them
whenever the datasets are rebuilt. `n_records` for the capped examples equals `max_records`.
"""
from __future__ import annotations
from dataclasses import dataclass, field
from typing import List
from . import config
from .estimate import Estimate, format_estimate
from .jobs import bucket_file_url, bucket_web_url
# Public bucket that holds the precomputed example outputs (must be public-readable).
EXAMPLES_BUCKET = "malteos/cc-repackage-examples"
@dataclass(frozen=True)
class Example:
key: str
title: str
description: str
# Form values the button fills in (mirrors the live filter fields).
sql_where: str = ""
domains: List[str] = field(default_factory=list)
hostnames: List[str] = field(default_factory=list)
languages: List[str] = field(default_factory=list)
crawls: List[str] = field(default_factory=list)
name: str = "repackaged"
max_records: int = 0
# Precomputed results (from the build) — n_records/total_bytes drive the estimate.
n_records: int = 0
total_bytes: int = 0
# Hosting on the public examples bucket.
bucket: str = EXAMPLES_BUCKET
path: str = ""
files: List[str] = field(default_factory=list) # e.g. ["commoncrawl-org-001.warc.gz"]
# Representative count for the "N parquet file(s) to scan" log line (from the build).
scan_files: int = 300
def download_urls(self) -> List[tuple]:
"""(filename, resolve-url) for each produced WARC — real, direct downloads."""
return [(f, bucket_file_url(self.bucket, self.path, f)) for f in self.files]
def folder_url(self) -> str:
return bucket_web_url(self.bucket, self.path)
# The "from build" values below were produced by scripts/build_examples.py and uploaded to
# EXAMPLES_BUCKET; re-run that script and refresh them whenever the datasets are rebuilt.
_CRAWL = "CC-MAIN-2026-21"
EXAMPLES: List[Example] = [
Example(
key="commoncrawl-org",
title="All commoncrawl.org pages",
description="Every capture of commoncrawl.org in one crawl — uses just the "
"Registered domains field, no record cap.",
domains=["commoncrawl.org"],
crawls=[_CRAWL],
name="commoncrawl-org",
max_records=0,
n_records=307, # from build
total_bytes=2_494_825, # from build
path="commoncrawl-org",
files=["commoncrawl-org-001.warc.gz"], # from build
scan_files=300, # from build
),
Example(
key="wikipedia-fr",
title="10 French Wikipedia pages",
description="French-language pages under wikipedia.org — uses the Registered domains "
"+ Content languages fields with a Max records cap (no SQL).",
domains=["wikipedia.org"],
languages=["fra"],
crawls=[_CRAWL],
name="wikipedia-fr",
max_records=10,
n_records=10,
total_bytes=334_225, # from build
path="wikipedia-fr",
files=["wikipedia-fr-001.warc.gz"], # from build
scan_files=300, # from build
),
Example(
key="pdfs",
title="10 PDF documents from one crawl",
description="The first 10 records the index detected as application/pdf in one crawl.",
sql_where="content_mime_detected = 'application/pdf'",
crawls=[_CRAWL],
name="pdfs",
max_records=10,
n_records=10,
total_bytes=19_857_520, # from build
path="pdfs",
files=["pdfs-001.warc.gz"], # from build
scan_files=8, # from build (capped via --max-files)
),
Example(
key="homepages",
title="500 homepages from one crawl",
description="The first 500 records whose URL path is '/' (site homepages) in one crawl.",
sql_where="url_path = '/'",
crawls=[_CRAWL],
name="homepages",
max_records=500,
n_records=500,
total_bytes=12_263_615, # from build
path="homepages",
files=["homepages-001.warc.gz"], # from build
scan_files=4, # from build (capped via --max-files)
),
]
EXAMPLES_BY_KEY = {ex.key: ex for ex in EXAMPLES}
# --- Human-readable helpers --------------------------------------------------
def _fmt_bytes(n: int) -> str:
step = 1024.0
val = float(n)
for unit in ("B", "KiB", "MiB", "GiB", "TiB"):
if val < step or unit == "TiB":
return f"{val:.0f} {unit}" if unit == "B" else f"{val:.1f} {unit}"
val /= step
return f"{n} B"
# --- Rendered markdown (reuses the real estimate/link builders) --------------
def estimate_markdown(ex: Example) -> str:
"""The same cost-estimate block a real estimate job would produce."""
return format_estimate(Estimate(ex.n_records, ex.total_bytes, config.DEFAULT_FLAVOR))
def result_markdown(ex: Example) -> str:
"""The 'repackaging complete' panel — with REAL download links to the hosted WARC(s)."""
downloads = "\n".join(
f"- **[⬇️ Download `{fname}`]({url})**" for fname, url in ex.download_urls()
)
return (
"### ✅ Repackaging complete\n"
"🧪 _This is a **precomputed example** — no HF Job ran and no sign-in was needed. "
"Sign in above to run your own repackaging._\n\n"
f"**{ex.n_records:,} records · {_fmt_bytes(ex.total_bytes)}** in the public "
"examples bucket:\n\n"
f"{downloads}\n\n"
f"**[📂 Browse the example folder]({ex.folder_url()})**"
)
# --- Simulated (but realistic) job logs, streamed line-by-line ---------------
def estimate_log_lines(ex: Example) -> List[str]:
"""Mirror src/index_query.py's real estimate-job output (SQL vs structured filter)."""
limit_note = f" limit={ex.max_records}" if ex.max_records > 0 else ""
if ex.sql_where:
filter_line = f"[index] mode=cdn crawls=['{ex.crawls[0]}'] sql_where={ex.sql_where!r}{limit_note}"
else:
filter_line = (
f"[index] mode=cdn crawls=['{ex.crawls[0]}'] hostnames={ex.hostnames} "
f"domains={ex.domains} languages={ex.languages}{limit_note}"
)
lines = [
"+ pip install --quiet duckdb huggingface_hub",
"Successfully installed duckdb huggingface_hub",
filter_line,
"[index] enumerating index parquet files via the bucket API…",
f"[index] {ex.scan_files} parquet file(s) to scan",
]
if ex.sql_where:
lines.append("[index] validating SQL filter…")
lines += [
"[index] running query + writing range-jobs CSV…",
f"ESTIMATE n_records={ex.n_records} total_bytes={ex.total_bytes}",
]
return lines
def fetch_log_lines(ex: Example) -> List[str]:
"""Mirror `cdxt -v repackage` fetch-job output."""
out = ex.files[0] if ex.files else f"{ex.name}-001.warc.gz"
return [
f"+ cdxt -v repackage --target-source csv --csv-path ranges.csv "
f"--warc-download-prefix hf://buckets/{config.CC_BUCKET} --hf-reader cdn "
f"--prefix hf://buckets/{ex.bucket}/{ex.path}/{ex.name}",
"INFO:cdx_toolkit:reading range jobs from ranges.csv",
f"INFO:cdx_toolkit:{ex.n_records} records to fetch",
"INFO:cdx_toolkit:fetching WARC ranges via hf-cdn (parallel_readers=48)",
f"INFO:cdx_toolkit:{ex.n_records} records extracted",
f"INFO:cdx_toolkit:wrote {out} ({_fmt_bytes(ex.total_bytes)})",
"INFO:cdx_toolkit:execution time: 6s",
]
|