File size: 8,033 Bytes
c380737
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
"""Precomputed example presets shown in the app so a visitor can try the tool WITHOUT
signing in and WITHOUT running any real HF job.

Each example's output WARC(s) were produced once by ``scripts/build_examples.py`` and
uploaded to a public HF bucket, so the "Download" link is a real, working file URL. The
app streams *simulated* (but realistic) job logs and renders the precomputed estimate and
result — no ``JobRunner`` call, no OAuth token needed.

The record counts / byte sizes / filenames below come from the actual build; update them
whenever the datasets are rebuilt. `n_records` for the capped examples equals `max_records`.
"""
from __future__ import annotations

from dataclasses import dataclass, field
from typing import List

from . import config
from .estimate import Estimate, format_estimate
from .jobs import bucket_file_url, bucket_web_url

# Public bucket that holds the precomputed example outputs (must be public-readable).
EXAMPLES_BUCKET = "malteos/cc-repackage-examples"


@dataclass(frozen=True)
class Example:
    key: str
    title: str
    description: str
    # Form values the button fills in (mirrors the live filter fields).
    sql_where: str = ""
    domains: List[str] = field(default_factory=list)
    hostnames: List[str] = field(default_factory=list)
    languages: List[str] = field(default_factory=list)
    crawls: List[str] = field(default_factory=list)
    name: str = "repackaged"
    max_records: int = 0
    # Precomputed results (from the build) — n_records/total_bytes drive the estimate.
    n_records: int = 0
    total_bytes: int = 0
    # Hosting on the public examples bucket.
    bucket: str = EXAMPLES_BUCKET
    path: str = ""
    files: List[str] = field(default_factory=list)  # e.g. ["commoncrawl-org-001.warc.gz"]
    # Representative count for the "N parquet file(s) to scan" log line (from the build).
    scan_files: int = 300

    def download_urls(self) -> List[tuple]:
        """(filename, resolve-url) for each produced WARC — real, direct downloads."""
        return [(f, bucket_file_url(self.bucket, self.path, f)) for f in self.files]

    def folder_url(self) -> str:
        return bucket_web_url(self.bucket, self.path)


# The "from build" values below were produced by scripts/build_examples.py and uploaded to
# EXAMPLES_BUCKET; re-run that script and refresh them whenever the datasets are rebuilt.
_CRAWL = "CC-MAIN-2026-21"

EXAMPLES: List[Example] = [
    Example(
        key="commoncrawl-org",
        title="All commoncrawl.org pages",
        description="Every capture of commoncrawl.org in one crawl — uses just the "
                    "Registered domains field, no record cap.",
        domains=["commoncrawl.org"],
        crawls=[_CRAWL],
        name="commoncrawl-org",
        max_records=0,
        n_records=307,        # from build
        total_bytes=2_494_825,  # from build
        path="commoncrawl-org",
        files=["commoncrawl-org-001.warc.gz"],  # from build
        scan_files=300,       # from build
    ),
    Example(
        key="wikipedia-fr",
        title="10 French Wikipedia pages",
        description="French-language pages under wikipedia.org — uses the Registered domains "
                    "+ Content languages fields with a Max records cap (no SQL).",
        domains=["wikipedia.org"],
        languages=["fra"],
        crawls=[_CRAWL],
        name="wikipedia-fr",
        max_records=10,
        n_records=10,
        total_bytes=334_225,  # from build
        path="wikipedia-fr",
        files=["wikipedia-fr-001.warc.gz"],  # from build
        scan_files=300,       # from build
    ),
    Example(
        key="pdfs",
        title="10 PDF documents from one crawl",
        description="The first 10 records the index detected as application/pdf in one crawl.",
        sql_where="content_mime_detected = 'application/pdf'",
        crawls=[_CRAWL],
        name="pdfs",
        max_records=10,
        n_records=10,
        total_bytes=19_857_520,  # from build
        path="pdfs",
        files=["pdfs-001.warc.gz"],  # from build
        scan_files=8,         # from build (capped via --max-files)
    ),
    Example(
        key="homepages",
        title="500 homepages from one crawl",
        description="The first 500 records whose URL path is '/' (site homepages) in one crawl.",
        sql_where="url_path = '/'",
        crawls=[_CRAWL],
        name="homepages",
        max_records=500,
        n_records=500,
        total_bytes=12_263_615,  # from build
        path="homepages",
        files=["homepages-001.warc.gz"],  # from build
        scan_files=4,         # from build (capped via --max-files)
    ),
]

EXAMPLES_BY_KEY = {ex.key: ex for ex in EXAMPLES}


# --- Human-readable helpers --------------------------------------------------

def _fmt_bytes(n: int) -> str:
    step = 1024.0
    val = float(n)
    for unit in ("B", "KiB", "MiB", "GiB", "TiB"):
        if val < step or unit == "TiB":
            return f"{val:.0f} {unit}" if unit == "B" else f"{val:.1f} {unit}"
        val /= step
    return f"{n} B"


# --- Rendered markdown (reuses the real estimate/link builders) --------------

def estimate_markdown(ex: Example) -> str:
    """The same cost-estimate block a real estimate job would produce."""
    return format_estimate(Estimate(ex.n_records, ex.total_bytes, config.DEFAULT_FLAVOR))


def result_markdown(ex: Example) -> str:
    """The 'repackaging complete' panel — with REAL download links to the hosted WARC(s)."""
    downloads = "\n".join(
        f"- **[⬇️ Download `{fname}`]({url})**" for fname, url in ex.download_urls()
    )
    return (
        "### ✅ Repackaging complete\n"
        "🧪 _This is a **precomputed example** — no HF Job ran and no sign-in was needed. "
        "Sign in above to run your own repackaging._\n\n"
        f"**{ex.n_records:,} records · {_fmt_bytes(ex.total_bytes)}** in the public "
        "examples bucket:\n\n"
        f"{downloads}\n\n"
        f"**[📂 Browse the example folder]({ex.folder_url()})**"
    )


# --- Simulated (but realistic) job logs, streamed line-by-line ---------------

def estimate_log_lines(ex: Example) -> List[str]:
    """Mirror src/index_query.py's real estimate-job output (SQL vs structured filter)."""
    limit_note = f" limit={ex.max_records}" if ex.max_records > 0 else ""
    if ex.sql_where:
        filter_line = f"[index] mode=cdn crawls=['{ex.crawls[0]}'] sql_where={ex.sql_where!r}{limit_note}"
    else:
        filter_line = (
            f"[index] mode=cdn crawls=['{ex.crawls[0]}'] hostnames={ex.hostnames} "
            f"domains={ex.domains} languages={ex.languages}{limit_note}"
        )
    lines = [
        "+ pip install --quiet duckdb huggingface_hub",
        "Successfully installed duckdb huggingface_hub",
        filter_line,
        "[index] enumerating index parquet files via the bucket API…",
        f"[index] {ex.scan_files} parquet file(s) to scan",
    ]
    if ex.sql_where:
        lines.append("[index] validating SQL filter…")
    lines += [
        "[index] running query + writing range-jobs CSV…",
        f"ESTIMATE n_records={ex.n_records} total_bytes={ex.total_bytes}",
    ]
    return lines


def fetch_log_lines(ex: Example) -> List[str]:
    """Mirror `cdxt -v repackage` fetch-job output."""
    out = ex.files[0] if ex.files else f"{ex.name}-001.warc.gz"
    return [
        f"+ cdxt -v repackage --target-source csv --csv-path ranges.csv "
        f"--warc-download-prefix hf://buckets/{config.CC_BUCKET} --hf-reader cdn "
        f"--prefix hf://buckets/{ex.bucket}/{ex.path}/{ex.name}",
        "INFO:cdx_toolkit:reading range jobs from ranges.csv",
        f"INFO:cdx_toolkit:{ex.n_records} records to fetch",
        "INFO:cdx_toolkit:fetching WARC ranges via hf-cdn (parallel_readers=48)",
        f"INFO:cdx_toolkit:{ex.n_records} records extracted",
        f"INFO:cdx_toolkit:wrote {out} ({_fmt_bytes(ex.total_bytes)})",
        "INFO:cdx_toolkit:execution time: 6s",
    ]