Spaces:
Running
Running
Download tests/test_examples.py from malteos/cc-repackage: direct link, hf CLI and curl.
- Browser
- Download file 3.36 kB
-
https://huggingface.co/spaces/malteos/cc-repackage/resolve/main/tests/test_examples.py
- Command line
-
hf download hf://spaces/malteos/cc-repackage/tests/test_examples.py
-
curl -L -o test_examples.py https://huggingface.co/spaces/malteos/cc-repackage/resolve/main/tests/test_examples.py
3.36 kB
| from src import examples | |
| def test_examples_well_formed(): | |
| keys = [ex.key for ex in examples.EXAMPLES] | |
| assert len(keys) == len(set(keys)) == 4 # 4 distinct examples | |
| assert set(examples.EXAMPLES_BY_KEY) == set(keys) | |
| for ex in examples.EXAMPLES: | |
| assert ex.crawls and ex.name and ex.path and ex.files | |
| # every example filters on SOMETHING (a SQL predicate or structured host/domain) | |
| assert ex.sql_where or ex.domains or ex.hostnames | |
| assert ex.n_records >= 0 and ex.total_bytes >= 0 | |
| # A mix: some examples use the advanced SQL filter, some use the structured fields. | |
| assert any(ex.sql_where for ex in examples.EXAMPLES) | |
| assert any(not ex.sql_where for ex in examples.EXAMPLES) | |
| by = examples.EXAMPLES_BY_KEY | |
| assert by["commoncrawl-org"].max_records == 0 and not by["commoncrawl-org"].sql_where | |
| assert by["wikipedia-fr"].domains == ["wikipedia.org"] and by["wikipedia-fr"].languages == ["fra"] | |
| assert by["pdfs"].max_records == 10 and by["pdfs"].sql_where | |
| assert by["homepages"].max_records == 500 and by["homepages"].sql_where | |
| # capped examples: n_records equals the cap | |
| for k in ("wikipedia-fr", "pdfs", "homepages"): | |
| assert by[k].n_records == by[k].max_records | |
| def test_download_urls_are_real_resolve_links(): | |
| ex = examples.EXAMPLES_BY_KEY["commoncrawl-org"] | |
| (fname, url), = ex.download_urls() | |
| assert fname == ex.files[0] | |
| assert url == ( | |
| f"https://huggingface.co/buckets/{ex.bucket}/resolve/{ex.path}/{ex.files[0]}" | |
| ) | |
| assert ex.folder_url().endswith(f"/tree/{ex.path}") | |
| def test_estimate_markdown_renders(): | |
| md = examples.estimate_markdown(examples.EXAMPLES_BY_KEY["homepages"]) | |
| assert "Cost estimate" in md | |
| assert "500" in md # record count | |
| def test_result_markdown_has_download_and_note(): | |
| ex = examples.EXAMPLES_BY_KEY["wikipedia-fr"] | |
| md = examples.result_markdown(ex) | |
| assert "precomputed example" in md.lower() | |
| assert "Download" in md | |
| assert ex.download_urls()[0][1] in md # the real resolve URL | |
| assert ex.folder_url() in md | |
| def test_estimate_log_lines_contains_estimate_summary(): | |
| ex = examples.EXAMPLES_BY_KEY["pdfs"] | |
| lines = examples.estimate_log_lines(ex) | |
| assert any(l.startswith("[index] mode=cdn") for l in lines) | |
| assert any(l.startswith("[index] validating SQL filter") for l in lines) # has sql_where | |
| assert any(f"ESTIMATE n_records={ex.n_records} total_bytes={ex.total_bytes}" in l for l in lines) | |
| def test_estimate_log_lines_structured_example(): | |
| """A structured example logs its domains/languages, not a sql_where, and skips the | |
| SQL-validation line.""" | |
| ex = examples.EXAMPLES_BY_KEY["wikipedia-fr"] | |
| lines = examples.estimate_log_lines(ex) | |
| assert any("domains=['wikipedia.org']" in l and "languages=['fra']" in l for l in lines) | |
| assert not any("sql_where=" in l for l in lines) | |
| assert not any("validating SQL filter" in l for l in lines) | |
| assert any(f"ESTIMATE n_records={ex.n_records}" in l for l in lines) | |
| def test_fetch_log_lines_realistic(): | |
| ex = examples.EXAMPLES_BY_KEY["homepages"] | |
| lines = examples.fetch_log_lines(ex) | |
| assert any("records extracted" in l for l in lines) | |
| assert any(ex.files[0] in l for l in lines) # names the produced WARC | |