Spaces:
Running on Zero
Running on Zero
Download tests/test_extract.py from spacedout-bits/Oracle: direct link, hf CLI and curl.
- Browser
- Download file 3.12 kB
-
https://huggingface.co/spaces/spacedout-bits/Oracle/resolve/main/tests/test_extract.py
- Command line
-
hf download hf://spaces/spacedout-bits/Oracle/tests/test_extract.py
-
curl -L -o test_extract.py https://huggingface.co/spaces/spacedout-bits/Oracle/resolve/main/tests/test_extract.py
3.12 kB
| import io | |
| from finbot.extract import extract, extract_csv, extract_xlsx, rows_to_text | |
| CSV_BYTES = ( | |
| b"Date,Narration,Withdrawal,Deposit\n" | |
| b"01/08/2026,SWIGGY BANGALORE,450.00,\n" | |
| b"03/08/2026,SALARY AUG,,85000.00\n" | |
| b"05/08/2026,NETFLIX.COM,649.00,\n" | |
| ) | |
| def make_xlsx() -> bytes: | |
| from openpyxl import Workbook | |
| workbook = Workbook() | |
| sheet = workbook.active | |
| sheet.append(["Date", "Description", "Amount"]) | |
| sheet.append(["2026-08-01", "Swiggy", 450]) | |
| sheet.append([None, None, None]) # blank row must be dropped | |
| sheet.append(["2026-08-05", "Netflix", 649]) | |
| buffer = io.BytesIO() | |
| workbook.save(buffer) | |
| return buffer.getvalue() | |
| class TestCSV: | |
| def test_parses_rows(self): | |
| doc = extract_csv(CSV_BYTES) | |
| assert doc.kind == "csv" | |
| assert len(doc.rows) == 4 # header + 3 | |
| assert doc.rows[1][1] == "SWIGGY BANGALORE" | |
| def test_semicolon_delimiter(self): | |
| doc = extract_csv(b"a;b;c\n1;2;3\n4;5;6\n") | |
| assert doc.rows[0] == ["a", "b", "c"] | |
| def test_skips_blank_rows(self): | |
| doc = extract_csv(b"a,b\n\n1,2\n") | |
| assert len(doc.rows) == 2 | |
| def test_handles_bom(self): | |
| doc = extract_csv(b"\xef\xbb\xbfDate,Amount\n2026-08-01,10\n") | |
| assert doc.rows[0][0] == "Date" | |
| def test_handles_non_utf8(self): | |
| doc = extract_csv("Caf\xe9,10\n".encode("latin-1")) | |
| assert "Caf" in doc.rows[0][0] | |
| def test_single_column_does_not_crash(self): | |
| doc = extract_csv(b"justonecolumn\nvalue\n") | |
| assert len(doc.rows) == 2 | |
| class TestXLSX: | |
| def test_parses_rows(self): | |
| doc = extract_xlsx(make_xlsx()) | |
| assert doc.kind == "xlsx" | |
| assert doc.rows[0] == ["Date", "Description", "Amount"] | |
| assert ["2026-08-05", "Netflix", "649"] in doc.rows | |
| def test_drops_blank_rows(self): | |
| doc = extract_xlsx(make_xlsx()) | |
| assert all(any(cell for cell in row) for row in doc.rows) | |
| class TestDispatch: | |
| def test_by_extension(self): | |
| assert extract(CSV_BYTES, "statement.csv").kind == "csv" | |
| assert extract(make_xlsx(), "book.xlsx").kind == "xlsx" | |
| def test_by_mime_when_no_extension(self): | |
| assert extract(CSV_BYTES, "", "text/csv").kind == "csv" | |
| def test_plain_text(self): | |
| doc = extract(b"hello", "notes.txt") | |
| assert doc.kind == "text" | |
| assert doc.text == "hello" | |
| def test_legacy_xls_is_reported_not_crashed(self): | |
| doc = extract(b"\xd0\xcf\x11\xe0", "old.xls") | |
| assert doc.kind == "unsupported" | |
| assert ".xlsx" in doc.note | |
| def test_unknown_type_explains_itself(self): | |
| doc = extract(b"\x00\x01", "thing.bin") | |
| assert doc.kind == "unsupported" | |
| assert doc.note | |
| def test_is_empty(self): | |
| assert extract(b"", "empty.txt").is_empty | |
| class TestRowsToText: | |
| def test_pipe_delimited(self): | |
| text = rows_to_text([["a", "b"], ["1", "2"]]) | |
| assert text == "a | b\n1 | 2" | |
| def test_respects_limit(self): | |
| text = rows_to_text([["x"]] * 100, limit=3) | |
| assert len(text.splitlines()) == 3 | |