File size: 3,326 Bytes
0ebef7d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
"""Accuracy harness — the release gate.

Runs the proven engine across the golden fixture corpus and asserts that every
statement (a) yields transactions and (b) RECONCILES (the engine's own integrity
check: running-balance and/or stated opening+credits-debits==closing).

Run directly for a readable table:   python tests/test_accuracy.py
Run under pytest for the gate:        pytest -q
"""
from __future__ import annotations

import sys
from pathlib import Path

ROOT = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(ROOT))

from app.engine.converter import BankStatementConverter  # noqa: E402

FIXTURES = ROOT / "tests" / "fixtures"

# The golden corpus: (fixture, expected number of statements). Every one is
# expected to extract, reconcile, AND section into the right number of statements
# (guards against over-/under-sectioning — e.g. a multi-page statement that wrongly
# splits one-per-page).
CASES = [
    ("sample_statement.pdf", 1),
    ("complex_statement.pdf", 1),
    ("real_style_statement.pdf", 1),
    ("two_block.pdf", 2),
    ("multipage_bf.pdf", 1),  # multi-page, repeated account + "brought forward" headers
    ("hard/t1_no_gridlines.pdf", 1),
    ("hard/t2_scanned.pdf", 1),
    ("hard/t3_renamed_headers.pdf", 1),
    ("hard/t4_signed_amount.pdf", 1),
    ("hard/t5_european_format.pdf", 1),
    ("hard/t6_multi_account.pdf", 2),
    ("hard/t7_diff_balance_labels.pdf", 1),
    ("hard/t8_text_dates.pdf", 1),
]


def _run(name: str):
    statements = BankStatementConverter(FIXTURES / name).extract()
    n_txns = sum(len(s.transactions) for s in statements)
    all_reconcile = bool(statements) and all(s.reconciles is True for s in statements)
    return statements, n_txns, all_reconcile


def test_corpus_reconciles():
    """Hard gate: every fixture extracts, reconciles, and sections correctly."""
    failures = []
    for name, expect in CASES:
        statements, n_txns, ok = _run(name)
        if not statements or n_txns == 0:
            failures.append(f"{name}: no transactions extracted")
            continue
        if len(statements) != expect:
            failures.append(f"{name}: expected {expect} statement(s), got {len(statements)}")
        if not ok:
            detail = " ; ".join(f"{s.reconciles}:{s.reconciliation_detail}" for s in statements)
            failures.append(f"{name}: did not reconcile -> {detail}")
    assert not failures, "Accuracy regressions:\n" + "\n".join(failures)


def _main() -> int:
    flag = {True: "PASS", False: "FAIL", None: "N/A "}
    passed = 0
    print("=" * 80)
    print(f"{'fixture':<34} {'stmts':>5} {'exp':>4} {'txns':>5}  reconcile")
    print("-" * 80)
    for name, expect in CASES:
        statements, n_txns, ok = _run(name)
        rec = " / ".join(flag[s.reconciles] for s in statements) or "—"
        good = bool(statements) and n_txns > 0 and ok and len(statements) == expect
        passed += 1 if good else 0
        mark = "" if len(statements) == expect else "  <-- SECTION MISMATCH"
        print(f"{name:<34} {len(statements):>5} {expect:>4} {n_txns:>5}  {rec}{mark}")
    print("=" * 80)
    print(f"RESULT: {passed}/{len(CASES)} fixtures extracted + reconciled + sectioned correctly")
    return 0 if passed == len(CASES) else 1


if __name__ == "__main__":
    raise SystemExit(_main())