Spaces:
Runtime error
Runtime error
Download api_export.py from sdudeja/agentic-extractor: direct link, hf CLI and curl.
- Browser
- Download file 6.33 kB
-
https://huggingface.co/spaces/sdudeja/agentic-extractor/resolve/main/api_export.py
- Command line
-
hf download hf://spaces/sdudeja/agentic-extractor/api_export.py
-
curl -L -o api_export.py https://huggingface.co/spaces/sdudeja/agentic-extractor/resolve/main/api_export.py
6.33 kB
| """ | |
| DocuLens — Export endpoints. | |
| Provides CSV and JSON export of extraction results | |
| from pipeline runs stored in Supabase. | |
| """ | |
| import csv | |
| import io | |
| import json | |
| import logging | |
| from fastapi import APIRouter, HTTPException, Depends, Query | |
| from fastapi.responses import StreamingResponse | |
| from middleware import check_rate_limit | |
| logger = logging.getLogger(__name__) | |
| export_router = APIRouter() | |
| # --------------------------------------------------------------------------- | |
| # Helpers | |
| # --------------------------------------------------------------------------- | |
| def _get_run_data(run_id: str) -> dict: | |
| """Fetch extraction_data from a pipeline run.""" | |
| from db.supabase import get_client | |
| client = get_client() | |
| if not client: | |
| raise HTTPException(status_code=503, detail="Database not configured") | |
| resp = ( | |
| client.table("pipeline_runs") | |
| .select("id, use_case, extraction_data, overall_result, processing_time_ms, started_at") | |
| .eq("id", run_id) | |
| .execute() | |
| ) | |
| if not resp.data: | |
| raise HTTPException(status_code=404, detail=f"Run '{run_id}' not found") | |
| return resp.data[0] | |
| def _flatten_fields(data: dict) -> dict: | |
| """Extract flat key-value fields from extraction_data.""" | |
| fields = {} | |
| # extracted_fields (primary — works for all use cases) | |
| if "extracted_fields" in data and isinstance(data["extracted_fields"], dict): | |
| for k, v in data["extracted_fields"].items(): | |
| fields[k] = v | |
| # Legacy named fields (invoice-specific) | |
| skip_keys = { | |
| "line_items", "bounding_boxes", "extracted_fields", | |
| "extraction_method", "model_id", "page_image", | |
| "page_images", "processing_time_ms", "confidence", | |
| "document_type", "raw_text", | |
| } | |
| for k, v in data.items(): | |
| if k not in skip_keys and k not in fields and not isinstance(v, (dict, list)): | |
| fields[k] = v | |
| return fields | |
| def _extraction_to_csv(data: dict) -> str: | |
| """Convert extraction data to CSV string.""" | |
| output = io.StringIO() | |
| # Section 1: Header fields | |
| fields = _flatten_fields(data) | |
| if fields: | |
| writer = csv.writer(output) | |
| writer.writerow(["Field", "Value"]) | |
| for k, v in fields.items(): | |
| writer.writerow([k, v]) | |
| output.write("\n") | |
| # Section 2: Line items | |
| line_items = data.get("line_items", []) | |
| if line_items: | |
| # Collect all unique keys across line items | |
| all_keys = [] | |
| seen = set() | |
| for item in line_items: | |
| if isinstance(item, dict): | |
| for k in item.keys(): | |
| if k not in seen: | |
| all_keys.append(k) | |
| seen.add(k) | |
| writer = csv.writer(output) | |
| writer.writerow(all_keys) | |
| for item in line_items: | |
| if isinstance(item, dict): | |
| writer.writerow([item.get(k, "") for k in all_keys]) | |
| return output.getvalue() | |
| def _extraction_to_json(data: dict, run_meta: dict) -> dict: | |
| """Structure extraction data for JSON export.""" | |
| fields = _flatten_fields(data) | |
| return { | |
| "run_id": run_meta.get("id"), | |
| "use_case": run_meta.get("use_case"), | |
| "status": run_meta.get("overall_result"), | |
| "processed_at": run_meta.get("started_at"), | |
| "processing_time_ms": run_meta.get("processing_time_ms"), | |
| "document_type": data.get("document_type", ""), | |
| "confidence": data.get("confidence", 0), | |
| "fields": fields, | |
| "line_items": data.get("line_items", []), | |
| } | |
| # --------------------------------------------------------------------------- | |
| # Endpoints | |
| # --------------------------------------------------------------------------- | |
| async def export_run( | |
| run_id: str, | |
| format: str = Query("json", regex="^(json|csv)$"), | |
| api_key: str = Depends(check_rate_limit), | |
| ): | |
| """ | |
| Export extraction results from a pipeline run. | |
| Query params: | |
| - format: "json" (default) or "csv" | |
| """ | |
| run = _get_run_data(run_id) | |
| data = run.get("extraction_data") or {} | |
| if not data: | |
| raise HTTPException(status_code=404, detail="No extraction data for this run") | |
| if format == "csv": | |
| csv_content = _extraction_to_csv(data) | |
| return StreamingResponse( | |
| io.BytesIO(csv_content.encode("utf-8")), | |
| media_type="text/csv", | |
| headers={ | |
| "Content-Disposition": f'attachment; filename="run_{run_id[:8]}.csv"', | |
| }, | |
| ) | |
| else: | |
| export_data = _extraction_to_json(data, run) | |
| json_content = json.dumps(export_data, indent=2, default=str) | |
| return StreamingResponse( | |
| io.BytesIO(json_content.encode("utf-8")), | |
| media_type="application/json", | |
| headers={ | |
| "Content-Disposition": f'attachment; filename="run_{run_id[:8]}.json"', | |
| }, | |
| ) | |
| async def export_inline( | |
| data: dict, | |
| format: str = Query("json", regex="^(json|csv)$"), | |
| api_key: str = Depends(check_rate_limit), | |
| ): | |
| """ | |
| Export extraction data directly (without a saved run). | |
| Accepts the extraction result JSON in the request body. | |
| Useful for exporting results from in-progress extractions | |
| that haven't been persisted yet. | |
| """ | |
| if format == "csv": | |
| csv_content = _extraction_to_csv(data) | |
| return StreamingResponse( | |
| io.BytesIO(csv_content.encode("utf-8")), | |
| media_type="text/csv", | |
| headers={ | |
| "Content-Disposition": 'attachment; filename="extraction.csv"', | |
| }, | |
| ) | |
| else: | |
| fields = _flatten_fields(data) | |
| export_data = { | |
| "document_type": data.get("document_type", ""), | |
| "confidence": data.get("confidence", 0), | |
| "fields": fields, | |
| "line_items": data.get("line_items", []), | |
| } | |
| json_content = json.dumps(export_data, indent=2, default=str) | |
| return StreamingResponse( | |
| io.BytesIO(json_content.encode("utf-8")), | |
| media_type="application/json", | |
| headers={ | |
| "Content-Disposition": 'attachment; filename="extraction.json"', | |
| }, | |
| ) | |