Spaces:
Running
Running
Download backend/api/pdf_handler.py from Luca448/APP-Backend: direct link, hf CLI and curl.
- Browser
- Download file 1.05 kB
-
https://huggingface.co/spaces/Luca448/APP-Backend/resolve/main/backend/api/pdf_handler.py
- Command line
-
hf download hf://spaces/Luca448/APP-Backend/backend/api/pdf_handler.py
-
curl -L -o pdf_handler.py https://huggingface.co/spaces/Luca448/APP-Backend/resolve/main/backend/api/pdf_handler.py
1.05 kB
| from fastapi import APIRouter, UploadFile, File, HTTPException | |
| import fitz # PyMuPDF | |
| import io | |
| router = APIRouter() | |
| async def extract_pdf(file: UploadFile = File(...)): | |
| """ | |
| Receives a PDF file and extracts its text content using PyMuPDF. | |
| """ | |
| if file.content_type != "application/pdf": | |
| raise HTTPException(status_code=400, detail="File must be a PDF") | |
| try: | |
| contents = await file.read() | |
| pdf_document = fitz.open(stream=contents, filetype="pdf") | |
| extracted_text = "" | |
| for page_num in range(len(pdf_document)): | |
| page = pdf_document.load_page(page_num) | |
| extracted_text += f"\n--- Seite {page_num + 1} ---\n" | |
| extracted_text += page.get_text() | |
| pdf_document.close() | |
| return { | |
| "filename": file.filename, | |
| "text": extracted_text, | |
| "pages": len(pdf_document) | |
| } | |
| except Exception as e: | |
| raise HTTPException(status_code=500, detail=str(e)) | |