Spaces:
Sleeping
Sleeping
Download app.py from gfathertech/markdown: direct link, hf CLI and curl.
- Browser
- Download file 11.1 kB
-
https://huggingface.co/spaces/gfathertech/markdown/resolve/main/app.py
- Command line
-
hf download hf://spaces/gfathertech/markdown/app.py
-
curl -L -o app.py https://huggingface.co/spaces/gfathertech/markdown/resolve/main/app.py
11.1 kB
| import os | |
| import shutil | |
| import subprocess | |
| import traceback | |
| import wave | |
| from uuid import uuid4 | |
| from fastapi import FastAPI, UploadFile, HTTPException | |
| from markitdown import MarkItDown | |
| import gradio as gr | |
| from PIL import Image | |
| import pytesseract | |
| from pdf2image import convert_from_path | |
| import riva.client as rcli | |
| app = FastAPI(title="Free MarkItDown & Real-Time Voice Processing") | |
| md = MarkItDown() | |
| IMAGE_EXTENSIONS = {".png", ".jpg", ".jpeg", ".bmp", ".tiff", ".webp"} | |
| AUDIO_EXTENSIONS = {".mp3", ".wav", ".m4a", ".ogg", ".flac", ".opus"} | |
| LEGACY_OFFICE_EXTENSIONS = {".doc", ".xls", ".ppt", ".rtf"} | |
| # NVIDIA Riva gRPC Configuration Endpoint | |
| NVIDIA_STAGE_SERVER = "grpc.nvcf.nvidia.com:443" | |
| NVIDIA_PARAKEET_FUNCTION_ID = "b0e8b4a5-217c-40b7-9b96-17d84e666317" | |
| def get_nvidia_api_keys(): | |
| """Dynamically collects NVIDIA API keys from environment variables.""" | |
| keys = [] | |
| # Check for numbered environment keys: NVIDIA_API_KEY_1, NVIDIA_API_KEY_2, etc. | |
| index = 1 | |
| while True: | |
| key = os.getenv(f"NVIDIA_API_KEY_{index}") | |
| if key and key.strip(): | |
| keys.append(key.strip()) | |
| index += 1 | |
| else: | |
| break | |
| # Fallback to single NVIDIA_API_KEY if numbered keys aren't set | |
| if not keys: | |
| single_key = os.getenv("NVIDIA_API_KEY") | |
| if single_key and single_key.strip(): | |
| keys.append(single_key.strip()) | |
| return keys | |
| def transcribe_with_nvidia_riva(audio_path: str) -> str: | |
| """Transcribes audio using NVIDIA's gRPC client with multi-key failover.""" | |
| api_keys = get_nvidia_api_keys() | |
| if not api_keys: | |
| return "Audio Error: No NVIDIA API keys (NVIDIA_API_KEY_1, NVIDIA_API_KEY_2...) set in environment variables." | |
| converted_wav = f"/tmp/{uuid4()}.wav" | |
| # 1. Convert any input format (MP3, OPUS, uploaded WAV, recorded WAV) | |
| # into strictly 16kHz, Mono, 16-bit PCM WAV using FFmpeg | |
| try: | |
| subprocess.run( | |
| [ | |
| "ffmpeg", "-y", "-i", audio_path, | |
| "-ac", "1", | |
| "-ar", "16000", | |
| "-sample_fmt", "s16", | |
| converted_wav | |
| ], | |
| check=True, | |
| stdout=subprocess.DEVNULL, | |
| stderr=subprocess.DEVNULL | |
| ) | |
| except Exception as e: | |
| return f"Audio Pre-processing Error (FFmpeg): {str(e)}" | |
| last_error = "" | |
| audio_data = None | |
| # 2. Extract ONLY raw PCM frames, stripping the 44-byte WAV header | |
| try: | |
| with wave.open(converted_wav, "rb") as wf: | |
| audio_data = wf.readframes(wf.getnframes()) | |
| except Exception as e: | |
| last_error = f"WAV Parse Error: {str(e)}" | |
| try: | |
| if audio_data: | |
| # 3. Rotate through API keys upon rate-limiting or errors | |
| for idx, key in enumerate(api_keys, start=1): | |
| try: | |
| auth = rcli.Auth( | |
| use_ssl=True, | |
| uri=NVIDIA_STAGE_SERVER, | |
| metadata_args=[ | |
| ("authorization", f"Bearer {key}"), | |
| ("function-id", NVIDIA_PARAKEET_FUNCTION_ID) | |
| ] | |
| ) | |
| asr_service = rcli.ASRService(auth) | |
| config = rcli.RecognitionConfig( | |
| encoding=rcli.AudioEncoding.LINEAR_PCM, | |
| sample_rate_hertz=16000, | |
| language_code="en-US", | |
| max_alternatives=1, | |
| audio_channel_count=1 | |
| ) | |
| response = asr_service.offline_recognize(audio_data, config) | |
| results = response.results | |
| transcripts = [res.alternatives[0].transcript for res in results if res.alternatives] | |
| text = " ".join(transcripts).strip() | |
| return f"## Audio Transcription\n\n{text}" if text else "Audio processed, but no speech was detected." | |
| except Exception as e: | |
| last_error = str(e) | |
| print(f"NVIDIA API Key #{idx} failed: {last_error}. Trying next key...") | |
| continue | |
| return f"Audio Transcription Error: All provided NVIDIA API keys failed. Last error: {last_error}" | |
| finally: | |
| if os.path.exists(converted_wav): | |
| os.remove(converted_wav) | |
| def run_free_ocr(file_path: str) -> str: | |
| """Fallback OCR pipeline using Tesseract (images and PDFs only).""" | |
| file_ext = os.path.splitext(file_path)[1].lower() | |
| if file_ext in IMAGE_EXTENSIONS: | |
| img = Image.open(file_path) | |
| return pytesseract.image_to_string(img) | |
| elif file_ext == ".pdf": | |
| pages = convert_from_path(file_path) | |
| extracted_text = [] | |
| for i, page in enumerate(pages): | |
| text = pytesseract.image_to_string(page) | |
| if text.strip(): | |
| extracted_text.append(f"## Page {i + 1}\n\n{text}") | |
| return "\n\n".join(extracted_text) | |
| return "" | |
| def convert_legacy_office_file(file_path: str, ext: str) -> str: | |
| """Converts legacy .doc, .xls, .ppt to modern XML format via LibreOffice.""" | |
| out_dir = "/tmp" | |
| target_ext = "docx" if ext == ".doc" else "xlsx" if ext == ".xls" else "pptx" | |
| subprocess.run( | |
| ["libreoffice", "--headless", "--convert-to", target_ext, file_path, "--outdir", out_dir], | |
| check=True | |
| ) | |
| base_name = os.path.splitext(os.path.basename(file_path))[0] | |
| return os.path.join(out_dir, f"{base_name}.{target_ext}") | |
| def process_input(source_path_or_url: str, original_filename: str = None) -> str: | |
| if not source_path_or_url or not source_path_or_url.strip(): | |
| return "Please supply a valid file or URL." | |
| target = source_path_or_url.strip() | |
| if target.startswith("http://") or target.startswith("https://"): | |
| try: | |
| res = md.convert(target) | |
| return res.text_content if res else "No text extracted from URL." | |
| except Exception as e: | |
| return f"URL Conversion Error: {str(e)}" | |
| ext = os.path.splitext(original_filename or target)[1].lower() | |
| working_path = target | |
| temp_with_ext = None | |
| if ext and not target.lower().endswith(ext): | |
| temp_with_ext = f"{target}{ext}" | |
| shutil.copyfile(target, temp_with_ext) | |
| working_path = temp_with_ext | |
| temp_converted_office = None | |
| try: | |
| # Route 1: Audio files (Uploaded or Recorded) -> NVIDIA Riva gRPC | |
| if ext in AUDIO_EXTENSIONS: | |
| return transcribe_with_nvidia_riva(working_path) | |
| # Route 2: Images -> Tesseract OCR | |
| if ext in IMAGE_EXTENSIONS: | |
| return run_free_ocr(working_path) | |
| # Route 3: Legacy Office (.doc, .xls) -> LibreOffice | |
| if ext in LEGACY_OFFICE_EXTENSIONS: | |
| temp_converted_office = convert_legacy_office_file(working_path, ext) | |
| working_path = temp_converted_office | |
| # Route 4: Standard MarkItDown extraction (PDFs, DOCX, XLSX, PPTX, HTML, TXT) | |
| result = md.convert(working_path) | |
| text_content = result.text_content if result else "" | |
| if not text_content.strip() and ext == ".pdf": | |
| text_content = run_free_ocr(working_path) | |
| return text_content if text_content.strip() else "Unable to extract text from document." | |
| except Exception as e: | |
| if ext in IMAGE_EXTENSIONS or ext == ".pdf": | |
| try: | |
| ocr_text = run_free_ocr(working_path) | |
| if ocr_text and ocr_text.strip(): | |
| return ocr_text | |
| except Exception: | |
| pass | |
| print("--- CONVERSION ERROR ---") | |
| traceback.print_exc() | |
| return f"Conversion Error: File type '{ext}' could not be processed." | |
| finally: | |
| if temp_with_ext and os.path.exists(temp_with_ext): | |
| os.remove(temp_with_ext) | |
| if temp_converted_office and os.path.exists(temp_converted_office): | |
| os.remove(temp_converted_office) | |
| # ------------------------------------------------------------- | |
| # FastAPI Endpoint | |
| # ------------------------------------------------------------- | |
| async def convert_file_api(file: UploadFile = None, url: str = None): | |
| if not file and not url: | |
| raise HTTPException(status_code=400, detail="Provide either a 'file' upload or a 'url' parameter.") | |
| if url: | |
| markdown_text = process_input(url) | |
| return {"source": url, "markdown": markdown_text} | |
| file_ext = os.path.splitext(file.filename)[1].lower() | |
| temp_path = f"/tmp/{uuid4()}{file_ext}" | |
| try: | |
| with open(temp_path, "wb") as buffer: | |
| shutil.copyfileobj(file.file, buffer) | |
| markdown_text = process_input(temp_path, original_filename=file.filename) | |
| return {"filename": file.filename, "markdown": markdown_text} | |
| except Exception as e: | |
| raise HTTPException(status_code=500, detail=f"Conversion error: {str(e)}") | |
| finally: | |
| if os.path.exists(temp_path): | |
| os.remove(temp_path) | |
| # ------------------------------------------------------------- | |
| # Gradio UI with Real-time Recording & File Handling | |
| # ------------------------------------------------------------- | |
| def gradio_handler(mic_file_path, uploaded_file_obj, url_input): | |
| """ | |
| Evaluates inputs with priority given to live microphone recordings. | |
| """ | |
| if mic_file_path is not None: | |
| return process_input(mic_file_path, original_filename="recorded_audio.wav") | |
| elif uploaded_file_obj is not None: | |
| orig_name = getattr(uploaded_file_obj, "orig_name", uploaded_file_obj.name) | |
| return process_input(uploaded_file_obj.name, original_filename=orig_name) | |
| elif url_input and url_input.strip(): | |
| return process_input(url_input) | |
| return "Please record voice, upload a file, or provide a URL." | |
| with gr.Blocks(title="Free Converter & Voice Transcriber") as demo: | |
| gr.Markdown("# 📄 Document & Real-Time Voice Transcriber") | |
| with gr.Row(): | |
| with gr.Column(): | |
| gr.Markdown("### Option 1: Record Live Voice") | |
| mic_input = gr.Audio( | |
| sources=["microphone"], | |
| type="filepath", | |
| label="Click microphone to start recording" | |
| ) | |
| gr.Markdown("---") | |
| gr.Markdown("### Option 2: Upload File or URL") | |
| file_input = gr.File(label="Upload File (.pdf, .doc, .mp3, .wav, .png, etc.)") | |
| url_input = gr.Textbox(label="Or enter URL", placeholder="https://example.com/document.pdf") | |
| submit_btn = gr.Button("Process & Transcribe", variant="primary") | |
| with gr.Column(): | |
| output_text = gr.Textbox(label="Result Output / Markdown", lines=24) | |
| submit_btn.click( | |
| fn=gradio_handler, | |
| inputs=[mic_input, file_input, url_input], | |
| outputs=output_text | |
| ) | |
| app = gr.mount_gradio_app(app, demo, path="/") |