markdown / app.py
gfathertech's picture
Update app.py
e2a54d1 verified
Raw History Blame Contribute Delete
11.1 kB
import os
import shutil
import subprocess
import traceback
import wave
from uuid import uuid4
from fastapi import FastAPI, UploadFile, HTTPException
from markitdown import MarkItDown
import gradio as gr
from PIL import Image
import pytesseract
from pdf2image import convert_from_path
import riva.client as rcli
app = FastAPI(title="Free MarkItDown & Real-Time Voice Processing")
md = MarkItDown()
IMAGE_EXTENSIONS = {".png", ".jpg", ".jpeg", ".bmp", ".tiff", ".webp"}
AUDIO_EXTENSIONS = {".mp3", ".wav", ".m4a", ".ogg", ".flac", ".opus"}
LEGACY_OFFICE_EXTENSIONS = {".doc", ".xls", ".ppt", ".rtf"}
# NVIDIA Riva gRPC Configuration Endpoint
NVIDIA_STAGE_SERVER = "grpc.nvcf.nvidia.com:443"
NVIDIA_PARAKEET_FUNCTION_ID = "b0e8b4a5-217c-40b7-9b96-17d84e666317"
def get_nvidia_api_keys():
"""Dynamically collects NVIDIA API keys from environment variables."""
keys = []
# Check for numbered environment keys: NVIDIA_API_KEY_1, NVIDIA_API_KEY_2, etc.
index = 1
while True:
key = os.getenv(f"NVIDIA_API_KEY_{index}")
if key and key.strip():
keys.append(key.strip())
index += 1
else:
break
# Fallback to single NVIDIA_API_KEY if numbered keys aren't set
if not keys:
single_key = os.getenv("NVIDIA_API_KEY")
if single_key and single_key.strip():
keys.append(single_key.strip())
return keys
def transcribe_with_nvidia_riva(audio_path: str) -> str:
"""Transcribes audio using NVIDIA's gRPC client with multi-key failover."""
api_keys = get_nvidia_api_keys()
if not api_keys:
return "Audio Error: No NVIDIA API keys (NVIDIA_API_KEY_1, NVIDIA_API_KEY_2...) set in environment variables."
converted_wav = f"/tmp/{uuid4()}.wav"
# 1. Convert any input format (MP3, OPUS, uploaded WAV, recorded WAV)
# into strictly 16kHz, Mono, 16-bit PCM WAV using FFmpeg
try:
subprocess.run(
[
"ffmpeg", "-y", "-i", audio_path,
"-ac", "1",
"-ar", "16000",
"-sample_fmt", "s16",
converted_wav
],
check=True,
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL
)
except Exception as e:
return f"Audio Pre-processing Error (FFmpeg): {str(e)}"
last_error = ""
audio_data = None
# 2. Extract ONLY raw PCM frames, stripping the 44-byte WAV header
try:
with wave.open(converted_wav, "rb") as wf:
audio_data = wf.readframes(wf.getnframes())
except Exception as e:
last_error = f"WAV Parse Error: {str(e)}"
try:
if audio_data:
# 3. Rotate through API keys upon rate-limiting or errors
for idx, key in enumerate(api_keys, start=1):
try:
auth = rcli.Auth(
use_ssl=True,
uri=NVIDIA_STAGE_SERVER,
metadata_args=[
("authorization", f"Bearer {key}"),
("function-id", NVIDIA_PARAKEET_FUNCTION_ID)
]
)
asr_service = rcli.ASRService(auth)
config = rcli.RecognitionConfig(
encoding=rcli.AudioEncoding.LINEAR_PCM,
sample_rate_hertz=16000,
language_code="en-US",
max_alternatives=1,
audio_channel_count=1
)
response = asr_service.offline_recognize(audio_data, config)
results = response.results
transcripts = [res.alternatives[0].transcript for res in results if res.alternatives]
text = " ".join(transcripts).strip()
return f"## Audio Transcription\n\n{text}" if text else "Audio processed, but no speech was detected."
except Exception as e:
last_error = str(e)
print(f"NVIDIA API Key #{idx} failed: {last_error}. Trying next key...")
continue
return f"Audio Transcription Error: All provided NVIDIA API keys failed. Last error: {last_error}"
finally:
if os.path.exists(converted_wav):
os.remove(converted_wav)
def run_free_ocr(file_path: str) -> str:
"""Fallback OCR pipeline using Tesseract (images and PDFs only)."""
file_ext = os.path.splitext(file_path)[1].lower()
if file_ext in IMAGE_EXTENSIONS:
img = Image.open(file_path)
return pytesseract.image_to_string(img)
elif file_ext == ".pdf":
pages = convert_from_path(file_path)
extracted_text = []
for i, page in enumerate(pages):
text = pytesseract.image_to_string(page)
if text.strip():
extracted_text.append(f"## Page {i + 1}\n\n{text}")
return "\n\n".join(extracted_text)
return ""
def convert_legacy_office_file(file_path: str, ext: str) -> str:
"""Converts legacy .doc, .xls, .ppt to modern XML format via LibreOffice."""
out_dir = "/tmp"
target_ext = "docx" if ext == ".doc" else "xlsx" if ext == ".xls" else "pptx"
subprocess.run(
["libreoffice", "--headless", "--convert-to", target_ext, file_path, "--outdir", out_dir],
check=True
)
base_name = os.path.splitext(os.path.basename(file_path))[0]
return os.path.join(out_dir, f"{base_name}.{target_ext}")
def process_input(source_path_or_url: str, original_filename: str = None) -> str:
if not source_path_or_url or not source_path_or_url.strip():
return "Please supply a valid file or URL."
target = source_path_or_url.strip()
if target.startswith("http://") or target.startswith("https://"):
try:
res = md.convert(target)
return res.text_content if res else "No text extracted from URL."
except Exception as e:
return f"URL Conversion Error: {str(e)}"
ext = os.path.splitext(original_filename or target)[1].lower()
working_path = target
temp_with_ext = None
if ext and not target.lower().endswith(ext):
temp_with_ext = f"{target}{ext}"
shutil.copyfile(target, temp_with_ext)
working_path = temp_with_ext
temp_converted_office = None
try:
# Route 1: Audio files (Uploaded or Recorded) -> NVIDIA Riva gRPC
if ext in AUDIO_EXTENSIONS:
return transcribe_with_nvidia_riva(working_path)
# Route 2: Images -> Tesseract OCR
if ext in IMAGE_EXTENSIONS:
return run_free_ocr(working_path)
# Route 3: Legacy Office (.doc, .xls) -> LibreOffice
if ext in LEGACY_OFFICE_EXTENSIONS:
temp_converted_office = convert_legacy_office_file(working_path, ext)
working_path = temp_converted_office
# Route 4: Standard MarkItDown extraction (PDFs, DOCX, XLSX, PPTX, HTML, TXT)
result = md.convert(working_path)
text_content = result.text_content if result else ""
if not text_content.strip() and ext == ".pdf":
text_content = run_free_ocr(working_path)
return text_content if text_content.strip() else "Unable to extract text from document."
except Exception as e:
if ext in IMAGE_EXTENSIONS or ext == ".pdf":
try:
ocr_text = run_free_ocr(working_path)
if ocr_text and ocr_text.strip():
return ocr_text
except Exception:
pass
print("--- CONVERSION ERROR ---")
traceback.print_exc()
return f"Conversion Error: File type '{ext}' could not be processed."
finally:
if temp_with_ext and os.path.exists(temp_with_ext):
os.remove(temp_with_ext)
if temp_converted_office and os.path.exists(temp_converted_office):
os.remove(temp_converted_office)
# -------------------------------------------------------------
# FastAPI Endpoint
# -------------------------------------------------------------
@app.post("/convert")
async def convert_file_api(file: UploadFile = None, url: str = None):
if not file and not url:
raise HTTPException(status_code=400, detail="Provide either a 'file' upload or a 'url' parameter.")
if url:
markdown_text = process_input(url)
return {"source": url, "markdown": markdown_text}
file_ext = os.path.splitext(file.filename)[1].lower()
temp_path = f"/tmp/{uuid4()}{file_ext}"
try:
with open(temp_path, "wb") as buffer:
shutil.copyfileobj(file.file, buffer)
markdown_text = process_input(temp_path, original_filename=file.filename)
return {"filename": file.filename, "markdown": markdown_text}
except Exception as e:
raise HTTPException(status_code=500, detail=f"Conversion error: {str(e)}")
finally:
if os.path.exists(temp_path):
os.remove(temp_path)
# -------------------------------------------------------------
# Gradio UI with Real-time Recording & File Handling
# -------------------------------------------------------------
def gradio_handler(mic_file_path, uploaded_file_obj, url_input):
"""
Evaluates inputs with priority given to live microphone recordings.
"""
if mic_file_path is not None:
return process_input(mic_file_path, original_filename="recorded_audio.wav")
elif uploaded_file_obj is not None:
orig_name = getattr(uploaded_file_obj, "orig_name", uploaded_file_obj.name)
return process_input(uploaded_file_obj.name, original_filename=orig_name)
elif url_input and url_input.strip():
return process_input(url_input)
return "Please record voice, upload a file, or provide a URL."
with gr.Blocks(title="Free Converter & Voice Transcriber") as demo:
gr.Markdown("# 📄 Document & Real-Time Voice Transcriber")
with gr.Row():
with gr.Column():
gr.Markdown("### Option 1: Record Live Voice")
mic_input = gr.Audio(
sources=["microphone"],
type="filepath",
label="Click microphone to start recording"
)
gr.Markdown("---")
gr.Markdown("### Option 2: Upload File or URL")
file_input = gr.File(label="Upload File (.pdf, .doc, .mp3, .wav, .png, etc.)")
url_input = gr.Textbox(label="Or enter URL", placeholder="https://example.com/document.pdf")
submit_btn = gr.Button("Process & Transcribe", variant="primary")
with gr.Column():
output_text = gr.Textbox(label="Result Output / Markdown", lines=24)
submit_btn.click(
fn=gradio_handler,
inputs=[mic_input, file_input, url_input],
outputs=output_text
)
app = gr.mount_gradio_app(app, demo, path="/")