import easyocr import os # Initialize the OCR reader once to save time # 'en' is for English language # Set both model storage and user network directories inside /app (writable) MODELS_DIR = "/app/easyocr_models" USER_NETWORK_DIR = "/app/easyocr_models/user_network" # Ensure directories exist os.makedirs(MODELS_DIR, exist_ok=True) os.makedirs(USER_NETWORK_DIR, exist_ok=True) # DEBUG: Verify the directories print(f"[DEBUG] EasyOCR models directory: {MODELS_DIR}") print(f"[DEBUG] EasyOCR user network directory: {USER_NETWORK_DIR}") # Initialize EasyOCR reader with both directories specified reader = easyocr.Reader( ['en'], model_storage_directory=MODELS_DIR, user_network_directory=USER_NETWORK_DIR ) def extract_keywords_from_report(file_path): """ Performs OCR on the uploaded file and extracts relevant text. """ try: results = reader.readtext(file_path, detail=0) full_text = " ".join(results).lower() return full_text except Exception as e: print(f"OCR Error: {e}") return "" def score_text_for_risk(text): """ Scores the extracted text and lists the keywords found. """ high_risk_keywords = [ "nodule", "abnormal cell", "squamous", "carcinoma", "malignant", "adenocarcinoma", "biopsy positive", "tumor", "mass" ] score = 0 keywords_found = [] for keyword in high_risk_keywords: if keyword in text: score += 0.1 keywords_found.append(keyword.title()) return min(score, 1.0), keywords_found # Example test if __name__ == '__main__': test_file_path = 'test_report.png' if os.path.exists(test_file_path): extracted_text = extract_keywords_from_report(test_file_path) risk_score = score_text_for_risk(extracted_text) print("--- OCR Test Results ---") print(f"Extracted Text: {extracted_text}") print(f"Calculated Risk Score: {risk_score}") else: print("Error: test_report.png not found. Cannot run direct test.")