| from PIL import Image |
| from typing import Dict, Any |
| from .base_agent import LLMAgent |
|
|
|
|
| class ImageProcessingAgent(LLMAgent): |
| name = "image_processing_agent" |
| description = "Enhances and processes images within PDF documents - handles resolution, contrast, noise reduction" |
|
|
| def build_prompt(self, pil_image: Image.Image) -> str: |
| width, height = pil_image.size |
| return f"""Analyze this image from a PDF ({width}x{height} pixels). |
| |
| Assess: |
| 1. Image quality - is it clear, blurry, or noisy? |
| 2. Contrast - good or poor? |
| 3. Resolution - adequate for the page size? |
| 4. Any distortion or artifacts? |
| 5. Are there multiple images or a single hero image? |
| 6. Should this image be enhanced or is it already optimal? |
| |
| Recommend processing steps if needed.""" |
|
|
| def execute(self, pil_image: Image.Image, reasoning: str) -> Dict[str, Any]: |
| return { |
| "processed": True, |
| "image_count": 1, |
| "quality": self._assess_quality(reasoning), |
| "needs_enhancement": "blur" in reasoning.lower() |
| or "noise" in reasoning.lower(), |
| "reasoning": reasoning, |
| } |
|
|
| def _assess_quality(self, reasoning: str) -> str: |
| reasoning_lower = reasoning.lower() |
| if "blur" in reasoning_lower or "noise" in reasoning_lower: |
| return "poor" |
| if "clear" in reasoning_lower or "sharp" in reasoning_lower: |
| return "high" |
| return "medium" |
|
|