annator / src /agents /layout_analysis_agent.py
techprotrade's picture
Add src directory
734b5b4 verified
Raw
History Blame Contribute Delete
1.65 kB
from PIL import Image
from typing import Dict, Any
from .base_agent import LLMAgent, AgentResponse
class LayoutAnalysisAgent(LLMAgent):
name = "layout_analysis_agent"
description = "Analyzes document layout structure - identifies columns, headers, footers, margins, and reading order"
def build_prompt(self, pil_image: Image.Image) -> str:
width, height = pil_image.size
return f"""Analyze the layout of this PDF page (dimensions: {width}x{height}).
Describe the layout structure:
1. Is it single or multi-column?
2. Where are the margins and gutters?
3. Are there headers, footers, or page numbers?
4. What is the reading order of content blocks?
5. Are there any sidebars, footnotes, or margin notes?
6. What type of document is this (letter, report, form, etc.)?
Provide specific coordinates for key layout elements if visible."""
def execute(self, pil_image: Image.Image, reasoning: str) -> Dict[str, Any]:
width, height = pil_image.size
return {
"layout_type": self._classify_layout(reasoning),
"width": width,
"height": height,
"blocks": self._extract_blocks(reasoning),
"reading_order": self._extract_reading_order(reasoning),
"reasoning": reasoning,
}
def _classify_layout(self, reasoning: str) -> str:
reasoning_lower = reasoning.lower()
if "column" in reasoning_lower:
return "multi_column"
return "single_column"
def _extract_blocks(self, reasoning: str) -> list:
return []
def _extract_reading_order(self, reasoning: str) -> list:
return []