File size: 2,919 Bytes
b20b1d8
 
 
6aae614
8fe992b
b20b1d8
9b5b26a
b20b1d8
 
9b5b26a
b20b1d8
9b5b26a
 
b20b1d8
 
 
 
9b5b26a
b20b1d8
ae7a494
6cacefe
 
 
 
 
 
 
673bdc8
 
 
 
 
 
 
 
 
ed9ac50
6cacefe
 
b20b1d8
e121372
6cacefe
b20b1d8
 
 
13d500a
8c01ffb
b20b1d8
 
8c01ffb
6cacefe
8c01ffb
8fe992b
6cacefe
8c01ffb
 
 
 
6cacefe
 
b20b1d8
8fe992b
 
6cacefe
 
 
 
 
 
 
 
b20b1d8
 
 
6cacefe
 
b20b1d8
 
6cacefe
b20b1d8
6cacefe
b20b1d8
6cacefe
 
b20b1d8
 
 
 
9b5b26a
6cacefe
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
import PyPDF2
import gradio as gr
from smolagents import CodeAgent, HfApiModel, tool
from tools.final_answer import FinalAnswerTool

# Define the PDF text extraction tool
@tool
def extract_text_from_pdf(pdf_path: str) -> str:
    """Extracts text from a given PDF file.
    Args:
        pdf_path: The path to the PDF file.
    """
    try:
        with open(pdf_path, "rb") as file:
            reader = PyPDF2.PdfReader(file)
            text = "\n".join([page.extract_text() for page in reader.pages if page.extract_text()])
        return text if text else "No text found in the PDF."
    except Exception as e:
        return f"Error extracting text from PDF: {str(e)}"

# Define the PDF summarization tool
@tool
def summarize_text(text: str) -> str:
    """Summarizes the extracted text using the AI model.
    Args:
        text: The extracted text from the PDF.
    """
    max_input_tokens = 15000  # ✅ Adjust input limit to stay within model constraints
    
    # Truncate text if it's too long
    truncated_text = text[:max_input_tokens]

    prompt = f"Summarize the following document:\n\n{truncated_text}"
    
    # ✅ Correct input format for HfApiModel
    response = model.__call__([{"role": "user", "content": prompt}])

    return response

# Initialize AI Model
model = HfApiModel(
    max_tokens=1024,  # Reduced token limit for summaries
    temperature=0.5,
    model_id='Qwen/Qwen2.5-Coder-32B-Instruct',
    custom_role_conversions=None,
)

# Initialize Final Answer Tool
final_answer = FinalAnswerTool()

# Create AI Agent with PDF extraction and summarization tools
agent = CodeAgent(
    model=model,
    tools=[final_answer, extract_text_from_pdf, summarize_text],  # Added summarization tool
    max_steps=6,
    verbosity_level=1,
    grammar=None,
    planning_interval=None,
    name="PDF Summary Agent",
    description="An AI agent that extracts and summarizes text from PDFs.",
    prompt_templates=None
)

# Define Gradio function to handle PDF upload and summarization
def extract_and_summarize_pdf(file_path):
    extracted_text = extract_text_from_pdf(file_path)
    if "Error" in extracted_text:
        return extracted_text, "No summary available."
    
    summary = summarize_text(extracted_text)
    return extracted_text, summary

# Gradio UI Setup
with gr.Blocks() as ui:
    gr.Markdown("# 📄 PDF Summarizer Agent")
    gr.Markdown("### Upload a PDF file, and the agent will extract and summarize its content.")

    with gr.Row():
        pdf_input = gr.File(type="filepath", label="Upload PDF")  # ✅ Fixed type
        output_text = gr.Textbox(label="Extracted Text", interactive=False)
        summary_output = gr.Textbox(label="Summary", interactive=False)

    extract_button = gr.Button("Extract & Summarize")
    extract_button.click(extract_and_summarize_pdf, inputs=pdf_input, outputs=[output_text, summary_output])

# Launch Gradio UI
ui.launch()