Spaces:
Sleeping
Sleeping
File size: 2,919 Bytes
b20b1d8 6aae614 8fe992b b20b1d8 9b5b26a b20b1d8 9b5b26a b20b1d8 9b5b26a b20b1d8 9b5b26a b20b1d8 ae7a494 6cacefe 673bdc8 ed9ac50 6cacefe b20b1d8 e121372 6cacefe b20b1d8 13d500a 8c01ffb b20b1d8 8c01ffb 6cacefe 8c01ffb 8fe992b 6cacefe 8c01ffb 6cacefe b20b1d8 8fe992b 6cacefe b20b1d8 6cacefe b20b1d8 6cacefe b20b1d8 6cacefe b20b1d8 6cacefe b20b1d8 9b5b26a 6cacefe | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 | import PyPDF2
import gradio as gr
from smolagents import CodeAgent, HfApiModel, tool
from tools.final_answer import FinalAnswerTool
# Define the PDF text extraction tool
@tool
def extract_text_from_pdf(pdf_path: str) -> str:
"""Extracts text from a given PDF file.
Args:
pdf_path: The path to the PDF file.
"""
try:
with open(pdf_path, "rb") as file:
reader = PyPDF2.PdfReader(file)
text = "\n".join([page.extract_text() for page in reader.pages if page.extract_text()])
return text if text else "No text found in the PDF."
except Exception as e:
return f"Error extracting text from PDF: {str(e)}"
# Define the PDF summarization tool
@tool
def summarize_text(text: str) -> str:
"""Summarizes the extracted text using the AI model.
Args:
text: The extracted text from the PDF.
"""
max_input_tokens = 15000 # ✅ Adjust input limit to stay within model constraints
# Truncate text if it's too long
truncated_text = text[:max_input_tokens]
prompt = f"Summarize the following document:\n\n{truncated_text}"
# ✅ Correct input format for HfApiModel
response = model.__call__([{"role": "user", "content": prompt}])
return response
# Initialize AI Model
model = HfApiModel(
max_tokens=1024, # Reduced token limit for summaries
temperature=0.5,
model_id='Qwen/Qwen2.5-Coder-32B-Instruct',
custom_role_conversions=None,
)
# Initialize Final Answer Tool
final_answer = FinalAnswerTool()
# Create AI Agent with PDF extraction and summarization tools
agent = CodeAgent(
model=model,
tools=[final_answer, extract_text_from_pdf, summarize_text], # Added summarization tool
max_steps=6,
verbosity_level=1,
grammar=None,
planning_interval=None,
name="PDF Summary Agent",
description="An AI agent that extracts and summarizes text from PDFs.",
prompt_templates=None
)
# Define Gradio function to handle PDF upload and summarization
def extract_and_summarize_pdf(file_path):
extracted_text = extract_text_from_pdf(file_path)
if "Error" in extracted_text:
return extracted_text, "No summary available."
summary = summarize_text(extracted_text)
return extracted_text, summary
# Gradio UI Setup
with gr.Blocks() as ui:
gr.Markdown("# 📄 PDF Summarizer Agent")
gr.Markdown("### Upload a PDF file, and the agent will extract and summarize its content.")
with gr.Row():
pdf_input = gr.File(type="filepath", label="Upload PDF") # ✅ Fixed type
output_text = gr.Textbox(label="Extracted Text", interactive=False)
summary_output = gr.Textbox(label="Summary", interactive=False)
extract_button = gr.Button("Extract & Summarize")
extract_button.click(extract_and_summarize_pdf, inputs=pdf_input, outputs=[output_text, summary_output])
# Launch Gradio UI
ui.launch()
|