KokTong's picture
Update agent.py
4b0f0d6 verified
Raw
History Blame Contribute Delete
4.65 kB
from smolagents import (CodeAgent,
GradioUI,
LiteLLMModel,
OpenAIServerModel,
ChatMessage,
ToolCallingAgent)
from smolagents.default_tools import (DuckDuckGoSearchTool,
VisitWebpageTool,
WikipediaSearchTool,
PythonInterpreterTool
)
import yaml
from tools import check_reasoning, ensure_formatting
from tools import (download_youtube_video, youtube_frames_to_images,
use_vision_model, analyze_frames_with_vision,
read_file, download_file_from_url, analyze_csv_file,
analyze_excel_file, youtube_transcribe,
transcribe_audio)
import os
from dotenv import load_dotenv
import time
load_dotenv()
# Load prompts from YAML file
with open("prompts.yaml", 'r') as stream:
prompt_templates = yaml.safe_load(stream)
class SlowLiteLLMModel(LiteLLMModel):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
def __call__(self, messages, **kwargs) -> ChatMessage:
time.sleep(15)
# prepend onto whatever messages the Agent built
return super().__call__(messages, **kwargs)
react_model_name = "gemini/gemini-2.5-flash-lite"
react_model = LiteLLMModel(model_id=react_model_name,
api_key=os.getenv("GEMINI_KEY"),
temperature=0.2
)
search_tool = DuckDuckGoSearchTool()
search_tool.name = "search"
visit_tool = VisitWebpageTool(max_output_length=500000)
visit_tool.name = "open_page"
wiki_tool = WikipediaSearchTool(extract_format='HTML')
wiki_tool.name = "wiki_search"
research_agent = ToolCallingAgent(
name="ResearchAgent",
description="""
Handles web search and information retrieval.
Tools:
- search(query: str): Search the web
- open_page(url: str): Open a webpage
- wiki_search(query: str): Search Wikipedia
""",
model=react_model,
tools=[search_tool, visit_tool, wiki_tool]
)
vision_tool = use_vision_model
vision_tool.name = "analyze_image"
vision_agent = ToolCallingAgent(
name="VisionAgent",
description="""
Handles image understanding and OCR.
Tools:
- extract_text(image_path: str): Extract text from image
- analyze_image(image_path: str): Analyze image content
""",
model=react_model,
tools=[vision_tool]
)
csv_tool = analyze_csv_file
csv_tool.name = "analyze_csv"
excel_tool = analyze_excel_file
excel_tool.name = "analyze_excel"
data_agent = ToolCallingAgent(
name="DataAgent",
description="""
Handles CSV and Excel analysis.
Tools:
- analyze_csv(file_path: str): Analyze CSV file
- analyze_excel(file_path: str): Analyze Excel file
""",
model=react_model,
tools=[csv_tool, excel_tool]
)
yt_transcribe_tool = youtube_transcribe
yt_transcribe_tool.name = "youtube_transcribe"
yt_frames_tool = youtube_frames_to_images
yt_frames_tool.name = "youtube_frames"
audio_tool = transcribe_audio
audio_tool.name = "transcribe_audio"
analyze_frame = analyze_frames_with_vision
analyze_frame.name = "analyze_frames"
media_agent = ToolCallingAgent(
name="MediaAgent",
description="""
Handles YouTube and audio processing.
Tools:
- youtube_transcribe(url: str): Get transcript from YouTube
- youtube_frames(url: str): Extract frames from downloaded video
- transcribe_audio(file_path: str): Convert audio to text
- analyze_frames(url: str): Analyze frames from video
""",
model=react_model,
tools=[
yt_transcribe_tool,
yt_frames_tool,
audio_tool,
analyze_frame,
]
)
manager_agent = CodeAgent(
model=react_model,
tools=[
read_file, download_file_from_url, download_youtube_video
],
managed_agents=[
research_agent,
vision_agent,
data_agent,
media_agent
],
additional_authorized_imports=['os', 'pandas', 'numpy', 'PIL', 'tempfile', 'PIL.Image'],
max_steps=20,
verbosity_level=1,
planning_interval=6,
name="Manager",
description="The manager of the team, responsible for overseeing and guiding the team's work. Routes tasks to the right specialist agents",
#final_answer_checks=[check_reasoning, ensure_formatting],
prompt_templates=prompt_templates
)
if __name__ == "__main__":
GradioUI(manager_agent).launch()