mralamdari's picture
Update app.py
ab97a62 verified
Raw History Blame Contribute Delete
13 kB
from smolagents import CodeAgent,DuckDuckGoSearchTool, HfApiModel,load_tool,Tool,tool, LiteLLMModel, TransformersModel
import datetime
import requests
import pytz
import yaml
import os
from tools.final_answer import FinalAnswerTool
from Gradio_UI import GradioUI
@tool
def anime_recommender(desc: str) -> str:
"""
Suggest an anime whose story is based on the description.
Args:
desc: The story of an anime.
"""
if len(desc) <= 50:
return 'One Punch Man'
elif len(desc) <= 100:
return 'Solo Leveling'
else:
return 'Anime Not Found'
class SimilarAnimeSuggester(Tool):
name = "similar_anime_name_suggester"
description = """
This tool suggest a similar anime based on the recommended anime
"""
inputs = {
'recommendation':{
'type': 'string',
"description": 'The name of an anime',
}
}
output_type = 'string'
def forward(self, recommendation: str):
if recommendation == 'One Punch Man':
return 'Mob 100'
elif recommendation == 'Solo Leveling':
return 'The Rising Of The Shield Hero'
else:
return 'This anime is unique'
final_answer = FinalAnswerTool()
from recipe_scrapers import scrape_html
from urllib.request import urlopen
from Gradio_UI import GradioUI
@tool
def food_recipe_recommender(url: str) -> str:
"""This tool look at the urls provided and returns the recipe from a wroking, existing page.
Args:
url: list of url pages to scrape the recipe from.
"""
scraper = None
try:
html = urlopen(url).read().decode("utf-8")
scraper = scrape_html(html, org_url=url)
except:
return f"Recipe not found for provided url: {url}."
ingredients = [f" - {ing}\n" for ing in scraper.ingredients()]
recipe = f"""
👩🏼‍🍳 Et voilà! The {scraper.title()}!
Preparation requires {scraper.total_time()} minutes, and it is for {scraper.yields()} people.
🧂 Ingredients:\n
{''.join(ingredients)}
🥘 Instructions:
{scraper.instructions()}
"""
return recipe
class VisitWebpageMarkdown(Tool):
name = "visit_webpage_markdown"
description = "Visits a webpage at the given url and reads its content as a markdown document via Jina. Use this to browse webpages. It skips images / medias"
inputs = {'url': {'type': 'string', 'description': 'The url of the webpage to visit.'}}
output_type = "string"
def forward(self, url: str) -> str:
import requests
try:
# Send a GET request to the URL
response = requests.get('https://r.jina.ai/' +url)
response.raise_for_status() # Raise an exception for bad status codes
markdown_content =response.text.strip()
return markdown_content
except Exception as e:
return f"An unexpected error occurred: {str(e)}"
def __init__(self, *args, **kwargs):
self.is_initialized = False
# class TravelDistanceDuration(Tool):
# name = "get_travel_duration"
# description = "Gets the travel time between two places."
# inputs = {"start_location":{"type":"string","description":"the place from which you start your ride"},"destination_location":{"type":"string","description":"the place of arrival"},"transportation_mode":{"type":"string","nullable":True,"description":"The transportation mode, in 'driving', 'walking', 'bicycling', or 'transit'. Defaults to 'driving'."}}
# output_type = "string"
# def forward(self, start_location: str, destination_location: str, transportation_mode: Optional[str] = None) -> str:
# """Gets the travel time between two places.
# Args:
# start_location: the place from which you start your ride
# destination_location: the place of arrival
# transportation_mode: The transportation mode, in 'driving', 'walking', 'bicycling', or 'transit'. Defaults to 'driving'.
# """
# import os # All imports are placed within the function, to allow for sharing to Hub.
# import googlemaps
# from datetime import datetime
# gmaps = googlemaps.Client(os.getenv("GMAPS_API_KEY"))
# if transportation_mode is None:
# transportation_mode = "driving"
# try:
# directions_result = gmaps.directions(
# start_location,
# destination_location,
# mode=transportation_mode,
# departure_time=datetime(2025, 12, 6, 11, 0), # At 11, date far in the future
# )
# if len(directions_result) == 0:
# return "No way found between these places with the required transportation mode."
# return directions_result[0]["legs"][0]["duration"]["text"]
# except Exception as e:
# print(e)
# return e
from typing import Any, Optional
class WebContentAnalyzer(Tool):
name = "web_content_analyzer"
description = "Analyzes web content using AI models."
inputs = {"url":{"type":"string","description":"The webpage URL to analyze."}}
output_type = "string"
def forward(self, url: str) -> str:
"""Analyzes web content using AI models.
Args:
url: The webpage URL to analyze.
Returns:
str: Analysis results in JSON format.
"""
import requests
from bs4 import BeautifulSoup
import re
from transformers import pipeline
import json
try:
# Fetch content
headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64)'}
response = requests.get(url, headers=headers, timeout=10)
# Parse HTML
soup = BeautifulSoup(response.text, 'html.parser')
for tag in soup(['script', 'style', 'meta']):
tag.decompose()
# Extract basic info
title = soup.title.string if soup.title else "No title found"
text = re.sub(r'\s+', ' ', soup.get_text()).strip()
if len(text) < 100:
return json.dumps({
"error": "Not enough content to analyze"
})
# Get summary
summarizer = pipeline("summarization", model="facebook/bart-large-cnn")
summary = summarizer(text[:1024], max_length=100, min_length=30)[0]['summary_text']
# Get sentiment
classifier = pipeline("text-classification",
model="nlptown/bert-base-multilingual-uncased-sentiment")
sentiment = classifier(text[:512])[0]
score = int(sentiment['label'][0])
mood = ["Very Negative", "Negative", "Neutral", "Positive", "Very Positive"][score-1]
# Format results
result = {
"title": title,
"summary": summary,
"sentiment": f"{mood} ({score}/5)",
"stats": {
"words": len(text.split()),
"chars": len(text)
}
}
return json.dumps(result)
except Exception as e:
return json.dumps({
"error": str(e)
})
"""------Applied TF-IDF for better semantic search------"""
import feedparser
import urllib.parse
import yaml
from tools.final_answer import FinalAnswerTool
import numpy as np
from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.metrics.pairwise import cosine_similarity
import gradio as gr
from smolagents import CodeAgent,DuckDuckGoSearchTool, HfApiModel,load_tool,tool
import nltk
import datetime
import requests
import pytz
from tools.final_answer import FinalAnswerTool
from Gradio_UI import GradioUI
nltk.download("stopwords")
from nltk.corpus import stopwords
@tool # ✅ Register the function properly as a SmolAgents tool
def fetch_latest_arxiv_papers(keywords: list, num_results: int = 5) -> list:
"""Fetches and ranks arXiv papers using TF-IDF and Cosine Similarity.
Args:
keywords: List of keywords for search.
num_results: Number of results to return.
Returns:
List of the most relevant papers based on TF-IDF ranking.
"""
try:
print(f"DEBUG: Searching arXiv papers with keywords: {keywords}")
# Use a general keyword search
query = "+AND+".join([f"all:{kw}" for kw in keywords])
query_encoded = urllib.parse.quote(query)
url = f"http://export.arxiv.org/api/query?search_query={query_encoded}&start=0&max_results=50&sortBy=submittedDate&sortOrder=descending"
print(f"DEBUG: Query URL - {url}")
feed = feedparser.parse(url)
papers = []
# Extract papers from arXiv
for entry in feed.entries:
papers.append({
"title": entry.title,
"authors": ", ".join(author.name for author in entry.authors),
"year": entry.published[:4],
"abstract": entry.summary,
"link": entry.link
})
if not papers:
return [{"error": "No results found. Try different keywords."}]
# Prepare TF-IDF Vectorization
corpus = [paper["title"] + " " + paper["abstract"] for paper in papers]
vectorizer = TfidfVectorizer(stop_words=stopwords.words('english')) # Remove stopwords
tfidf_matrix = vectorizer.fit_transform(corpus)
# Transform Query into TF-IDF Vector
query_str = " ".join(keywords)
query_vec = vectorizer.transform([query_str])
#Compute Cosine Similarity
similarity_scores = cosine_similarity(query_vec, tfidf_matrix).flatten()
#Sort papers based on similarity score
ranked_papers = sorted(zip(papers, similarity_scores), key=lambda x: x[1], reverse=True)
# Return the most relevant papers
return [paper[0] for paper in ranked_papers[:num_results]]
except Exception as e:
print(f"ERROR: {str(e)}")
return [{"error": f"Error fetching research papers: {str(e)}"}]
# # Create Gradio UI
# with gr.Blocks() as demo:
# gr.Markdown("# ScholarAgent")
# keyword_input = gr.Textbox(label="Enter keywords (comma-separated)", placeholder="e.g., deep learning, reinforcement learning")
# output_display = gr.Markdown()
# search_button = gr.Button("Search")
# search_button.click(search_papers, inputs=[keyword_input], outputs=[output_display])
# print("DEBUG: Gradio UI is running. Waiting for user input...")
# # Launch Gradio App
# demo.launch()
class HFModelDownloadsTool(Tool):
name = "model_download_counter"
description = """
This is a tool that returns the most downloaded model of a given task on the Hugging Face Hub.
It returns the name of the checkpoint."""
inputs = {'task': {'type': 'string', 'description': 'the task category (such as text-classification, depth-estimation, etc)'}}
output_type = "string"
def forward(self, task: str):
from huggingface_hub import list_models
model = next(iter(list_models(filter=task, sort="downloads", direction=-1)))
return model.id
def __init__(self, *args, **kwargs):
self.is_initialized = False
# hugging face is getting the hug of death so lets use litellm for now
model = HfApiModel(
max_tokens=2096,
temperature=0.5,
# model_id='Qwen/Qwen2.5-Coder-32B-Instruct',
model_id='https://pflgm2locj2t89co.us-east-1.aws.endpoints.huggingface.cloud',
custom_role_conversions=None,
)
# model = LiteLLMModel(
# model_id="gemini/gemini-2.0-flash-exp",
# max_tokens=2096,
# temperature=0.6,
# api_key=os.getenv("LITELLM_API_KEY")
# )
# ollama
# model = LiteLLMModel(
# model_id="ollama_chat/deepseek-r1:7b",
# max_tokens=2096,
# temperature=0.6,
# api_base="http://localhost:11434",
# num_ctx=8192
# )
# transformer
# model = TransformersModel(
# model_id="deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B",
# device_map="auto",
# torch_dtype="auto",
# max_new_tokens=2096,
# temperature=0.6,
# )
# Import tool from Hub
image_generation_tool = load_tool("agents-course/text-to-image", trust_remote_code=True)
with open("prompts.yaml", 'r') as stream:
prompt_templates = yaml.safe_load(stream)
agent = CodeAgent(
model=model,
tools=[final_answer,
anime_recommender,
SimilarAnimeSuggester(),
# TravelDistanceDuration(),
food_recipe_recommender,
VisitWebpageMarkdown(),
fetch_latest_arxiv_papers,
WebContentAnalyzer(),
HFModelDownloadsTool(),
], ## add your tools here (don't remove final answer)
max_steps=6,
verbosity_level=1,
grammar=None,
planning_interval=None,
name=None,
description=None,
prompt_templates=prompt_templates
)
GradioUI(agent).launch()