ResumeBot / app.py
sudeep1610's picture
ResumeBot: model, code and docs
3d00929 verified
Raw History Blame Contribute Delete
6.79 kB
"""ResumeBot: a FLAN-T5 model fine-tuned on Sudeep's resume, with retrieval and hallucination guards.
Run: pip install -r requirements.txt && python app.py (then open the local link it prints)"""
import os, re, json, inspect
import numpy as np, torch, gradio as gr
from huggingface_hub import hf_hub_download
from sentence_transformers import SentenceTransformer
from transformers import AutoTokenizer, AutoModelForSeq2SeqLM
MODEL_REPO = "sudeep1610/ResumeBot" # the Hugging Face repo with the fine-tuned model
device = "cuda" if torch.cuda.is_available() else "cpu"
# ---- knowledge base (resume facts, profile, prompt, known questions) ----
cfg_path = "bot_config.json" if os.path.exists("bot_config.json") else hf_hub_download(MODEL_REPO, "bot_config.json")
cfg = json.load(open(cfg_path))
FACTS = [tuple(f) for f in cfg["facts"]]
PROFILE, PROMPT = cfg["profile"], cfg["prompt"]
# ---- models ----
tokenizer = AutoTokenizer.from_pretrained(MODEL_REPO)
model = AutoModelForSeq2SeqLM.from_pretrained(MODEL_REPO).to(device).eval()
embedder = SentenceTransformer("sentence-transformers/all-MiniLM-L6-v2", device=device)
bank, bank_fid = cfg["bank"], cfg["bank_fid"]
bank_vecs = embedder.encode(bank, normalize_embeddings=True, convert_to_numpy=True)
def retrieve(question, k=2):
sims = bank_vecs @ embedder.encode([question], normalize_embeddings=True, convert_to_numpy=True)[0]
best = {}
for i in np.argsort(-sims):
best.setdefault(bank_fid[i], float(sims[i]))
if len(best) == k: break
fids = list(best)
return fids, best[fids[0]]
# ---- answer function with hallucination guards ----
PERSON_WORDS = {"sudeep", "candidate", "he", "his", "him", "you", "your", "resume", "cv", "profile"}
STOP = set("a an the is are was were of in on at to for and or with by from as it this that his he sudeep s".split())
GREETING = "Hi! πŸ‘‹ I'm ResumeBot. Ask me anything about Sudeep: his education, skills, projects, certifications or achievements."
OUT_OF_SCOPE = ("I only answer questions about Sudeep's resume, so I won't guess about that. "
"Try asking about his education, skills, projects, certifications or achievements.")
def words(text):
return {w for w in re.findall(r"[a-z0-9]+", text.lower()) if w not in STOP}
def generate(prompt):
inputs = tokenizer(prompt, return_tensors="pt", truncation=True, max_length=256).to(model.device)
with torch.no_grad():
out = model.generate(**inputs, max_new_tokens=128, num_beams=4, no_repeat_ngram_size=3, early_stopping=True)
return tokenizer.decode(out[0], skip_special_tokens=True).strip()
def answer(question):
q = question.strip()
ql = re.findall(r"[a-z]+", q.lower())
if not q or set(ql) <= {"hi", "hello", "hey", "hii", "good", "morning", "evening", "namaste"}:
return GREETING, "Greeting"
if set(ql) & {"thanks", "thank", "thankyou"}:
return "You're welcome! Ask me anything else about Sudeep. 😊", "Greeting"
fids, score = retrieve(q)
about_him = bool(PERSON_WORDS & set(ql))
if score < (0.35 if about_him else 0.55): # guard 1: not about the resume -> don't guess
return OUT_OF_SCOPE, "Out of scope"
context = " ".join(FACTS[f][1] for f in fids)
reply = generate(PROMPT.format(context=context, question=q))
grounded = len(words(reply) - words(context)) <= 1 and len(reply) > 3
if not grounded: # guard 2: answer must come from the resume text
reply = FACTS[fids[0]][1]
return reply, FACTS[fids[0]][0]
# ---- Gradio website ----
CSS = """
.gradio-container {max-width: 1100px !important; margin: auto;}
#hero {background: linear-gradient(135deg,#4f46e5 0%,#7c3aed 55%,#db2777 100%); border-radius: 18px; padding: 28px 32px; color: white;}
#hero h1 {margin: 0; font-size: 34px; color: white;}
#hero p {margin: 6px 0 0; opacity: .92; font-size: 16px; color: white;}
.card {border: 1px solid var(--border-color-primary); border-radius: 16px; padding: 18px 20px; background: var(--block-background-fill);}
.card h3 {margin: 0 0 8px; font-size: 15px; text-transform: uppercase; letter-spacing: .06em; opacity: .7;}
.chip {display: inline-block; padding: 4px 10px; margin: 3px 4px 3px 0; border-radius: 999px; font-size: 13px;
background: rgba(99,102,241,.12); border: 1px solid rgba(99,102,241,.35);}
.meta {font-size: 14px; line-height: 1.8;}
footer {display: none !important;}
"""
P = PROFILE
HERO = f"""<div id="hero"><h1>πŸ€– ResumeBot</h1>
<p>Chat with my personal AI, a FLAN-T5 model fine-tuned on my resume. It answers only from verified resume facts.</p></div>"""
PROFILE_CARD = f"""<div class="card"><h3>Profile</h3>
<div style="font-size:24px;font-weight:700">{P['name']}</div><div style="opacity:.8;margin-bottom:10px">{P['role']}</div>
<div class="meta">πŸ“ {P['location']}<br>πŸŽ“ {P['education']}<br>βœ‰οΈ {P['email']}<br>
πŸ”— <a href="https://{P['linkedin']}" target="_blank">LinkedIn</a> Β· <a href="https://{P['github']}" target="_blank">GitHub</a></div></div>
<div class="card" style="margin-top:12px"><h3>Skills</h3>{''.join(f'<span class="chip">{s}</span>' for s in P['skills'])}</div>
<div class="card" style="margin-top:12px"><h3>How it works</h3><div class="meta">
1️⃣ Finds the matching resume section<br>2️⃣ Fine-tuned FLAN-T5 writes the answer<br>3️⃣ Answer is checked against the resume</div></div>"""
def chat(message, history):
reply, source = answer(message)
if source in ("Greeting",):
return reply
return f"{reply}\n\n<sub>πŸ“„ Resume section: {source}</sub>" if source != "Out of scope" else f"{reply}\n\n<sub>πŸ›‘οΈ Out of scope</sub>"
theme = gr.themes.Soft(primary_hue="indigo", secondary_hue="violet", font=[gr.themes.GoogleFont("Inter"), "sans-serif"])
style = dict(theme=theme, css=CSS)
new_gradio = "css" in inspect.signature(gr.Blocks.launch).parameters # Gradio 6 moved theme/css to launch()
with gr.Blocks(title="ResumeBot | Sudeep", **({} if new_gradio else style)) as demo:
gr.HTML(HERO)
with gr.Row(equal_height=False):
with gr.Column(scale=1, min_width=280):
gr.HTML(PROFILE_CARD)
with gr.Column(scale=2):
gr.ChatInterface(
fn=chat,
chatbot=gr.Chatbot(height=460, show_label=False, placeholder="Ask me about Sudeep πŸ‘‡"),
textbox=gr.Textbox(placeholder="e.g. What projects has Sudeep built?", show_label=False, scale=7),
examples=["Tell me about Sudeep", "What are his technical skills?", "What is his best project?",
"Which certifications does he have?", "Why should we hire him?", "Has he done any internships?"],
)
demo.launch(**(style if new_gradio else {}))