"""ResumeBot: a FLAN-T5 model fine-tuned on Sudeep's resume, with retrieval and hallucination guards. Run: pip install -r requirements.txt && python app.py (then open the local link it prints)""" import os, re, json, inspect import numpy as np, torch, gradio as gr from huggingface_hub import hf_hub_download from sentence_transformers import SentenceTransformer from transformers import AutoTokenizer, AutoModelForSeq2SeqLM MODEL_REPO = "sudeep1610/ResumeBot" # the Hugging Face repo with the fine-tuned model device = "cuda" if torch.cuda.is_available() else "cpu" # ---- knowledge base (resume facts, profile, prompt, known questions) ---- cfg_path = "bot_config.json" if os.path.exists("bot_config.json") else hf_hub_download(MODEL_REPO, "bot_config.json") cfg = json.load(open(cfg_path)) FACTS = [tuple(f) for f in cfg["facts"]] PROFILE, PROMPT = cfg["profile"], cfg["prompt"] # ---- models ---- tokenizer = AutoTokenizer.from_pretrained(MODEL_REPO) model = AutoModelForSeq2SeqLM.from_pretrained(MODEL_REPO).to(device).eval() embedder = SentenceTransformer("sentence-transformers/all-MiniLM-L6-v2", device=device) bank, bank_fid = cfg["bank"], cfg["bank_fid"] bank_vecs = embedder.encode(bank, normalize_embeddings=True, convert_to_numpy=True) def retrieve(question, k=2): sims = bank_vecs @ embedder.encode([question], normalize_embeddings=True, convert_to_numpy=True)[0] best = {} for i in np.argsort(-sims): best.setdefault(bank_fid[i], float(sims[i])) if len(best) == k: break fids = list(best) return fids, best[fids[0]] # ---- answer function with hallucination guards ---- PERSON_WORDS = {"sudeep", "candidate", "he", "his", "him", "you", "your", "resume", "cv", "profile"} STOP = set("a an the is are was were of in on at to for and or with by from as it this that his he sudeep s".split()) GREETING = "Hi! ๐ I'm ResumeBot. Ask me anything about Sudeep: his education, skills, projects, certifications or achievements." OUT_OF_SCOPE = ("I only answer questions about Sudeep's resume, so I won't guess about that. " "Try asking about his education, skills, projects, certifications or achievements.") def words(text): return {w for w in re.findall(r"[a-z0-9]+", text.lower()) if w not in STOP} def generate(prompt): inputs = tokenizer(prompt, return_tensors="pt", truncation=True, max_length=256).to(model.device) with torch.no_grad(): out = model.generate(**inputs, max_new_tokens=128, num_beams=4, no_repeat_ngram_size=3, early_stopping=True) return tokenizer.decode(out[0], skip_special_tokens=True).strip() def answer(question): q = question.strip() ql = re.findall(r"[a-z]+", q.lower()) if not q or set(ql) <= {"hi", "hello", "hey", "hii", "good", "morning", "evening", "namaste"}: return GREETING, "Greeting" if set(ql) & {"thanks", "thank", "thankyou"}: return "You're welcome! Ask me anything else about Sudeep. ๐", "Greeting" fids, score = retrieve(q) about_him = bool(PERSON_WORDS & set(ql)) if score < (0.35 if about_him else 0.55): # guard 1: not about the resume -> don't guess return OUT_OF_SCOPE, "Out of scope" context = " ".join(FACTS[f][1] for f in fids) reply = generate(PROMPT.format(context=context, question=q)) grounded = len(words(reply) - words(context)) <= 1 and len(reply) > 3 if not grounded: # guard 2: answer must come from the resume text reply = FACTS[fids[0]][1] return reply, FACTS[fids[0]][0] # ---- Gradio website ---- CSS = """ .gradio-container {max-width: 1100px !important; margin: auto;} #hero {background: linear-gradient(135deg,#4f46e5 0%,#7c3aed 55%,#db2777 100%); border-radius: 18px; padding: 28px 32px; color: white;} #hero h1 {margin: 0; font-size: 34px; color: white;} #hero p {margin: 6px 0 0; opacity: .92; font-size: 16px; color: white;} .card {border: 1px solid var(--border-color-primary); border-radius: 16px; padding: 18px 20px; background: var(--block-background-fill);} .card h3 {margin: 0 0 8px; font-size: 15px; text-transform: uppercase; letter-spacing: .06em; opacity: .7;} .chip {display: inline-block; padding: 4px 10px; margin: 3px 4px 3px 0; border-radius: 999px; font-size: 13px; background: rgba(99,102,241,.12); border: 1px solid rgba(99,102,241,.35);} .meta {font-size: 14px; line-height: 1.8;} footer {display: none !important;} """ P = PROFILE HERO = f"""
Chat with my personal AI, a FLAN-T5 model fine-tuned on my resume. It answers only from verified resume facts.