reranker / src /jd /parse.py
Hemprasad Badgujar
Add bm25 prefilter stage; unify candidates cache; validate reasoning quality
cd5f9d4
Raw History Blame Contribute Delete
18.7 kB
"""JD extraction β€” TWO greedy JSON calls: OBJECTIVE (filters + BM25 keywords) and SUBJECTIVE (embedding).
1. OBJECTIVE β†’ one JSON object of keyword bags + filter facts matched against the candidate ``text``.
2. SUBJECTIVE β†’ the ideal-candidate narrative (candidate-profile register) for embedding similarity.
``_extract_json_object`` pulls the first balanced ``{...}`` (models wrap JSON in prose); the ``_coerce_*``
helpers clean it. Objective keys are aligned to the candidate schema. Returns None on any failure / empty
result so the caller can degrade.
"""
from __future__ import annotations
import time
from typing import TYPE_CHECKING, Any
from common.logging import get_logger
if TYPE_CHECKING:
from models.llm import LlmClient
# Pins the persona so the extractor (Qwen3-1.7B) emits ONLY a JSON object and never slips into prose or a
# planning loop. The client appends Qwen's /no_think soft switch (gated to Qwen models in LlmClient).
_SYSTEM = (
"You are a precise information-extraction engine for recruitment job descriptions. "
"You output ONLY one valid JSON object β€” no prose, no markdown fences, no commentary. "
"You never explain, plan, apologize, or add conversational filler."
)
# CALL 1 β€” OBJECTIVE (filters + BM25). Role-Task-Context-Constraints-Output layout; KEYWORDS, not sentences.
# Substituted via str.replace (NOT str.format) so the literal JSON braces below are safe.
_OBJECTIVE_PROMPT = """# Role
Expert technical recruiter extracting structured data from the hiring JD below. Your JSON feeds a BM25
keyword search and hard filters β€” every list value must be a SHORT keyword, never a phrase or sentence.
# Task
Extract ONE JSON object with EXACTLY these keys (values must come only from THIS job description, never
copied from these field descriptions):
- must_have_skills: EVERY REQUIRED and PREFERRED/BONUS skill the JD names β€” every language, framework,
library, tool, platform, database, algorithm, technique, and evaluation metric; include BOTH the broad
concept AND each specifically-named product/library/metric. SKILL RECALL IS THE PRIORITY β€” scan the
responsibilities/narrative as well as the skills lists, not only the bullet headers; missing a named skill
is the worst error. Expand any parenthetical, comma, or slash list into separate keywords. A skill-dense JD
like this one typically names 20-30+ distinct skills across its sections β€” a short list is almost always
under-extraction, keep re-scanning the text before finishing.
- nice_to_have: EXTRA closely-RELATED skills NOT literally named but strongly implied for this role β€” common
synonyms, sub-techniques, adjacent tools a strong candidate would also list (e.g. for an embeddings /
retrieval / ranking role: RAG, semantic search, dense/sparse retrieval, ANN, HNSW, re-ranking,
learning-to-rank, recommendation systems). ONLY terms tightly adjacent to named skills; [] if none.
- education_fields: field-of-study keywords
- years_min / years_max: min/max years of experience (numbers)
- locations: EVERY city the JD names β€” primary work cities AND every OTHER city it says candidates are
"welcome" from, can "apply" from, or are otherwise acceptable; list them ALL (e.g. primary "Pune/Noida" PLUS
"Hyderabad, Mumbai, Delhi NCR welcome" β‡’ list all five)
- acceptable_countries: countries candidates are accepted from β€” INFER from the cities and phrases like
"outside <country>" (e.g. Pune/Noida, or "outside India" β‡’ ["India"]); [] only if truly remote/worldwide
- job_titles: 12-18 relevant titles the candidate may hold now or previously. START with the JD's EXACT posted
title, verbatim. THEN add the role family for the JD's field β€” the adjacent job titles that do the same core
work (for an ML/AI role these are e.g. AI Engineer, Machine Learning Engineer, Search Engineer, Ranking
Engineer, Applied Scientist, Research Scientist, Research Engineer, Data Scientist, MLOps Engineer, Software
Engineer; for a different field use THAT field's equivalent roles) β€” and ALWAYS include the role titles named
after the JD's CORE activities (e.g. a search/ranking/retrieval JD β‡’ Search Engineer, Ranking Engineer,
Information Retrieval Engineer) β€” plus seniority variants of the main role (e.g. "Senior <role>", "Staff
<role>", "Principal <role>"). Full, real role titles only β€” no bare words, NO duplicates; keep every title
on-topic for this JD's work
# Rules
- SHORT keywords in list fields β€” never phrases or sentences.
- ``nice_to_have`` is the ONLY field allowed terms not literally in the JD (still tightly adjacent). Every
other field's values must appear in, or be directly inferable from, the JD.
- Do NOT put into must_have_skills any expertise the JD explicitly REJECTS as a disqualifier (e.g. a domain it
says it does "NOT want", like computer vision / speech / robotics here).
- Output EVERY key above; [] for an empty list, a bare number for the years fields.
- No reasoning, planning, or explanation β€” output your immediate final answer.
- Return ONLY the JSON object β€” no markdown fences and no text before or after it.
# Output shape (placeholders only, NOT values to copy)
{"must_have_skills":[...],"nice_to_have":[...],"education_fields":[...],"years_min":0,"years_max":0,"locations":[...],"acceptable_countries":[...],"job_titles":[...]}
# Job Description
{jd_text}
# JSON
"""
# CALL 2 β€” SUBJECTIVE summary, in the CANDIDATE-PROFILE register (matches the embedded candidate summary).
# Only ``ideal_summary`` is needed β€” it is the text embedded and compared against candidate profiles.
_SUBJECTIVE_PROMPT = """# Role
Expert technical recruiter. The summary you write is embedded and compared by vector similarity against
CANDIDATE profile summaries β€” so it must read like a candidate profile, not a job post.
# Task
Write the ideal candidate as ONE JSON object with a single key, ideal_summary: a 3-5 sentence profile summary
in the register of a candidate profile β€” open like "<role> with N years of experience in <domains>. Core
skills: ...; Recent roles: ..." β€” factual profile prose. Never job-posting language; do NOT fabricate metrics
or numbers not in the JD.
# Rules
- No reasoning or planning β€” output your immediate final answer.
- Return ONLY the JSON object β€” no markdown fences and no text before or after it.
# Output shape
{"ideal_summary":"Senior AI engineer with 7 years of experience in search and ranking. Core skills: ..."}
# Job Description
{jd_text}
"""
def _extract_json_object(raw_text: str) -> dict[str, Any] | None:
"""Pull the first balanced ``{...}`` from model output and parse it; None if none/invalid."""
import orjson
start = raw_text.find("{")
if start < 0:
return None
depth = 0
for index in range(start, len(raw_text)):
if raw_text[index] == "{":
depth += 1
elif raw_text[index] == "}":
depth -= 1
if depth == 0:
try:
parsed = orjson.loads(raw_text[start : index + 1])
except orjson.JSONDecodeError:
return None
return parsed if isinstance(parsed, dict) else None
return None
def _as_str_list(value: object) -> list[str]:
"""Clean a JSON value to a list[str]: strip, drop empties, dedupe case-insensitively (keep first casing/order)."""
if not isinstance(value, list):
return []
out: list[str] = []
seen: set[str] = set()
for item in value:
text = str(item).strip()
key = text.lower()
if text and key not in seen:
out.append(text)
seen.add(key)
return out
def _as_years(value: object, fallback: float) -> float:
try:
return float(value) # type: ignore[arg-type]
except (TypeError, ValueError):
return fallback
# City β†’ country fallback for acceptable_countries. The OBJECTIVE prompt asks the model to INFER the country
# from the JD's cities, but the small extractor often returns [] β€” which silently disables the composite
# location hard-drop. This deterministic gazetteer recovers the signal: keyed by lowercase city (the common
# tech hubs), matched as a substring so "Pune, India" / "Noida (Remote)" still resolve. Extend as needed.
_CITY_COUNTRY: dict[str, str] = {
# India
"pune": "India", "noida": "India", "bangalore": "India", "bengaluru": "India", "hyderabad": "India",
"mumbai": "India", "new delhi": "India", "delhi": "India", "gurugram": "India", "gurgaon": "India",
"chennai": "India", "kolkata": "India", "ahmedabad": "India", "jaipur": "India", "indore": "India",
"chandigarh": "India", "kochi": "India", "coimbatore": "India", "nagpur": "India", "thiruvananthapuram": "India",
# United States
"san francisco": "United States", "new york": "United States", "seattle": "United States",
"austin": "United States", "boston": "United States", "mountain view": "United States",
# Other common hubs
"london": "United Kingdom", "manchester": "United Kingdom", "berlin": "Germany", "munich": "Germany",
"toronto": "Canada", "singapore": "Singapore", "dublin": "Ireland", "amsterdam": "Netherlands",
"sydney": "Australia",
}
def _infer_countries(locations: list[str]) -> list[str]:
"""Infer acceptable_countries from JD city names (deduped, order-stable); [] if none recognized.
Substring match on a lowercased location so "Pune, India" or "Noida (Remote)" still resolve. Used only
as a fallback when the model omits acceptable_countries β€” never overrides an explicit model value.
"""
found: list[str] = []
for location in locations:
key = str(location).lower()
for city, country in _CITY_COUNTRY.items():
if city in key and country not in found:
found.append(country)
return found
def _coerce_objective(parsed: dict[str, Any]) -> dict[str, Any] | None:
"""Clean the objective object into the JDSpec-shaped dict; None if no role AND no skills (unusable).
The prompt emits: ``must_have_skills`` (named), ``nice_to_have`` (adjacent/related), ``education_fields``,
the years, and a single ``locations`` (preferred+acceptable merged). ``target_industries`` and
``exclude_companies`` are no longer REQUESTED from the model β€” retired from the prompt to free prompt/
output budget for the two fields that matter most (skills, job_titles) β€” but both keys are still written
to the objective dict as ``[]`` (not simply omitted), so the artifact schema stays stable for any future
reader that indexes them directly rather than ``.get()``-ing; JDSpec/soft_filter/composite already treat
an empty list as a no-op for these two signals. ``certifications`` isn't requested either β†’ defaults to [].
Older keys are still read as a fallback.
"""
locations = _as_str_list(parsed.get("locations")) or _as_str_list(parsed.get("locations_preferred"))
# acceptable_countries: trust the model first, else INFER from the cities (the prompt asks for this, but
# the small extractor frequently returns [] β€” without this the composite location hard-drop is a no-op).
acceptable_countries = _as_str_list(parsed.get("acceptable_countries")) or _infer_countries(locations)
objective = {
"must_have_skills": _as_str_list(parsed.get("must_have_skills")),
"nice_to_have": _as_str_list(parsed.get("nice_to_have")), # adjacent/related skills (BM25 recall boost)
"job_titles": _as_str_list(parsed.get("job_titles")),
"target_industries": [], # retired from the prompt; kept as [] so the artifact schema doesn't shrink
"exclude_companies": [], # (soft_filter / composite BM25 already no-op gracefully on an empty list)
"education_fields": _as_str_list(parsed.get("education_fields")),
"certifications": _as_str_list(parsed.get("certifications")),
"years_min": _as_years(parsed.get("years_min"), 5.0),
"years_max": _as_years(parsed.get("years_max"), 9.0),
"locations_preferred": locations,
"locations_acceptable": _as_str_list(parsed.get("locations_acceptable")),
"acceptable_countries": acceptable_countries,
}
if not objective["job_titles"] and not objective["must_have_skills"]:
return None
return objective
def _coerce_subjective(parsed: dict[str, Any]) -> dict[str, Any] | None:
"""Clean the subjective object to just ``ideal_summary`` (the embedding text); None if empty."""
ideal_summary = str(parsed.get("ideal_summary", "")).strip()
return {"ideal_summary": ideal_summary} if ideal_summary else None
def _generate_json(
prompt: str, client: LlmClient, concern: str, max_tokens: int = 1024, temperature: float = 0.0
) -> dict[str, Any] | None:
"""One generation β†’ parsed JSON object; None on any failure. temperature=0 β‡’ greedy/deterministic."""
try:
raw_text = client.generate(prompt, system=_SYSTEM, temperature=temperature, max_tokens=max_tokens)
except Exception as failure:
get_logger("jd").warning(f"jd.{concern}.failed", reason=type(failure).__name__)
return None
return _extract_json_object(raw_text)
def _skill_count(objective: dict[str, Any]) -> int:
"""NAMED required-skill count (must_have only) β€” the recall measure best-of-N maximizes.
Excludes ``nice_to_have``: those are inferred/adjacent terms now always populated by the prompt, so
counting them would inflate the total and early-stop the best-of-N before literal must_have recall peaks.
"""
return len(objective.get("must_have_skills", []))
def _generate_objective(
prompt: str, client: LlmClient, max_tokens: int, temperature: float, retries: int, log: Any,
) -> dict[str, Any] | None:
"""One objective generate+coerce pass, retried up to ``retries`` extra times on a HARD failure (bad/
truncated JSON, an exception, or valid JSON with no titles AND no skills). A retry re-samples from the
SAME already-loaded model instance β€” its RNG state has already advanced past the failed call, so a retry
genuinely explores a different completion rather than reproducing the identical failure (confirmed:
repeated same-process calls measured real variance, e.g. a 10/32/24-skill spread across 3 trials)."""
for attempt in range(retries + 1):
parsed = _generate_json(prompt, client, "objective", max_tokens=max_tokens, temperature=temperature)
objective = _coerce_objective(parsed) if parsed is not None else None
if objective is not None:
return objective
if attempt < retries:
log.warning("jd.objective.retry", retry=attempt + 1, of=retries)
return None
def extract_objective(
jd_text: str, client: LlmClient, *,
temperature: float | None = None, attempts: int | None = None, min_skills: int | None = None,
retry_on_failure: int | None = None,
) -> dict[str, Any] | None:
"""CALL 1 β€” objective keyword bags + filter facts, with a best-of-N recall pass.
Samples the objective parse at ``temperature`` up to ``attempts`` times and KEEPS THE RICHEST result (most
skills), stopping early once ``min_skills`` is reached. A greedy temperature (0.0) collapses to a single
deterministic pass (identical samples), as does ``attempts=1``. Overrides fall back to the JD config.
Each attempt slot ALSO retries on a hard failure (``retry_on_failure`` extra tries) before giving up on
that slot β€” measured on real runs, a single attempt (temperature=0.5) fails outright (unparseable JSON /
empty) roughly 1 in 9 times, and with ``parse_attempts=1`` there was no recovery from that. No wall-clock
cap here β€” the JD spec is the single most important input to the whole funnel (every filter and retrieval
query derives from it), so this is intentionally allowed to take as long as it needs; the attempt/retry
counts are the only bound. Returns None on failure / nothing usable.
"""
if not jd_text.strip():
return None
from config import load_settings # lazy import β€” keeps this module config/backend-free at import time
jd_cfg = load_settings().jd
temperature = jd_cfg.parse_temperature if temperature is None else temperature
attempts = jd_cfg.parse_attempts if attempts is None else attempts
min_skills = jd_cfg.min_skills if min_skills is None else min_skills
retry_on_failure = jd_cfg.parse_retry_on_failure if retry_on_failure is None else retry_on_failure
attempts = max(1, attempts) if temperature > 0 else 1 # greedy β‡’ identical samples β‡’ one pass only
prompt = _OBJECTIVE_PROMPT.replace("{jd_text}", jd_text)
log = get_logger("jd")
best: dict[str, Any] | None = None
start = time.perf_counter()
for attempt in range(1, attempts + 1):
objective = _generate_objective(prompt, client, jd_cfg.parse_max_tokens, temperature, retry_on_failure, log)
if objective is None:
continue
if best is None or _skill_count(objective) > _skill_count(best):
best = objective
log.info("jd.objective.sample", attempt=attempt, attempts=attempts,
skills=_skill_count(objective), best=_skill_count(best), target=min_skills,
seconds=round(time.perf_counter() - start, 3))
if _skill_count(best) >= min_skills:
break
return best
def extract_subjective(
jd_text: str, client: LlmClient, *, retry_on_failure: int | None = None
) -> dict[str, Any] | None:
"""CALL 2 β€” subjective ideal_summary (the embedding text); None if every try fails. Short β†’ small
max_tokens. Retries up to ``retry_on_failure`` extra times on a hard failure, same as the objective call β€”
cheap insurance since this call rarely fails but a lost ideal_summary still degrades the embedding query
for the whole funnel."""
if not jd_text.strip():
return None
from config import load_settings # lazy import β€” keeps this module config/backend-free at import time
retries = load_settings().jd.parse_retry_on_failure if retry_on_failure is None else retry_on_failure
log = get_logger("jd")
prompt = _SUBJECTIVE_PROMPT.replace("{jd_text}", jd_text)
for attempt in range(retries + 1):
parsed = _generate_json(prompt, client, "subjective", max_tokens=384)
subjective = _coerce_subjective(parsed) if parsed is not None else None
if subjective is not None:
return subjective
if attempt < retries:
log.warning("jd.subjective.retry", retry=attempt + 1, of=retries)
return None