"""JD extraction — TWO greedy JSON calls: OBJECTIVE (filters + BM25 keywords) and SUBJECTIVE (embedding). 1. OBJECTIVE → one JSON object of keyword bags + filter facts matched against the candidate ``text``. 2. SUBJECTIVE → the ideal-candidate narrative (candidate-profile register) for embedding similarity. ``_extract_json_object`` pulls the first balanced ``{...}`` (models wrap JSON in prose); the ``_coerce_*`` helpers clean it. Objective keys are aligned to the candidate schema. Returns None on any failure / empty result so the caller can degrade. """ from __future__ import annotations import time from typing import TYPE_CHECKING, Any from common.logging import get_logger if TYPE_CHECKING: from models.llm import LlmClient # Pins the persona so the extractor (Qwen3-1.7B) emits ONLY a JSON object and never slips into prose or a # planning loop. The client appends Qwen's /no_think soft switch (gated to Qwen models in LlmClient). _SYSTEM = ( "You are a precise information-extraction engine for recruitment job descriptions. " "You output ONLY one valid JSON object — no prose, no markdown fences, no commentary. " "You never explain, plan, apologize, or add conversational filler." ) # CALL 1 — OBJECTIVE (filters + BM25). Role-Task-Context-Constraints-Output layout; KEYWORDS, not sentences. # Substituted via str.replace (NOT str.format) so the literal JSON braces below are safe. _OBJECTIVE_PROMPT = """# Role Expert technical recruiter extracting structured data from the hiring JD below. Your JSON feeds a BM25 keyword search and hard filters — every list value must be a SHORT keyword, never a phrase or sentence. # Task Extract ONE JSON object with EXACTLY these keys (values must come only from THIS job description, never copied from these field descriptions): - must_have_skills: EVERY REQUIRED and PREFERRED/BONUS skill the JD names — every language, framework, library, tool, platform, database, algorithm, technique, and evaluation metric; include BOTH the broad concept AND each specifically-named product/library/metric. SKILL RECALL IS THE PRIORITY — scan the responsibilities/narrative as well as the skills lists, not only the bullet headers; missing a named skill is the worst error. Expand any parenthetical, comma, or slash list into separate keywords. A skill-dense JD like this one typically names 20-30+ distinct skills across its sections — a short list is almost always under-extraction, keep re-scanning the text before finishing. - nice_to_have: EXTRA closely-RELATED skills NOT literally named but strongly implied for this role — common synonyms, sub-techniques, adjacent tools a strong candidate would also list (e.g. for an embeddings / retrieval / ranking role: RAG, semantic search, dense/sparse retrieval, ANN, HNSW, re-ranking, learning-to-rank, recommendation systems). ONLY terms tightly adjacent to named skills; [] if none. - education_fields: field-of-study keywords - years_min / years_max: min/max years of experience (numbers) - locations: EVERY city the JD names — primary work cities AND every OTHER city it says candidates are "welcome" from, can "apply" from, or are otherwise acceptable; list them ALL (e.g. primary "Pune/Noida" PLUS "Hyderabad, Mumbai, Delhi NCR welcome" ⇒ list all five) - acceptable_countries: countries candidates are accepted from — INFER from the cities and phrases like "outside " (e.g. Pune/Noida, or "outside India" ⇒ ["India"]); [] only if truly remote/worldwide - job_titles: 12-18 relevant titles the candidate may hold now or previously. START with the JD's EXACT posted title, verbatim. THEN add the role family for the JD's field — the adjacent job titles that do the same core work (for an ML/AI role these are e.g. AI Engineer, Machine Learning Engineer, Search Engineer, Ranking Engineer, Applied Scientist, Research Scientist, Research Engineer, Data Scientist, MLOps Engineer, Software Engineer; for a different field use THAT field's equivalent roles) — and ALWAYS include the role titles named after the JD's CORE activities (e.g. a search/ranking/retrieval JD ⇒ Search Engineer, Ranking Engineer, Information Retrieval Engineer) — plus seniority variants of the main role (e.g. "Senior ", "Staff ", "Principal "). Full, real role titles only — no bare words, NO duplicates; keep every title on-topic for this JD's work # Rules - SHORT keywords in list fields — never phrases or sentences. - ``nice_to_have`` is the ONLY field allowed terms not literally in the JD (still tightly adjacent). Every other field's values must appear in, or be directly inferable from, the JD. - Do NOT put into must_have_skills any expertise the JD explicitly REJECTS as a disqualifier (e.g. a domain it says it does "NOT want", like computer vision / speech / robotics here). - Output EVERY key above; [] for an empty list, a bare number for the years fields. - No reasoning, planning, or explanation — output your immediate final answer. - Return ONLY the JSON object — no markdown fences and no text before or after it. # Output shape (placeholders only, NOT values to copy) {"must_have_skills":[...],"nice_to_have":[...],"education_fields":[...],"years_min":0,"years_max":0,"locations":[...],"acceptable_countries":[...],"job_titles":[...]} # Job Description {jd_text} # JSON """ # CALL 2 — SUBJECTIVE summary, in the CANDIDATE-PROFILE register (matches the embedded candidate summary). # Only ``ideal_summary`` is needed — it is the text embedded and compared against candidate profiles. _SUBJECTIVE_PROMPT = """# Role Expert technical recruiter. The summary you write is embedded and compared by vector similarity against CANDIDATE profile summaries — so it must read like a candidate profile, not a job post. # Task Write the ideal candidate as ONE JSON object with a single key, ideal_summary: a 3-5 sentence profile summary in the register of a candidate profile — open like " with N years of experience in . Core skills: ...; Recent roles: ..." — factual profile prose. Never job-posting language; do NOT fabricate metrics or numbers not in the JD. # Rules - No reasoning or planning — output your immediate final answer. - Return ONLY the JSON object — no markdown fences and no text before or after it. # Output shape {"ideal_summary":"Senior AI engineer with 7 years of experience in search and ranking. Core skills: ..."} # Job Description {jd_text} """ def _extract_json_object(raw_text: str) -> dict[str, Any] | None: """Pull the first balanced ``{...}`` from model output and parse it; None if none/invalid.""" import orjson start = raw_text.find("{") if start < 0: return None depth = 0 for index in range(start, len(raw_text)): if raw_text[index] == "{": depth += 1 elif raw_text[index] == "}": depth -= 1 if depth == 0: try: parsed = orjson.loads(raw_text[start : index + 1]) except orjson.JSONDecodeError: return None return parsed if isinstance(parsed, dict) else None return None def _as_str_list(value: object) -> list[str]: """Clean a JSON value to a list[str]: strip, drop empties, dedupe case-insensitively (keep first casing/order).""" if not isinstance(value, list): return [] out: list[str] = [] seen: set[str] = set() for item in value: text = str(item).strip() key = text.lower() if text and key not in seen: out.append(text) seen.add(key) return out def _as_years(value: object, fallback: float) -> float: try: return float(value) # type: ignore[arg-type] except (TypeError, ValueError): return fallback # City → country fallback for acceptable_countries. The OBJECTIVE prompt asks the model to INFER the country # from the JD's cities, but the small extractor often returns [] — which silently disables the composite # location hard-drop. This deterministic gazetteer recovers the signal: keyed by lowercase city (the common # tech hubs), matched as a substring so "Pune, India" / "Noida (Remote)" still resolve. Extend as needed. _CITY_COUNTRY: dict[str, str] = { # India "pune": "India", "noida": "India", "bangalore": "India", "bengaluru": "India", "hyderabad": "India", "mumbai": "India", "new delhi": "India", "delhi": "India", "gurugram": "India", "gurgaon": "India", "chennai": "India", "kolkata": "India", "ahmedabad": "India", "jaipur": "India", "indore": "India", "chandigarh": "India", "kochi": "India", "coimbatore": "India", "nagpur": "India", "thiruvananthapuram": "India", # United States "san francisco": "United States", "new york": "United States", "seattle": "United States", "austin": "United States", "boston": "United States", "mountain view": "United States", # Other common hubs "london": "United Kingdom", "manchester": "United Kingdom", "berlin": "Germany", "munich": "Germany", "toronto": "Canada", "singapore": "Singapore", "dublin": "Ireland", "amsterdam": "Netherlands", "sydney": "Australia", } def _infer_countries(locations: list[str]) -> list[str]: """Infer acceptable_countries from JD city names (deduped, order-stable); [] if none recognized. Substring match on a lowercased location so "Pune, India" or "Noida (Remote)" still resolve. Used only as a fallback when the model omits acceptable_countries — never overrides an explicit model value. """ found: list[str] = [] for location in locations: key = str(location).lower() for city, country in _CITY_COUNTRY.items(): if city in key and country not in found: found.append(country) return found def _coerce_objective(parsed: dict[str, Any]) -> dict[str, Any] | None: """Clean the objective object into the JDSpec-shaped dict; None if no role AND no skills (unusable). The prompt emits: ``must_have_skills`` (named), ``nice_to_have`` (adjacent/related), ``education_fields``, the years, and a single ``locations`` (preferred+acceptable merged). ``target_industries`` and ``exclude_companies`` are no longer REQUESTED from the model — retired from the prompt to free prompt/ output budget for the two fields that matter most (skills, job_titles) — but both keys are still written to the objective dict as ``[]`` (not simply omitted), so the artifact schema stays stable for any future reader that indexes them directly rather than ``.get()``-ing; JDSpec/soft_filter/composite already treat an empty list as a no-op for these two signals. ``certifications`` isn't requested either → defaults to []. Older keys are still read as a fallback. """ locations = _as_str_list(parsed.get("locations")) or _as_str_list(parsed.get("locations_preferred")) # acceptable_countries: trust the model first, else INFER from the cities (the prompt asks for this, but # the small extractor frequently returns [] — without this the composite location hard-drop is a no-op). acceptable_countries = _as_str_list(parsed.get("acceptable_countries")) or _infer_countries(locations) objective = { "must_have_skills": _as_str_list(parsed.get("must_have_skills")), "nice_to_have": _as_str_list(parsed.get("nice_to_have")), # adjacent/related skills (BM25 recall boost) "job_titles": _as_str_list(parsed.get("job_titles")), "target_industries": [], # retired from the prompt; kept as [] so the artifact schema doesn't shrink "exclude_companies": [], # (soft_filter / composite BM25 already no-op gracefully on an empty list) "education_fields": _as_str_list(parsed.get("education_fields")), "certifications": _as_str_list(parsed.get("certifications")), "years_min": _as_years(parsed.get("years_min"), 5.0), "years_max": _as_years(parsed.get("years_max"), 9.0), "locations_preferred": locations, "locations_acceptable": _as_str_list(parsed.get("locations_acceptable")), "acceptable_countries": acceptable_countries, } if not objective["job_titles"] and not objective["must_have_skills"]: return None return objective def _coerce_subjective(parsed: dict[str, Any]) -> dict[str, Any] | None: """Clean the subjective object to just ``ideal_summary`` (the embedding text); None if empty.""" ideal_summary = str(parsed.get("ideal_summary", "")).strip() return {"ideal_summary": ideal_summary} if ideal_summary else None def _generate_json( prompt: str, client: LlmClient, concern: str, max_tokens: int = 1024, temperature: float = 0.0 ) -> dict[str, Any] | None: """One generation → parsed JSON object; None on any failure. temperature=0 ⇒ greedy/deterministic.""" try: raw_text = client.generate(prompt, system=_SYSTEM, temperature=temperature, max_tokens=max_tokens) except Exception as failure: get_logger("jd").warning(f"jd.{concern}.failed", reason=type(failure).__name__) return None return _extract_json_object(raw_text) def _skill_count(objective: dict[str, Any]) -> int: """NAMED required-skill count (must_have only) — the recall measure best-of-N maximizes. Excludes ``nice_to_have``: those are inferred/adjacent terms now always populated by the prompt, so counting them would inflate the total and early-stop the best-of-N before literal must_have recall peaks. """ return len(objective.get("must_have_skills", [])) def _generate_objective( prompt: str, client: LlmClient, max_tokens: int, temperature: float, retries: int, log: Any, ) -> dict[str, Any] | None: """One objective generate+coerce pass, retried up to ``retries`` extra times on a HARD failure (bad/ truncated JSON, an exception, or valid JSON with no titles AND no skills). A retry re-samples from the SAME already-loaded model instance — its RNG state has already advanced past the failed call, so a retry genuinely explores a different completion rather than reproducing the identical failure (confirmed: repeated same-process calls measured real variance, e.g. a 10/32/24-skill spread across 3 trials).""" for attempt in range(retries + 1): parsed = _generate_json(prompt, client, "objective", max_tokens=max_tokens, temperature=temperature) objective = _coerce_objective(parsed) if parsed is not None else None if objective is not None: return objective if attempt < retries: log.warning("jd.objective.retry", retry=attempt + 1, of=retries) return None def extract_objective( jd_text: str, client: LlmClient, *, temperature: float | None = None, attempts: int | None = None, min_skills: int | None = None, retry_on_failure: int | None = None, ) -> dict[str, Any] | None: """CALL 1 — objective keyword bags + filter facts, with a best-of-N recall pass. Samples the objective parse at ``temperature`` up to ``attempts`` times and KEEPS THE RICHEST result (most skills), stopping early once ``min_skills`` is reached. A greedy temperature (0.0) collapses to a single deterministic pass (identical samples), as does ``attempts=1``. Overrides fall back to the JD config. Each attempt slot ALSO retries on a hard failure (``retry_on_failure`` extra tries) before giving up on that slot — measured on real runs, a single attempt (temperature=0.5) fails outright (unparseable JSON / empty) roughly 1 in 9 times, and with ``parse_attempts=1`` there was no recovery from that. No wall-clock cap here — the JD spec is the single most important input to the whole funnel (every filter and retrieval query derives from it), so this is intentionally allowed to take as long as it needs; the attempt/retry counts are the only bound. Returns None on failure / nothing usable. """ if not jd_text.strip(): return None from config import load_settings # lazy import — keeps this module config/backend-free at import time jd_cfg = load_settings().jd temperature = jd_cfg.parse_temperature if temperature is None else temperature attempts = jd_cfg.parse_attempts if attempts is None else attempts min_skills = jd_cfg.min_skills if min_skills is None else min_skills retry_on_failure = jd_cfg.parse_retry_on_failure if retry_on_failure is None else retry_on_failure attempts = max(1, attempts) if temperature > 0 else 1 # greedy ⇒ identical samples ⇒ one pass only prompt = _OBJECTIVE_PROMPT.replace("{jd_text}", jd_text) log = get_logger("jd") best: dict[str, Any] | None = None start = time.perf_counter() for attempt in range(1, attempts + 1): objective = _generate_objective(prompt, client, jd_cfg.parse_max_tokens, temperature, retry_on_failure, log) if objective is None: continue if best is None or _skill_count(objective) > _skill_count(best): best = objective log.info("jd.objective.sample", attempt=attempt, attempts=attempts, skills=_skill_count(objective), best=_skill_count(best), target=min_skills, seconds=round(time.perf_counter() - start, 3)) if _skill_count(best) >= min_skills: break return best def extract_subjective( jd_text: str, client: LlmClient, *, retry_on_failure: int | None = None ) -> dict[str, Any] | None: """CALL 2 — subjective ideal_summary (the embedding text); None if every try fails. Short → small max_tokens. Retries up to ``retry_on_failure`` extra times on a hard failure, same as the objective call — cheap insurance since this call rarely fails but a lost ideal_summary still degrades the embedding query for the whole funnel.""" if not jd_text.strip(): return None from config import load_settings # lazy import — keeps this module config/backend-free at import time retries = load_settings().jd.parse_retry_on_failure if retry_on_failure is None else retry_on_failure log = get_logger("jd") prompt = _SUBJECTIVE_PROMPT.replace("{jd_text}", jd_text) for attempt in range(retries + 1): parsed = _generate_json(prompt, client, "subjective", max_tokens=384) subjective = _coerce_subjective(parsed) if parsed is not None else None if subjective is not None: return subjective if attempt < retries: log.warning("jd.subjective.retry", retry=attempt + 1, of=retries) return None