reranker / src /preprocess /quality.py
Hemprasad Badgujar
Add CPU threading, T1-only LLM polish, experiment script
a41b023
Raw History Blame Contribute Delete
8.45 kB
"""Preprocess sub-phase — quality pillars: 7 query-independent sub-scores + blended candidate_quality.
A candidate-only prior (no JD dependency): each pillar is a capped-linear blend of materialized features,
and ``candidate_quality`` is their weighted sum (weights from central config). Adds the seniority / degree /
institution-tier ranks the pillars need. Never drops a row.
"""
from __future__ import annotations
import polars as pl
from common.logging import step
from config import QualitySettings, load_settings
# Ordinal ranks (0 = no signal). Higher = more senior / more prestigious / higher degree.
_SENIORITY_RANK = {"unknown": 0, "ic": 1, "lead": 2, "manager": 3, "director": 4, "vp": 5, "c_level": 6}
_TIER_RANK = {"unknown": 0, "tier_4": 1, "tier_3": 2, "tier_2": 3, "tier_1": 4}
_DEGREE_RANK = {"unknown": 0, "none": 0, "diploma": 1, "bachelor": 2, "master": 3, "phd": 4}
_PILLAR_COLUMNS = [
"score_skills", "score_experience", "score_reliability", "score_reputation",
"score_engagement", "score_verification", "score_education", "candidate_quality",
]
def _cap01(expr: pl.Expr, cap: float) -> pl.Expr:
"""Capped linear norm to [0, 1]: ``expr / cap`` clipped."""
return (expr / cap).clip(0.0, 1.0)
def _rank_label_exprs() -> list[pl.Expr]:
"""Derive the seniority / degree / institution-tier LABELS the pillars rank off."""
title = pl.col("current_title").cast(pl.String).fill_null("").str.to_lowercase()
seniority = (
pl.when(title.str.len_chars() == 0).then(pl.lit("unknown"))
.when(title.str.contains(r"\b(?:chief|ceo|cto|cpo|cfo|coo|founder)\b")).then(pl.lit("c_level"))
.when(title.str.contains(r"\b(?:vp|vice president)\b")).then(pl.lit("vp"))
.when(title.str.contains(r"\bdirector\b")).then(pl.lit("director"))
.when(title.str.contains(r"\b(?:manager|head of)\b")).then(pl.lit("manager"))
.when(title.str.contains(r"\b(?:lead|principal|staff)\b")).then(pl.lit("lead"))
.otherwise(pl.lit("ic"))
.alias("seniority_level")
)
degree = pl.element().struct.field("degree").cast(pl.String).str.to_lowercase()
highest_degree = pl.col("education").list.eval(
pl.when(degree.str.contains("phd|doctor")).then(pl.lit("phd"))
.when(degree.str.contains("master|msc|mba")).then(pl.lit("master"))
.when(degree.str.contains("bachelor|btech|bsc|be |ba ")).then(pl.lit("bachelor"))
.when(degree.str.contains("diploma")).then(pl.lit("diploma"))
.otherwise(pl.lit("unknown"))
).list.max().fill_null("unknown").alias("highest_degree_level")
best_tier = (
pl.col("flat_institution_tiers").list.eval(pl.element().filter(pl.element() != ""))
.list.max().fill_null("unknown").alias("best_institution_tier")
)
return [seniority, highest_degree, best_tier]
def _rank_code_exprs() -> list[pl.Expr]:
"""Map the labels to their ordinal ranks (the numeric inputs to the experience/education pillars)."""
return [
pl.col("seniority_level").replace_strict(_SENIORITY_RANK, default=0, return_dtype=pl.Int32).alias("seniority_rank"),
pl.col("highest_degree_level").replace_strict(_DEGREE_RANK, default=0, return_dtype=pl.Int32).alias("highest_degree_rank"),
pl.col("best_institution_tier").replace_strict(_TIER_RANK, default=0, return_dtype=pl.Int32).alias("best_institution_tier_rank"),
]
def _pillar_exprs(weights: QualitySettings) -> list[pl.Expr]:
"""The 7 capped-linear pillars + their weighted blend ``candidate_quality`` (clipped to [0, 1])."""
skills = (
_cap01(pl.col("num_advanced_skills").cast(pl.Float64), 8) * 0.25
+ _cap01(pl.col("num_expert_skills").cast(pl.Float64), 5) * 0.20
+ _cap01(pl.col("avg_skill_assessment").fill_null(0), 100) * 0.30
+ _cap01(pl.col("assessment_coverage").fill_null(0), 1) * 0.25
)
experience = (
_cap01(pl.col("years_of_experience"), 15) * 0.35
+ _cap01(pl.col("seniority_rank").cast(pl.Float64), 6) * 0.25
+ _cap01(pl.col("avg_tenure_months"), 48) * 0.20
+ _cap01(pl.col("num_industries").cast(pl.Float64), 5) * 0.20
)
reliability = (
_cap01(pl.col("interview_completion_rate").fill_null(0), 1) * 0.50
+ _cap01(pl.col("offer_acceptance_rate_clean").fill_null(0), 1) * 0.35
+ pl.col("has_offer_history").cast(pl.Float64) * 0.15
)
reputation = (
_cap01(pl.col("endorsements_received").cast(pl.Float64), 200) * 0.20
+ _cap01(pl.col("connection_count").cast(pl.Float64), 500) * 0.15
+ _cap01(pl.col("saved_by_recruiters_30d").cast(pl.Float64), 20) * 0.20
+ _cap01(pl.col("search_appearance_30d").cast(pl.Float64), 50) * 0.15
+ _cap01(pl.col("github_score_clean").fill_null(0), 100) * 0.20
+ pl.col("has_github").cast(pl.Float64) * 0.10
)
engagement = (
(1.0 - _cap01(pl.col("days_since_active").cast(pl.Float64), 180)) * 0.40
+ _cap01(pl.col("recruiter_response_rate").fill_null(0), 1) * 0.35
+ (1.0 - _cap01(pl.col("avg_response_time_hours").fill_null(72), 72)) * 0.25
)
verification = (
_cap01(pl.col("verification_count").cast(pl.Float64), 3) * 0.50
+ _cap01(pl.col("profile_completeness_score"), 100) * 0.50
)
education = (
_cap01(pl.col("best_institution_tier_rank").cast(pl.Float64), 4) * 0.55
+ _cap01(pl.col("highest_degree_rank").cast(pl.Float64), 4) * 0.45
)
candidate_quality = (
weights.skills * skills + weights.experience * experience + weights.reliability * reliability
+ weights.reputation * reputation + weights.engagement * engagement
+ weights.verification * verification + weights.education * education
).clip(0.0, 1.0)
return [
skills.alias("score_skills"), experience.alias("score_experience"),
reliability.alias("score_reliability"), reputation.alias("score_reputation"),
engagement.alias("score_engagement"), verification.alias("score_verification"),
education.alias("score_education"), candidate_quality.alias("candidate_quality"),
]
# Pillar → human label for the strengths clause folded into the embedded text (computed parameters → words the
# granite embedder can use). A pillar is named only when it's a genuine strength (score above the threshold).
_PILLAR_LABELS = {
"score_skills": "skills", "score_experience": "experience", "score_reliability": "reliability",
"score_reputation": "reputation", "score_engagement": "engagement",
"score_verification": "verification", "score_education": "education",
}
_STRENGTH_THRESHOLD = 0.6
def _strengths_clause() -> pl.Expr:
"""' Strengths: <strong pillars>.' from the pillars (empty when none clear the threshold)."""
labels = pl.concat_list([
pl.when(pl.col(column) >= _STRENGTH_THRESHOLD).then(pl.lit(label)).otherwise(None)
for column, label in _PILLAR_LABELS.items()
]).list.drop_nulls().list.join(", ")
return pl.when(labels.str.len_chars() > 0).then(pl.lit(" Strengths: ") + labels + pl.lit(".")).otherwise(pl.lit(""))
def quality(candidates: pl.DataFrame, weights: QualitySettings | None = None) -> pl.DataFrame:
"""Add the seniority/degree/tier ranks, the 7 quality pillars, ``candidate_quality``, and fold a strengths
clause into the embedded ``text``. Never drops."""
pillar_weights = weights or load_settings().quality
with step("preprocess.quality", rows_in=candidates.height) as metrics:
out = (
candidates.with_columns(_rank_label_exprs())
.with_columns(_rank_code_exprs())
.with_columns(_pillar_exprs(pillar_weights))
.with_columns([pl.col(column).round(3) for column in _PILLAR_COLUMNS])
)
if "text" in out.columns: # enrich the granite-embedded body with the computed pillar strengths
out = out.with_columns((pl.col("text") + _strengths_clause()).str.strip_chars().alias("text"))
if "embed_source_hash" in out.columns: # re-hash: materialize.py's hash predates this text change
from .materialize import _sha16
out = out.with_columns(pl.Series("embed_source_hash", _sha16(out.get_column("text").to_list())))
metrics["rows_out"] = out.height
metrics["mean_quality"] = round(float(out.select(pl.col("candidate_quality").mean()).item() or 0), 4)
return out