"""Preprocess sub-phase — quality pillars: 7 query-independent sub-scores + blended candidate_quality. A candidate-only prior (no JD dependency): each pillar is a capped-linear blend of materialized features, and ``candidate_quality`` is their weighted sum (weights from central config). Adds the seniority / degree / institution-tier ranks the pillars need. Never drops a row. """ from __future__ import annotations import polars as pl from common.logging import step from config import QualitySettings, load_settings # Ordinal ranks (0 = no signal). Higher = more senior / more prestigious / higher degree. _SENIORITY_RANK = {"unknown": 0, "ic": 1, "lead": 2, "manager": 3, "director": 4, "vp": 5, "c_level": 6} _TIER_RANK = {"unknown": 0, "tier_4": 1, "tier_3": 2, "tier_2": 3, "tier_1": 4} _DEGREE_RANK = {"unknown": 0, "none": 0, "diploma": 1, "bachelor": 2, "master": 3, "phd": 4} _PILLAR_COLUMNS = [ "score_skills", "score_experience", "score_reliability", "score_reputation", "score_engagement", "score_verification", "score_education", "candidate_quality", ] def _cap01(expr: pl.Expr, cap: float) -> pl.Expr: """Capped linear norm to [0, 1]: ``expr / cap`` clipped.""" return (expr / cap).clip(0.0, 1.0) def _rank_label_exprs() -> list[pl.Expr]: """Derive the seniority / degree / institution-tier LABELS the pillars rank off.""" title = pl.col("current_title").cast(pl.String).fill_null("").str.to_lowercase() seniority = ( pl.when(title.str.len_chars() == 0).then(pl.lit("unknown")) .when(title.str.contains(r"\b(?:chief|ceo|cto|cpo|cfo|coo|founder)\b")).then(pl.lit("c_level")) .when(title.str.contains(r"\b(?:vp|vice president)\b")).then(pl.lit("vp")) .when(title.str.contains(r"\bdirector\b")).then(pl.lit("director")) .when(title.str.contains(r"\b(?:manager|head of)\b")).then(pl.lit("manager")) .when(title.str.contains(r"\b(?:lead|principal|staff)\b")).then(pl.lit("lead")) .otherwise(pl.lit("ic")) .alias("seniority_level") ) degree = pl.element().struct.field("degree").cast(pl.String).str.to_lowercase() highest_degree = pl.col("education").list.eval( pl.when(degree.str.contains("phd|doctor")).then(pl.lit("phd")) .when(degree.str.contains("master|msc|mba")).then(pl.lit("master")) .when(degree.str.contains("bachelor|btech|bsc|be |ba ")).then(pl.lit("bachelor")) .when(degree.str.contains("diploma")).then(pl.lit("diploma")) .otherwise(pl.lit("unknown")) ).list.max().fill_null("unknown").alias("highest_degree_level") best_tier = ( pl.col("flat_institution_tiers").list.eval(pl.element().filter(pl.element() != "")) .list.max().fill_null("unknown").alias("best_institution_tier") ) return [seniority, highest_degree, best_tier] def _rank_code_exprs() -> list[pl.Expr]: """Map the labels to their ordinal ranks (the numeric inputs to the experience/education pillars).""" return [ pl.col("seniority_level").replace_strict(_SENIORITY_RANK, default=0, return_dtype=pl.Int32).alias("seniority_rank"), pl.col("highest_degree_level").replace_strict(_DEGREE_RANK, default=0, return_dtype=pl.Int32).alias("highest_degree_rank"), pl.col("best_institution_tier").replace_strict(_TIER_RANK, default=0, return_dtype=pl.Int32).alias("best_institution_tier_rank"), ] def _pillar_exprs(weights: QualitySettings) -> list[pl.Expr]: """The 7 capped-linear pillars + their weighted blend ``candidate_quality`` (clipped to [0, 1]).""" skills = ( _cap01(pl.col("num_advanced_skills").cast(pl.Float64), 8) * 0.25 + _cap01(pl.col("num_expert_skills").cast(pl.Float64), 5) * 0.20 + _cap01(pl.col("avg_skill_assessment").fill_null(0), 100) * 0.30 + _cap01(pl.col("assessment_coverage").fill_null(0), 1) * 0.25 ) experience = ( _cap01(pl.col("years_of_experience"), 15) * 0.35 + _cap01(pl.col("seniority_rank").cast(pl.Float64), 6) * 0.25 + _cap01(pl.col("avg_tenure_months"), 48) * 0.20 + _cap01(pl.col("num_industries").cast(pl.Float64), 5) * 0.20 ) reliability = ( _cap01(pl.col("interview_completion_rate").fill_null(0), 1) * 0.50 + _cap01(pl.col("offer_acceptance_rate_clean").fill_null(0), 1) * 0.35 + pl.col("has_offer_history").cast(pl.Float64) * 0.15 ) reputation = ( _cap01(pl.col("endorsements_received").cast(pl.Float64), 200) * 0.20 + _cap01(pl.col("connection_count").cast(pl.Float64), 500) * 0.15 + _cap01(pl.col("saved_by_recruiters_30d").cast(pl.Float64), 20) * 0.20 + _cap01(pl.col("search_appearance_30d").cast(pl.Float64), 50) * 0.15 + _cap01(pl.col("github_score_clean").fill_null(0), 100) * 0.20 + pl.col("has_github").cast(pl.Float64) * 0.10 ) engagement = ( (1.0 - _cap01(pl.col("days_since_active").cast(pl.Float64), 180)) * 0.40 + _cap01(pl.col("recruiter_response_rate").fill_null(0), 1) * 0.35 + (1.0 - _cap01(pl.col("avg_response_time_hours").fill_null(72), 72)) * 0.25 ) verification = ( _cap01(pl.col("verification_count").cast(pl.Float64), 3) * 0.50 + _cap01(pl.col("profile_completeness_score"), 100) * 0.50 ) education = ( _cap01(pl.col("best_institution_tier_rank").cast(pl.Float64), 4) * 0.55 + _cap01(pl.col("highest_degree_rank").cast(pl.Float64), 4) * 0.45 ) candidate_quality = ( weights.skills * skills + weights.experience * experience + weights.reliability * reliability + weights.reputation * reputation + weights.engagement * engagement + weights.verification * verification + weights.education * education ).clip(0.0, 1.0) return [ skills.alias("score_skills"), experience.alias("score_experience"), reliability.alias("score_reliability"), reputation.alias("score_reputation"), engagement.alias("score_engagement"), verification.alias("score_verification"), education.alias("score_education"), candidate_quality.alias("candidate_quality"), ] # Pillar → human label for the strengths clause folded into the embedded text (computed parameters → words the # granite embedder can use). A pillar is named only when it's a genuine strength (score above the threshold). _PILLAR_LABELS = { "score_skills": "skills", "score_experience": "experience", "score_reliability": "reliability", "score_reputation": "reputation", "score_engagement": "engagement", "score_verification": "verification", "score_education": "education", } _STRENGTH_THRESHOLD = 0.6 def _strengths_clause() -> pl.Expr: """' Strengths: .' from the pillars (empty when none clear the threshold).""" labels = pl.concat_list([ pl.when(pl.col(column) >= _STRENGTH_THRESHOLD).then(pl.lit(label)).otherwise(None) for column, label in _PILLAR_LABELS.items() ]).list.drop_nulls().list.join(", ") return pl.when(labels.str.len_chars() > 0).then(pl.lit(" Strengths: ") + labels + pl.lit(".")).otherwise(pl.lit("")) def quality(candidates: pl.DataFrame, weights: QualitySettings | None = None) -> pl.DataFrame: """Add the seniority/degree/tier ranks, the 7 quality pillars, ``candidate_quality``, and fold a strengths clause into the embedded ``text``. Never drops.""" pillar_weights = weights or load_settings().quality with step("preprocess.quality", rows_in=candidates.height) as metrics: out = ( candidates.with_columns(_rank_label_exprs()) .with_columns(_rank_code_exprs()) .with_columns(_pillar_exprs(pillar_weights)) .with_columns([pl.col(column).round(3) for column in _PILLAR_COLUMNS]) ) if "text" in out.columns: # enrich the granite-embedded body with the computed pillar strengths out = out.with_columns((pl.col("text") + _strengths_clause()).str.strip_chars().alias("text")) if "embed_source_hash" in out.columns: # re-hash: materialize.py's hash predates this text change from .materialize import _sha16 out = out.with_columns(pl.Series("embed_source_hash", _sha16(out.get_column("text").to_list()))) metrics["rows_out"] = out.height metrics["mean_quality"] = round(float(out.select(pl.col("candidate_quality").mean()).item() or 0), 4) return out