File size: 8,446 Bytes
700e896 fb57bfa 700e896 fb57bfa 700e896 fb57bfa a41b023 700e896 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 | """Preprocess sub-phase — quality pillars: 7 query-independent sub-scores + blended candidate_quality.
A candidate-only prior (no JD dependency): each pillar is a capped-linear blend of materialized features,
and ``candidate_quality`` is their weighted sum (weights from central config). Adds the seniority / degree /
institution-tier ranks the pillars need. Never drops a row.
"""
from __future__ import annotations
import polars as pl
from common.logging import step
from config import QualitySettings, load_settings
# Ordinal ranks (0 = no signal). Higher = more senior / more prestigious / higher degree.
_SENIORITY_RANK = {"unknown": 0, "ic": 1, "lead": 2, "manager": 3, "director": 4, "vp": 5, "c_level": 6}
_TIER_RANK = {"unknown": 0, "tier_4": 1, "tier_3": 2, "tier_2": 3, "tier_1": 4}
_DEGREE_RANK = {"unknown": 0, "none": 0, "diploma": 1, "bachelor": 2, "master": 3, "phd": 4}
_PILLAR_COLUMNS = [
"score_skills", "score_experience", "score_reliability", "score_reputation",
"score_engagement", "score_verification", "score_education", "candidate_quality",
]
def _cap01(expr: pl.Expr, cap: float) -> pl.Expr:
"""Capped linear norm to [0, 1]: ``expr / cap`` clipped."""
return (expr / cap).clip(0.0, 1.0)
def _rank_label_exprs() -> list[pl.Expr]:
"""Derive the seniority / degree / institution-tier LABELS the pillars rank off."""
title = pl.col("current_title").cast(pl.String).fill_null("").str.to_lowercase()
seniority = (
pl.when(title.str.len_chars() == 0).then(pl.lit("unknown"))
.when(title.str.contains(r"\b(?:chief|ceo|cto|cpo|cfo|coo|founder)\b")).then(pl.lit("c_level"))
.when(title.str.contains(r"\b(?:vp|vice president)\b")).then(pl.lit("vp"))
.when(title.str.contains(r"\bdirector\b")).then(pl.lit("director"))
.when(title.str.contains(r"\b(?:manager|head of)\b")).then(pl.lit("manager"))
.when(title.str.contains(r"\b(?:lead|principal|staff)\b")).then(pl.lit("lead"))
.otherwise(pl.lit("ic"))
.alias("seniority_level")
)
degree = pl.element().struct.field("degree").cast(pl.String).str.to_lowercase()
highest_degree = pl.col("education").list.eval(
pl.when(degree.str.contains("phd|doctor")).then(pl.lit("phd"))
.when(degree.str.contains("master|msc|mba")).then(pl.lit("master"))
.when(degree.str.contains("bachelor|btech|bsc|be |ba ")).then(pl.lit("bachelor"))
.when(degree.str.contains("diploma")).then(pl.lit("diploma"))
.otherwise(pl.lit("unknown"))
).list.max().fill_null("unknown").alias("highest_degree_level")
best_tier = (
pl.col("flat_institution_tiers").list.eval(pl.element().filter(pl.element() != ""))
.list.max().fill_null("unknown").alias("best_institution_tier")
)
return [seniority, highest_degree, best_tier]
def _rank_code_exprs() -> list[pl.Expr]:
"""Map the labels to their ordinal ranks (the numeric inputs to the experience/education pillars)."""
return [
pl.col("seniority_level").replace_strict(_SENIORITY_RANK, default=0, return_dtype=pl.Int32).alias("seniority_rank"),
pl.col("highest_degree_level").replace_strict(_DEGREE_RANK, default=0, return_dtype=pl.Int32).alias("highest_degree_rank"),
pl.col("best_institution_tier").replace_strict(_TIER_RANK, default=0, return_dtype=pl.Int32).alias("best_institution_tier_rank"),
]
def _pillar_exprs(weights: QualitySettings) -> list[pl.Expr]:
"""The 7 capped-linear pillars + their weighted blend ``candidate_quality`` (clipped to [0, 1])."""
skills = (
_cap01(pl.col("num_advanced_skills").cast(pl.Float64), 8) * 0.25
+ _cap01(pl.col("num_expert_skills").cast(pl.Float64), 5) * 0.20
+ _cap01(pl.col("avg_skill_assessment").fill_null(0), 100) * 0.30
+ _cap01(pl.col("assessment_coverage").fill_null(0), 1) * 0.25
)
experience = (
_cap01(pl.col("years_of_experience"), 15) * 0.35
+ _cap01(pl.col("seniority_rank").cast(pl.Float64), 6) * 0.25
+ _cap01(pl.col("avg_tenure_months"), 48) * 0.20
+ _cap01(pl.col("num_industries").cast(pl.Float64), 5) * 0.20
)
reliability = (
_cap01(pl.col("interview_completion_rate").fill_null(0), 1) * 0.50
+ _cap01(pl.col("offer_acceptance_rate_clean").fill_null(0), 1) * 0.35
+ pl.col("has_offer_history").cast(pl.Float64) * 0.15
)
reputation = (
_cap01(pl.col("endorsements_received").cast(pl.Float64), 200) * 0.20
+ _cap01(pl.col("connection_count").cast(pl.Float64), 500) * 0.15
+ _cap01(pl.col("saved_by_recruiters_30d").cast(pl.Float64), 20) * 0.20
+ _cap01(pl.col("search_appearance_30d").cast(pl.Float64), 50) * 0.15
+ _cap01(pl.col("github_score_clean").fill_null(0), 100) * 0.20
+ pl.col("has_github").cast(pl.Float64) * 0.10
)
engagement = (
(1.0 - _cap01(pl.col("days_since_active").cast(pl.Float64), 180)) * 0.40
+ _cap01(pl.col("recruiter_response_rate").fill_null(0), 1) * 0.35
+ (1.0 - _cap01(pl.col("avg_response_time_hours").fill_null(72), 72)) * 0.25
)
verification = (
_cap01(pl.col("verification_count").cast(pl.Float64), 3) * 0.50
+ _cap01(pl.col("profile_completeness_score"), 100) * 0.50
)
education = (
_cap01(pl.col("best_institution_tier_rank").cast(pl.Float64), 4) * 0.55
+ _cap01(pl.col("highest_degree_rank").cast(pl.Float64), 4) * 0.45
)
candidate_quality = (
weights.skills * skills + weights.experience * experience + weights.reliability * reliability
+ weights.reputation * reputation + weights.engagement * engagement
+ weights.verification * verification + weights.education * education
).clip(0.0, 1.0)
return [
skills.alias("score_skills"), experience.alias("score_experience"),
reliability.alias("score_reliability"), reputation.alias("score_reputation"),
engagement.alias("score_engagement"), verification.alias("score_verification"),
education.alias("score_education"), candidate_quality.alias("candidate_quality"),
]
# Pillar → human label for the strengths clause folded into the embedded text (computed parameters → words the
# granite embedder can use). A pillar is named only when it's a genuine strength (score above the threshold).
_PILLAR_LABELS = {
"score_skills": "skills", "score_experience": "experience", "score_reliability": "reliability",
"score_reputation": "reputation", "score_engagement": "engagement",
"score_verification": "verification", "score_education": "education",
}
_STRENGTH_THRESHOLD = 0.6
def _strengths_clause() -> pl.Expr:
"""' Strengths: <strong pillars>.' from the pillars (empty when none clear the threshold)."""
labels = pl.concat_list([
pl.when(pl.col(column) >= _STRENGTH_THRESHOLD).then(pl.lit(label)).otherwise(None)
for column, label in _PILLAR_LABELS.items()
]).list.drop_nulls().list.join(", ")
return pl.when(labels.str.len_chars() > 0).then(pl.lit(" Strengths: ") + labels + pl.lit(".")).otherwise(pl.lit(""))
def quality(candidates: pl.DataFrame, weights: QualitySettings | None = None) -> pl.DataFrame:
"""Add the seniority/degree/tier ranks, the 7 quality pillars, ``candidate_quality``, and fold a strengths
clause into the embedded ``text``. Never drops."""
pillar_weights = weights or load_settings().quality
with step("preprocess.quality", rows_in=candidates.height) as metrics:
out = (
candidates.with_columns(_rank_label_exprs())
.with_columns(_rank_code_exprs())
.with_columns(_pillar_exprs(pillar_weights))
.with_columns([pl.col(column).round(3) for column in _PILLAR_COLUMNS])
)
if "text" in out.columns: # enrich the granite-embedded body with the computed pillar strengths
out = out.with_columns((pl.col("text") + _strengths_clause()).str.strip_chars().alias("text"))
if "embed_source_hash" in out.columns: # re-hash: materialize.py's hash predates this text change
from .materialize import _sha16
out = out.with_columns(pl.Series("embed_source_hash", _sha16(out.get_column("text").to_list())))
metrics["rows_out"] = out.height
metrics["mean_quality"] = round(float(out.select(pl.col("candidate_quality").mean()).item() or 0), 4)
return out
|