File size: 8,446 Bytes
700e896
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
fb57bfa
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
700e896
fb57bfa
 
700e896
 
 
 
 
 
 
 
fb57bfa
 
a41b023
 
 
 
700e896
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
"""Preprocess sub-phase — quality pillars: 7 query-independent sub-scores + blended candidate_quality.

A candidate-only prior (no JD dependency): each pillar is a capped-linear blend of materialized features,
and ``candidate_quality`` is their weighted sum (weights from central config). Adds the seniority / degree /
institution-tier ranks the pillars need. Never drops a row.
"""

from __future__ import annotations

import polars as pl

from common.logging import step
from config import QualitySettings, load_settings

# Ordinal ranks (0 = no signal). Higher = more senior / more prestigious / higher degree.
_SENIORITY_RANK = {"unknown": 0, "ic": 1, "lead": 2, "manager": 3, "director": 4, "vp": 5, "c_level": 6}
_TIER_RANK = {"unknown": 0, "tier_4": 1, "tier_3": 2, "tier_2": 3, "tier_1": 4}
_DEGREE_RANK = {"unknown": 0, "none": 0, "diploma": 1, "bachelor": 2, "master": 3, "phd": 4}

_PILLAR_COLUMNS = [
    "score_skills", "score_experience", "score_reliability", "score_reputation",
    "score_engagement", "score_verification", "score_education", "candidate_quality",
]


def _cap01(expr: pl.Expr, cap: float) -> pl.Expr:
    """Capped linear norm to [0, 1]: ``expr / cap`` clipped."""
    return (expr / cap).clip(0.0, 1.0)


def _rank_label_exprs() -> list[pl.Expr]:
    """Derive the seniority / degree / institution-tier LABELS the pillars rank off."""
    title = pl.col("current_title").cast(pl.String).fill_null("").str.to_lowercase()
    seniority = (
        pl.when(title.str.len_chars() == 0).then(pl.lit("unknown"))
        .when(title.str.contains(r"\b(?:chief|ceo|cto|cpo|cfo|coo|founder)\b")).then(pl.lit("c_level"))
        .when(title.str.contains(r"\b(?:vp|vice president)\b")).then(pl.lit("vp"))
        .when(title.str.contains(r"\bdirector\b")).then(pl.lit("director"))
        .when(title.str.contains(r"\b(?:manager|head of)\b")).then(pl.lit("manager"))
        .when(title.str.contains(r"\b(?:lead|principal|staff)\b")).then(pl.lit("lead"))
        .otherwise(pl.lit("ic"))
        .alias("seniority_level")
    )
    degree = pl.element().struct.field("degree").cast(pl.String).str.to_lowercase()
    highest_degree = pl.col("education").list.eval(
        pl.when(degree.str.contains("phd|doctor")).then(pl.lit("phd"))
        .when(degree.str.contains("master|msc|mba")).then(pl.lit("master"))
        .when(degree.str.contains("bachelor|btech|bsc|be |ba ")).then(pl.lit("bachelor"))
        .when(degree.str.contains("diploma")).then(pl.lit("diploma"))
        .otherwise(pl.lit("unknown"))
    ).list.max().fill_null("unknown").alias("highest_degree_level")
    best_tier = (
        pl.col("flat_institution_tiers").list.eval(pl.element().filter(pl.element() != ""))
        .list.max().fill_null("unknown").alias("best_institution_tier")
    )
    return [seniority, highest_degree, best_tier]


def _rank_code_exprs() -> list[pl.Expr]:
    """Map the labels to their ordinal ranks (the numeric inputs to the experience/education pillars)."""
    return [
        pl.col("seniority_level").replace_strict(_SENIORITY_RANK, default=0, return_dtype=pl.Int32).alias("seniority_rank"),
        pl.col("highest_degree_level").replace_strict(_DEGREE_RANK, default=0, return_dtype=pl.Int32).alias("highest_degree_rank"),
        pl.col("best_institution_tier").replace_strict(_TIER_RANK, default=0, return_dtype=pl.Int32).alias("best_institution_tier_rank"),
    ]


def _pillar_exprs(weights: QualitySettings) -> list[pl.Expr]:
    """The 7 capped-linear pillars + their weighted blend ``candidate_quality`` (clipped to [0, 1])."""
    skills = (
        _cap01(pl.col("num_advanced_skills").cast(pl.Float64), 8) * 0.25
        + _cap01(pl.col("num_expert_skills").cast(pl.Float64), 5) * 0.20
        + _cap01(pl.col("avg_skill_assessment").fill_null(0), 100) * 0.30
        + _cap01(pl.col("assessment_coverage").fill_null(0), 1) * 0.25
    )
    experience = (
        _cap01(pl.col("years_of_experience"), 15) * 0.35
        + _cap01(pl.col("seniority_rank").cast(pl.Float64), 6) * 0.25
        + _cap01(pl.col("avg_tenure_months"), 48) * 0.20
        + _cap01(pl.col("num_industries").cast(pl.Float64), 5) * 0.20
    )
    reliability = (
        _cap01(pl.col("interview_completion_rate").fill_null(0), 1) * 0.50
        + _cap01(pl.col("offer_acceptance_rate_clean").fill_null(0), 1) * 0.35
        + pl.col("has_offer_history").cast(pl.Float64) * 0.15
    )
    reputation = (
        _cap01(pl.col("endorsements_received").cast(pl.Float64), 200) * 0.20
        + _cap01(pl.col("connection_count").cast(pl.Float64), 500) * 0.15
        + _cap01(pl.col("saved_by_recruiters_30d").cast(pl.Float64), 20) * 0.20
        + _cap01(pl.col("search_appearance_30d").cast(pl.Float64), 50) * 0.15
        + _cap01(pl.col("github_score_clean").fill_null(0), 100) * 0.20
        + pl.col("has_github").cast(pl.Float64) * 0.10
    )
    engagement = (
        (1.0 - _cap01(pl.col("days_since_active").cast(pl.Float64), 180)) * 0.40
        + _cap01(pl.col("recruiter_response_rate").fill_null(0), 1) * 0.35
        + (1.0 - _cap01(pl.col("avg_response_time_hours").fill_null(72), 72)) * 0.25
    )
    verification = (
        _cap01(pl.col("verification_count").cast(pl.Float64), 3) * 0.50
        + _cap01(pl.col("profile_completeness_score"), 100) * 0.50
    )
    education = (
        _cap01(pl.col("best_institution_tier_rank").cast(pl.Float64), 4) * 0.55
        + _cap01(pl.col("highest_degree_rank").cast(pl.Float64), 4) * 0.45
    )
    candidate_quality = (
        weights.skills * skills + weights.experience * experience + weights.reliability * reliability
        + weights.reputation * reputation + weights.engagement * engagement
        + weights.verification * verification + weights.education * education
    ).clip(0.0, 1.0)
    return [
        skills.alias("score_skills"), experience.alias("score_experience"),
        reliability.alias("score_reliability"), reputation.alias("score_reputation"),
        engagement.alias("score_engagement"), verification.alias("score_verification"),
        education.alias("score_education"), candidate_quality.alias("candidate_quality"),
    ]


# Pillar → human label for the strengths clause folded into the embedded text (computed parameters → words the
# granite embedder can use). A pillar is named only when it's a genuine strength (score above the threshold).
_PILLAR_LABELS = {
    "score_skills": "skills", "score_experience": "experience", "score_reliability": "reliability",
    "score_reputation": "reputation", "score_engagement": "engagement",
    "score_verification": "verification", "score_education": "education",
}
_STRENGTH_THRESHOLD = 0.6


def _strengths_clause() -> pl.Expr:
    """' Strengths: <strong pillars>.' from the pillars (empty when none clear the threshold)."""
    labels = pl.concat_list([
        pl.when(pl.col(column) >= _STRENGTH_THRESHOLD).then(pl.lit(label)).otherwise(None)
        for column, label in _PILLAR_LABELS.items()
    ]).list.drop_nulls().list.join(", ")
    return pl.when(labels.str.len_chars() > 0).then(pl.lit(" Strengths: ") + labels + pl.lit(".")).otherwise(pl.lit(""))


def quality(candidates: pl.DataFrame, weights: QualitySettings | None = None) -> pl.DataFrame:
    """Add the seniority/degree/tier ranks, the 7 quality pillars, ``candidate_quality``, and fold a strengths
    clause into the embedded ``text``. Never drops."""
    pillar_weights = weights or load_settings().quality
    with step("preprocess.quality", rows_in=candidates.height) as metrics:
        out = (
            candidates.with_columns(_rank_label_exprs())
            .with_columns(_rank_code_exprs())
            .with_columns(_pillar_exprs(pillar_weights))
            .with_columns([pl.col(column).round(3) for column in _PILLAR_COLUMNS])
        )
        if "text" in out.columns:  # enrich the granite-embedded body with the computed pillar strengths
            out = out.with_columns((pl.col("text") + _strengths_clause()).str.strip_chars().alias("text"))
            if "embed_source_hash" in out.columns:  # re-hash: materialize.py's hash predates this text change
                from .materialize import _sha16

                out = out.with_columns(pl.Series("embed_source_hash", _sha16(out.get_column("text").to_list())))
        metrics["rows_out"] = out.height
        metrics["mean_quality"] = round(float(out.select(pl.col("candidate_quality").mean()).item() or 0), 4)
        return out