Download src/preprocess/quality.py from hembad/reranker: direct link, hf CLI and curl.
- Browser
- Download file 8.45 kB
-
https://huggingface.co/spaces/hembad/reranker/resolve/main/src/preprocess/quality.py
- Command line
-
hf download hf://spaces/hembad/reranker/src/preprocess/quality.py
-
curl -L -o quality.py https://huggingface.co/spaces/hembad/reranker/resolve/main/src/preprocess/quality.py
8.45 kB
| """Preprocess sub-phase — quality pillars: 7 query-independent sub-scores + blended candidate_quality. | |
| A candidate-only prior (no JD dependency): each pillar is a capped-linear blend of materialized features, | |
| and ``candidate_quality`` is their weighted sum (weights from central config). Adds the seniority / degree / | |
| institution-tier ranks the pillars need. Never drops a row. | |
| """ | |
| from __future__ import annotations | |
| import polars as pl | |
| from common.logging import step | |
| from config import QualitySettings, load_settings | |
| # Ordinal ranks (0 = no signal). Higher = more senior / more prestigious / higher degree. | |
| _SENIORITY_RANK = {"unknown": 0, "ic": 1, "lead": 2, "manager": 3, "director": 4, "vp": 5, "c_level": 6} | |
| _TIER_RANK = {"unknown": 0, "tier_4": 1, "tier_3": 2, "tier_2": 3, "tier_1": 4} | |
| _DEGREE_RANK = {"unknown": 0, "none": 0, "diploma": 1, "bachelor": 2, "master": 3, "phd": 4} | |
| _PILLAR_COLUMNS = [ | |
| "score_skills", "score_experience", "score_reliability", "score_reputation", | |
| "score_engagement", "score_verification", "score_education", "candidate_quality", | |
| ] | |
| def _cap01(expr: pl.Expr, cap: float) -> pl.Expr: | |
| """Capped linear norm to [0, 1]: ``expr / cap`` clipped.""" | |
| return (expr / cap).clip(0.0, 1.0) | |
| def _rank_label_exprs() -> list[pl.Expr]: | |
| """Derive the seniority / degree / institution-tier LABELS the pillars rank off.""" | |
| title = pl.col("current_title").cast(pl.String).fill_null("").str.to_lowercase() | |
| seniority = ( | |
| pl.when(title.str.len_chars() == 0).then(pl.lit("unknown")) | |
| .when(title.str.contains(r"\b(?:chief|ceo|cto|cpo|cfo|coo|founder)\b")).then(pl.lit("c_level")) | |
| .when(title.str.contains(r"\b(?:vp|vice president)\b")).then(pl.lit("vp")) | |
| .when(title.str.contains(r"\bdirector\b")).then(pl.lit("director")) | |
| .when(title.str.contains(r"\b(?:manager|head of)\b")).then(pl.lit("manager")) | |
| .when(title.str.contains(r"\b(?:lead|principal|staff)\b")).then(pl.lit("lead")) | |
| .otherwise(pl.lit("ic")) | |
| .alias("seniority_level") | |
| ) | |
| degree = pl.element().struct.field("degree").cast(pl.String).str.to_lowercase() | |
| highest_degree = pl.col("education").list.eval( | |
| pl.when(degree.str.contains("phd|doctor")).then(pl.lit("phd")) | |
| .when(degree.str.contains("master|msc|mba")).then(pl.lit("master")) | |
| .when(degree.str.contains("bachelor|btech|bsc|be |ba ")).then(pl.lit("bachelor")) | |
| .when(degree.str.contains("diploma")).then(pl.lit("diploma")) | |
| .otherwise(pl.lit("unknown")) | |
| ).list.max().fill_null("unknown").alias("highest_degree_level") | |
| best_tier = ( | |
| pl.col("flat_institution_tiers").list.eval(pl.element().filter(pl.element() != "")) | |
| .list.max().fill_null("unknown").alias("best_institution_tier") | |
| ) | |
| return [seniority, highest_degree, best_tier] | |
| def _rank_code_exprs() -> list[pl.Expr]: | |
| """Map the labels to their ordinal ranks (the numeric inputs to the experience/education pillars).""" | |
| return [ | |
| pl.col("seniority_level").replace_strict(_SENIORITY_RANK, default=0, return_dtype=pl.Int32).alias("seniority_rank"), | |
| pl.col("highest_degree_level").replace_strict(_DEGREE_RANK, default=0, return_dtype=pl.Int32).alias("highest_degree_rank"), | |
| pl.col("best_institution_tier").replace_strict(_TIER_RANK, default=0, return_dtype=pl.Int32).alias("best_institution_tier_rank"), | |
| ] | |
| def _pillar_exprs(weights: QualitySettings) -> list[pl.Expr]: | |
| """The 7 capped-linear pillars + their weighted blend ``candidate_quality`` (clipped to [0, 1]).""" | |
| skills = ( | |
| _cap01(pl.col("num_advanced_skills").cast(pl.Float64), 8) * 0.25 | |
| + _cap01(pl.col("num_expert_skills").cast(pl.Float64), 5) * 0.20 | |
| + _cap01(pl.col("avg_skill_assessment").fill_null(0), 100) * 0.30 | |
| + _cap01(pl.col("assessment_coverage").fill_null(0), 1) * 0.25 | |
| ) | |
| experience = ( | |
| _cap01(pl.col("years_of_experience"), 15) * 0.35 | |
| + _cap01(pl.col("seniority_rank").cast(pl.Float64), 6) * 0.25 | |
| + _cap01(pl.col("avg_tenure_months"), 48) * 0.20 | |
| + _cap01(pl.col("num_industries").cast(pl.Float64), 5) * 0.20 | |
| ) | |
| reliability = ( | |
| _cap01(pl.col("interview_completion_rate").fill_null(0), 1) * 0.50 | |
| + _cap01(pl.col("offer_acceptance_rate_clean").fill_null(0), 1) * 0.35 | |
| + pl.col("has_offer_history").cast(pl.Float64) * 0.15 | |
| ) | |
| reputation = ( | |
| _cap01(pl.col("endorsements_received").cast(pl.Float64), 200) * 0.20 | |
| + _cap01(pl.col("connection_count").cast(pl.Float64), 500) * 0.15 | |
| + _cap01(pl.col("saved_by_recruiters_30d").cast(pl.Float64), 20) * 0.20 | |
| + _cap01(pl.col("search_appearance_30d").cast(pl.Float64), 50) * 0.15 | |
| + _cap01(pl.col("github_score_clean").fill_null(0), 100) * 0.20 | |
| + pl.col("has_github").cast(pl.Float64) * 0.10 | |
| ) | |
| engagement = ( | |
| (1.0 - _cap01(pl.col("days_since_active").cast(pl.Float64), 180)) * 0.40 | |
| + _cap01(pl.col("recruiter_response_rate").fill_null(0), 1) * 0.35 | |
| + (1.0 - _cap01(pl.col("avg_response_time_hours").fill_null(72), 72)) * 0.25 | |
| ) | |
| verification = ( | |
| _cap01(pl.col("verification_count").cast(pl.Float64), 3) * 0.50 | |
| + _cap01(pl.col("profile_completeness_score"), 100) * 0.50 | |
| ) | |
| education = ( | |
| _cap01(pl.col("best_institution_tier_rank").cast(pl.Float64), 4) * 0.55 | |
| + _cap01(pl.col("highest_degree_rank").cast(pl.Float64), 4) * 0.45 | |
| ) | |
| candidate_quality = ( | |
| weights.skills * skills + weights.experience * experience + weights.reliability * reliability | |
| + weights.reputation * reputation + weights.engagement * engagement | |
| + weights.verification * verification + weights.education * education | |
| ).clip(0.0, 1.0) | |
| return [ | |
| skills.alias("score_skills"), experience.alias("score_experience"), | |
| reliability.alias("score_reliability"), reputation.alias("score_reputation"), | |
| engagement.alias("score_engagement"), verification.alias("score_verification"), | |
| education.alias("score_education"), candidate_quality.alias("candidate_quality"), | |
| ] | |
| # Pillar → human label for the strengths clause folded into the embedded text (computed parameters → words the | |
| # granite embedder can use). A pillar is named only when it's a genuine strength (score above the threshold). | |
| _PILLAR_LABELS = { | |
| "score_skills": "skills", "score_experience": "experience", "score_reliability": "reliability", | |
| "score_reputation": "reputation", "score_engagement": "engagement", | |
| "score_verification": "verification", "score_education": "education", | |
| } | |
| _STRENGTH_THRESHOLD = 0.6 | |
| def _strengths_clause() -> pl.Expr: | |
| """' Strengths: <strong pillars>.' from the pillars (empty when none clear the threshold).""" | |
| labels = pl.concat_list([ | |
| pl.when(pl.col(column) >= _STRENGTH_THRESHOLD).then(pl.lit(label)).otherwise(None) | |
| for column, label in _PILLAR_LABELS.items() | |
| ]).list.drop_nulls().list.join(", ") | |
| return pl.when(labels.str.len_chars() > 0).then(pl.lit(" Strengths: ") + labels + pl.lit(".")).otherwise(pl.lit("")) | |
| def quality(candidates: pl.DataFrame, weights: QualitySettings | None = None) -> pl.DataFrame: | |
| """Add the seniority/degree/tier ranks, the 7 quality pillars, ``candidate_quality``, and fold a strengths | |
| clause into the embedded ``text``. Never drops.""" | |
| pillar_weights = weights or load_settings().quality | |
| with step("preprocess.quality", rows_in=candidates.height) as metrics: | |
| out = ( | |
| candidates.with_columns(_rank_label_exprs()) | |
| .with_columns(_rank_code_exprs()) | |
| .with_columns(_pillar_exprs(pillar_weights)) | |
| .with_columns([pl.col(column).round(3) for column in _PILLAR_COLUMNS]) | |
| ) | |
| if "text" in out.columns: # enrich the granite-embedded body with the computed pillar strengths | |
| out = out.with_columns((pl.col("text") + _strengths_clause()).str.strip_chars().alias("text")) | |
| if "embed_source_hash" in out.columns: # re-hash: materialize.py's hash predates this text change | |
| from .materialize import _sha16 | |
| out = out.with_columns(pl.Series("embed_source_hash", _sha16(out.get_column("text").to_list()))) | |
| metrics["rows_out"] = out.height | |
| metrics["mean_quality"] = round(float(out.select(pl.col("candidate_quality").mean()).item() or 0), 4) | |
| return out | |