job-processor / services /processor_utils.py
Eng-Musa's picture
own processor
2d765ca
Raw
History Blame Contribute Delete
71.2 kB
"""
processor_utils.py
==================
Contains the static skill taxonomy (copied from the scraper service) and
re-exports the utility functions and constants used by cv_chunker.py
and job_matcher.py.
All other modules import from here:
from services.processor_utils import (
DEFAULT_SKILL_ALIASES, DEFAULT_CATEGORY_SKILLS, SOFT_SKILL_KEYS,
SENIORITY_ORDER, _TECH_LOC_BLACKLIST,
_normalize, _has_alias, _extract_skills, _infer_category,
_detect_seniority, _extract_years, compute_years_from_experience,
refine_seniority_with_years, _extract_cv_title,
)
"""
from __future__ import annotations
import logging
import re
from datetime import datetime as _dt
from typing import Any, Dict, List, Optional, Set, Tuple
logger = logging.getLogger(__name__)
# ---------------------------------------------------------------------------
# Configuration
# ---------------------------------------------------------------------------
DEFAULT_SKILL_ALIASES: Dict[str, List[str]] = {
"3d design": [
"3d design",
"blender",
"cinema 4d",
"3d modeling",
"3d rendering",
"maya",
],
"3pl": [
"3pl",
"third party logistics",
"outsourced logistics",
"logistics provider",
],
"acca": ["acca", "chartered accountant", "cpa", "cima", "icpak", "aca"],
"account management": [
"account management",
"key account management",
"client management",
"portfolio management",
],
"accounting": ["accounting", "accountancy", "bookkeeping", "accounts"],
"accounts payable": ["accounts payable", "vendor payments", "creditors", "ap"],
"accounts receivable": [
"accounts receivable",
"invoicing",
"debtors",
"collections",
"ar",
],
"active directory": [
"active directory",
"ldap",
"azure ad",
"microsoft entra",
"entra id",
"domain services",
],
"actuarial": ["actuarial", "actuary", "actuarial science", "actuarial analysis"],
"adaptability": ["adaptability", "adaptable", "flexible", "versatile"],
"adobe after effects": ["after effects", "adobe after effects", "motion graphics"],
"adobe illustrator": ["illustrator", "adobe illustrator", "vector graphics"],
"adobe photoshop": ["photoshop", "adobe photoshop"],
"adobe premiere": ["adobe premiere", "premiere pro", "video editing"],
"adobe xd": ["adobe xd", "xd"],
"affiliate marketing": [
"affiliate marketing",
"performance marketing",
"influencer marketing",
"partnership marketing",
],
"agile": [
"agile",
"agile methodology",
"agile development",
"agile framework",
"scrum",
"scrum master",
"safe",
"scaled agile",
],
"aml": [
"aml",
"anti-money laundering",
"anti money laundering",
"financial crime",
"fraud detection",
],
"android": ["android", "android studio", "android sdk", "android development"],
"angular": ["angular", "angularjs", "angular.js", "angular 2+"],
"ansible": ["ansible", "ansible playbook", "configuration management"],
"apache airflow": ["apache airflow", "airflow", "workflow orchestration", "dag"],
"apache hadoop": ["apache hadoop", "hadoop", "hdfs", "hive", "hbase"],
"apache spark": ["apache spark", "pyspark", "spark streaming", "spark sql"],
"arbitration": [
"arbitration",
"mediation",
"dispute resolution",
"adr",
"alternative dispute resolution",
],
"assessment": [
"assessment",
"grading",
"marking",
"examination",
"evaluation",
"formative assessment",
],
"attention to detail": [
"attention to detail",
"accuracy",
"meticulous",
"detail oriented",
"detail-oriented",
],
"auditing": [
"auditing",
"internal audit",
"external audit",
"audit",
"statutory audit",
"audit trail",
],
"autocad": ["autocad", "auto cad", "cad", "computer aided design", "drafting"],
"aws": [
"aws",
"amazon web services",
"ec2",
"s3",
"lambda",
"ecs",
"eks",
"rds",
"cloudformation",
"cdk",
],
"azure": [
"azure",
"microsoft azure",
"azure devops",
"azure functions",
"azure pipelines",
"arm templates",
"bicep",
],
"backup recovery": [
"backup",
"backup and recovery",
"backup strategy",
"data backup",
"veeam",
"commvault",
],
"bash": [
"bash",
"shell scripting",
"bash scripting",
"shell script",
"unix scripting",
],
"bloomberg": ["bloomberg", "bloomberg terminal", "bloomberg api"],
"brand design": ["brand design", "branding", "visual identity", "brand guidelines"],
"brand management": ["brand management", "brand strategy"],
"budgeting": [
"budgeting",
"budget management",
"budget planning",
"forecasting",
"variance analysis",
"capex",
"opex",
],
"business analysis": [
"business analysis",
"business analyst",
"requirements gathering",
"process improvement",
"business case",
],
"business process": [
"business process",
"bpm",
"business process management",
"process mapping",
"process documentation",
"sop",
],
"c": ["c language", "c programming", "embedded c"],
"c#": ["c#", "csharp", "c sharp"],
"c++": ["c++", "cpp", "c plus plus"],
"call centre": [
"call centre",
"call center",
"contact centre",
"contact center",
"inbound calls",
"outbound calls",
],
"canva": ["canva"],
"cash flow": ["cash flow", "cash management", "treasury", "liquidity management"],
"cassandra": ["cassandra", "apache cassandra"],
"cfa": ["cfa", "chartered financial analyst"],
"change management": [
"change management",
"organisational change",
"change leadership",
"transformation",
],
"chef": ["chef", "chef infra"],
"ci/cd": [
"ci/cd",
"ci cd",
"github actions",
"jenkins",
"gitlab ci",
"bitbucket pipelines",
"circle ci",
"travis ci",
"argocd",
"tekton",
],
"cisco": ["cisco", "ccna", "ccnp", "ccie", "cisco ios", "cisco networking"],
"civil 3d": ["civil 3d", "autocad civil 3d", "road design", "drainage design"],
"civil engineering": [
"civil engineering",
"structural engineering",
"geotechnical",
"quantity surveying",
"qs",
],
"classroom management": [
"classroom management",
"student management",
"behaviour management",
"student discipline",
],
"clickhouse": ["clickhouse"],
"clinical skills": [
"clinical skills",
"clinical assessment",
"clinical procedures",
"clinical competencies",
],
"clinical trials": [
"clinical trials",
"good clinical practice",
"gcp",
"clinical research",
"research protocol",
],
"cloud security": ["cloud security", "aws security", "azure security", "cspm"],
"cloudwatch": ["cloudwatch", "aws cloudwatch"],
"communication": [
"communication",
"interpersonal",
"verbal communication",
"written communication",
"correspondence",
],
"compensation": [
"compensation",
"benefits",
"compensation and benefits",
"remuneration",
"total rewards",
"salary benchmarking",
],
"computer vision": [
"computer vision",
"image recognition",
"object detection",
"opencv",
"image processing",
],
"conflict resolution": [
"conflict resolution",
"dispute management",
"mediation skills",
"grievance resolution",
],
"confluence": ["confluence", "atlassian confluence"],
"content writing": [
"content writing",
"copywriting",
"content creation",
"blog writing",
"content strategy",
"editorial",
],
"context api": ["context api", "react context"],
"contract drafting": [
"contract drafting",
"contract review",
"drafting agreements",
"legal drafting",
"contract management",
],
"conveyancing": [
"conveyancing",
"property law",
"land transactions",
"title search",
],
"core banking": [
"core banking",
"banking system",
"flexcube",
"temenos",
"finacle",
"murex",
"t24",
],
"corporate law": [
"corporate law",
"company law",
"corporate governance",
"company secretarial",
],
"credit analysis": [
"credit analysis",
"credit risk",
"loan appraisal",
"credit scoring",
"underwriting",
],
"crm": [
"crm",
"salesforce",
"hubspot",
"zoho crm",
"customer relationship management",
"pipedrive",
"dynamics crm",
],
"cro": [
"cro",
"conversion rate optimization",
"conversion optimisation",
"a/b testing",
"landing page optimization",
],
"css": ["css", "css3", "sass", "scss", "less", "stylesheets"],
"cucumber": [
"cucumber",
"gherkin",
"bdd",
"behavior driven development",
"behaviour driven development",
],
"culture": [
"culture",
"employee engagement",
"employer branding",
"employee experience",
],
"curriculum development": [
"curriculum development",
"curriculum design",
"lesson planning",
"scheme of work",
"curriculum mapping",
],
"customer satisfaction": [
"customer satisfaction",
"csat",
"net promoter score",
"nps",
"customer feedback",
],
"customer service": [
"customer service",
"client service",
"customer support",
"client relations",
"client care",
],
"customer success": [
"customer success",
"customer retention",
"account growth",
"upselling",
"cross-selling",
],
"customs": [
"customs",
"customs clearance",
"import export",
"trade compliance",
"incoterms",
],
"cypress": ["cypress", "cypress.io"],
"dart": ["dart"],
"data analysis": [
"data analysis",
"analytics",
"data analytics",
"business intelligence",
],
"data entry": [
"data entry",
"typing",
"data input",
"data capture",
"data processing",
],
"data governance": [
"data governance",
"data quality",
"data catalog",
"metadata management",
"data stewardship",
],
"data modeling": [
"data modeling",
"data modelling",
"star schema",
"dimensional modeling",
"entity relationship",
],
"data pipeline": [
"data pipeline",
"etl",
"elt",
"data integration",
"data ingestion",
"data warehouse",
"data lake",
"data lakehouse",
],
"data science": ["data science", "data scientist"],
"data visualization": [
"data visualization",
"data visualisation",
"data viz",
"dashboarding",
],
"databricks": ["databricks", "lakehouse"],
"datadog": ["datadog", "data dog"],
"dbt": ["dbt", "data build tool", "analytics engineering"],
"decision making": [
"decision making",
"decision-making",
"strategic thinking",
"judgement",
],
"dei": [
"diversity equity inclusion",
"dei",
"diversity and inclusion",
"equal opportunity",
"inclusive workplace",
],
"demand planning": [
"demand planning",
"demand forecasting",
"s&op",
"sales and operations planning",
],
"derivatives": [
"derivatives",
"futures",
"options",
"swaps",
"structured products",
],
"devsecops": [
"devsecops",
"security as code",
"shift left security",
"sast",
"dast",
],
"digital marketing": [
"digital marketing",
"online marketing",
"marketing campaigns",
"marketing strategy",
"performance marketing",
],
"distance learning": [
"distance learning",
"online teaching",
"remote teaching",
"blended learning",
"hybrid learning",
],
"django": ["django", "django rest framework", "drf"],
"docker": [
"docker",
"docker compose",
"dockerfile",
"containerization",
"containerisation",
],
"dotnet": [
".net",
".net core",
"dotnet",
"asp.net",
"asp.net core",
"net core",
".net framework",
".net maui",
"blazor",
"wpf",
"xamarin",
],
"due diligence": [
"due diligence",
"legal due diligence",
"commercial due diligence",
],
"dynamodb": ["dynamodb", "dynamo db", "amazon dynamodb"],
"dynatrace": ["dynatrace"],
"e-learning": [
"e-learning",
"elearning",
"lms",
"moodle",
"instructional design",
"online course development",
],
"ehr systems": [
"ehr",
"emr",
"electronic health record",
"epic",
"cerner",
"meditech",
"allscripts",
"hospital information system",
],
"elasticsearch": ["elasticsearch", "elastic search", "opensearch", "elk"],
"electrical engineering": [
"electrical engineering",
"power systems",
"plc",
"scada",
"instrumentation",
"hv",
"lv",
],
"elixir": ["elixir", "phoenix framework", "phoenix"],
"elk stack": ["elk stack", "logstash", "kibana", "elastic stack", "fluentd"],
"email marketing": [
"email marketing",
"mailchimp",
"email campaigns",
"newsletter",
"klaviyo",
"sendgrid",
],
"emotional intelligence": [
"emotional intelligence",
"eq",
"empathy",
"self-awareness",
],
"employee relations": [
"employee relations",
"labour relations",
"labor relations",
"grievance handling",
"disciplinary",
],
"employment law": [
"employment law",
"labour law",
"labor law",
"employment tribunal",
],
"entity framework": ["entity framework", "ef core", "entity framework core"],
"erp": [
"erp",
"enterprise resource planning",
"sap erp",
"oracle erp",
"dynamics 365",
"netsuite",
"odoo",
],
"event management": [
"event management",
"events coordination",
"venue management",
"conference management",
],
"excel analytics": [
"pivot tables",
"advanced excel",
"excel analytics",
"vlookup",
"xlookup",
],
"f#": ["f#", "fsharp", "f sharp"],
"facebook ads": [
"facebook ads",
"meta ads",
"social media advertising",
"paid social",
],
"fastapi": ["fastapi", "fast api"],
"figma": ["figma"],
"filing": [
"filing",
"document management",
"records management",
"file management",
"document control",
"record keeping",
],
"financial analysis": [
"financial analysis",
"financial modelling",
"financial modeling",
"financial reporting",
"financial planning",
"fp&a",
],
"financial compliance": [
"financial compliance",
"financial regulation",
"regulatory reporting",
"cbk",
"sec compliance",
],
"firestore": [
"firestore",
"firebase",
"cloud firestore",
"firebase realtime database",
],
"first aid": [
"first aid",
"cpr",
"bls",
"basic life support",
"acls",
"emergency response",
"resuscitation",
],
"flask": ["flask", "flask api"],
"flutter": ["flutter"],
"food safety": [
"food safety",
"haccp",
"food hygiene",
"food handling",
"food service",
],
"gaap": ["gaap", "generally accepted accounting principles", "us gaap", "uk gaap"],
"gcp": [
"gcp",
"google cloud",
"google cloud platform",
"cloud run",
"gke",
"bigquery",
"cloud functions",
],
"gdpr": ["gdpr", "data protection regulation", "ccpa", "data privacy law"],
"generative ai": [
"generative ai",
"gen ai",
"llm",
"large language model",
"gpt",
"langchain",
"hugging face",
"transformers",
"openai",
],
"git": [
"git",
"github",
"gitlab",
"bitbucket",
"version control",
"source control",
],
"go": ["golang", "go lang", "go programming"],
"google ads": [
"google ads",
"google adwords",
"ppc",
"pay per click",
"sem",
"paid search",
],
"google analytics": [
"google analytics",
"ga4",
"web analytics",
"google tag manager",
"gtm",
],
"google workspace": [
"google workspace",
"google docs",
"google drive",
"gsuite",
"g suite",
],
"grafana": ["grafana", "grafana dashboard"],
"graphic design": ["graphic design", "graphics", "visual design", "print design"],
"graphql": ["graphql", "graph ql", "apollo", "apollo server", "apollo client"],
"groovy": ["groovy"],
"grpc": ["grpc", "protocol buffers", "protobuf"],
"haskell": ["haskell"],
"health and safety": [
"health and safety",
"hse",
"ohse",
"occupational health",
"workplace safety",
"risk assessment",
],
"help desk": [
"help desk",
"helpdesk",
"it support",
"technical support",
"service desk",
"l1 support",
"l2 support",
],
"hibernate": ["hibernate", "jpa", "java persistence api", "orm"],
"high availability": [
"high availability",
"ha cluster",
"failover",
"load balancing",
"disaster recovery",
"rto",
"rpo",
"bcp",
"business continuity",
"fault tolerance",
],
"hl7": [
"hl7",
"hl7 fhir",
"fhir",
"healthcare interoperability",
"health data exchange",
],
"hospitality management": [
"hospitality management",
"hotel management",
"f&b",
"food and beverage",
"front office",
],
"housekeeping": [
"housekeeping",
"rooms division",
"facilities management",
"janitorial",
],
"hr compliance": [
"hr compliance",
"employment law",
"labour law",
"labor law",
"hr policy",
"hr regulation",
"employment relations",
],
"hris": [
"hris",
"hr information system",
"workday",
"bamboohr",
"zoho hr",
"oracle hcm",
"successfactors",
"peoplesoft",
],
"html": ["html", "html5", "semantic html"],
"hyper-v": ["hyper-v", "hyperv", "hyper v"],
"icd coding": [
"icd",
"icd-10",
"icd-11",
"medical coding",
"cpt coding",
"clinical coding",
],
"ifrs": ["ifrs", "international financial reporting standards"],
"infection control": [
"infection control",
"infection prevention",
"ips",
"sterilization",
"ppe",
],
"influxdb": ["influxdb", "influx db", "time series database", "timescaledb"],
"intellectual property": [
"intellectual property",
"ip law",
"trademark",
"patent",
"copyright",
"ip rights",
"ip management",
],
"inventory management": [
"inventory management",
"stock control",
"stock management",
"stock checking",
"inventory tracking",
],
"invision": ["invision", "invision studio"],
"ionic": ["ionic", "capacitor", "cordova", "hybrid app"],
"ios": ["ios", "xcode", "ios development", "iphone development"],
"iso 27001": ["iso 27001", "iso27001", "information security management", "isms"],
"itil": ["itil", "it service management", "itsm", "itil v4", "service lifecycle"],
"jasmine": ["jasmine"],
"java": ["java"],
"javascript": ["javascript", "js", "es6", "es2015", "ecmascript", "vanilla js"],
"jest": ["jest", "jest.js"],
"jetpack compose": ["jetpack compose", "android compose", "compose multiplatform"],
"jira": ["jira", "atlassian jira"],
"jmeter": ["jmeter", "apache jmeter", "load testing", "performance testing"],
"job analysis": [
"job analysis",
"job evaluation",
"job grading",
"job description",
],
"junit": ["junit", "junit5", "junit 4", "junit jupiter"],
"jwt": ["jwt", "json web token", "json web tokens"],
"kafka": ["kafka", "apache kafka", "kafka streams", "confluent", "event streaming"],
"kanban": ["kanban", "kanban board"],
"kotlin": ["kotlin", "kotlin multiplatform"],
"kubernetes": ["kubernetes", "k8s", "helm", "openshift", "container orchestration"],
"kyc": [
"kyc",
"know your customer",
"customer due diligence",
"cdd",
"customer onboarding compliance",
],
"laboratory": [
"laboratory",
"lab technician",
"medical laboratory",
"specimen processing",
"pathology",
],
"laravel": ["laravel"],
"lead generation": [
"lead generation",
"prospecting",
"cold calling",
"cold outreach",
"demand generation",
],
"leadership": [
"leadership",
"team leadership",
"people management",
"people leader",
],
"lean": [
"lean",
"lean manufacturing",
"lean six sigma",
"continuous improvement",
"kaizen",
"5s",
"value stream mapping",
],
"legal compliance": [
"legal compliance",
"regulatory affairs",
"compliance management",
"regulatory framework",
],
"legal research": [
"legal research",
"case research",
"statute interpretation",
"case law",
],
"legal tech": [
"legal tech",
"legaltech",
"contract automation",
"clm",
"e-discovery",
],
"legal writing": [
"legal writing",
"pleadings",
"legal opinions",
"memoranda",
"legal briefs",
],
"linux": [
"linux",
"ubuntu",
"centos",
"rhel",
"red hat",
"debian",
"fedora",
"unix",
"linux administration",
],
"litigation": [
"litigation",
"court proceedings",
"legal proceedings",
"advocacy",
"trial",
],
"live chat": ["live chat", "chat support", "intercom", "drift", "livechat"],
"lms": [
"learning management system",
"blackboard",
"canvas lms",
"google classroom",
"schoology",
"d2l",
"brightspace",
],
"logistics": [
"logistics",
"fleet management",
"transportation management",
"freight",
"delivery",
"last mile delivery",
],
"looker": ["looker", "looker studio", "google data studio"],
"lua": ["lua"],
"machine learning": [
"machine learning",
"ml",
"deep learning",
"neural networks",
"artificial intelligence",
"ai",
],
"mariadb": ["mariadb", "maria db"],
"marketing automation": [
"marketing automation",
"hubspot marketing",
"marketo",
"pardot",
"salesforce marketing cloud",
],
"mechanical engineering": [
"mechanical engineering",
"hvac",
"fluid mechanics",
"thermodynamics",
"piping",
],
"medical records": [
"medical records",
"electronic health records",
"medical documentation",
"health records",
],
"mental health": [
"mental health",
"psychiatry",
"psychology",
"counselling",
"counseling",
"psychotherapy",
"cbt",
],
"mentoring": [
"mentoring",
"mentorship",
"coaching",
"academic guidance",
"career coaching",
],
"mergers acquisitions": [
"mergers and acquisitions",
"m&a",
"corporate restructuring",
"merger",
"acquisition",
"takeover",
],
"message broker": [
"message broker",
"activemq",
"apache activemq",
"nats",
"zeromq",
"pub/sub",
"pubsub",
"azure service bus",
"aws sqs",
"google pub/sub",
],
"microservices": [
"microservices",
"micro-services",
"microservice architecture",
"service-oriented",
"soa",
"event-driven",
],
"microsoft excel": [
"microsoft excel",
"ms excel",
"excel",
"spreadsheets",
"google sheets",
],
"microsoft office": [
"microsoft office",
"ms office",
"office suite",
"office 365",
"microsoft 365",
],
"microsoft outlook": ["microsoft outlook", "ms outlook", "outlook"],
"microsoft powerpoint": [
"microsoft powerpoint",
"ms powerpoint",
"powerpoint",
"power point",
"google slides",
"presentations",
],
"microsoft word": ["microsoft word", "ms word", "word processor"],
"midwifery": [
"midwifery",
"obstetrics",
"maternity",
"antenatal",
"postnatal",
"labour ward",
],
"minute taking": ["minute taking", "meeting minutes", "notetaking", "secretarial"],
"miro": ["miro", "miroboard", "whiteboarding", "collaborative design"],
"mobx": ["mobx"],
"mocha": ["mocha", "mocha.js"],
"mongodb": ["mongodb", "mongo"],
"mqtt": ["mqtt", "hivemq", "mosquitto", "emqx", "iot messaging"],
"ms project": [
"ms project",
"microsoft project",
"project planning software",
"primavera",
],
"mysql": ["mysql"],
"negotiation": [
"negotiation",
"contract negotiation",
"deal closing",
"deal making",
],
"neo4j": ["neo4j", "graph database", "graph db"],
"nestjs": ["nestjs", "nest.js", "nestjs framework"],
"networking": [
"networking",
"network administration",
"network engineering",
"tcp/ip",
"dns",
"dhcp",
"vpn",
"lan",
"wan",
"sd-wan",
"sdwan",
"firewall",
"routing",
"switching",
"vlan",
"mpls",
],
"new relic": ["new relic", "newrelic"],
"next.js": ["next.js", "nextjs", "next js"],
"ngrx": ["ngrx", "ngxs"],
"nist": ["nist", "nist framework", "nist cybersecurity"],
"nlp": [
"nlp",
"natural language processing",
"text mining",
"sentiment analysis",
"text analytics",
],
"node.js": ["node", "node.js", "nodejs", "express", "express.js"],
"nunit": ["nunit"],
"nursing": [
"nursing",
"registered nurse",
"rn",
"clinical nursing",
"enrolled nurse",
"bscn",
],
"nutrition": [
"nutrition",
"dietetics",
"dietitian",
"nutritionist",
"food science",
],
"oauth": ["oauth", "oauth2", "oauth 2.0", "openid connect", "oidc", "pkce"],
"onboarding": [
"staff onboarding",
"new hire orientation",
"employee induction",
"new employee onboarding",
],
"openapi": [
"openapi",
"swagger",
"openapi spec",
"swagger ui",
"api spec",
"api documentation",
],
"opentelemetry": [
"opentelemetry",
"open telemetry",
"distributed tracing",
"jaeger",
"zipkin",
"observability",
],
"oracle db": ["oracle database", "oracle db", "oracle sql", "pl/sql"],
"owasp": ["owasp", "owasp top 10", "web application security", "appsec"],
"partnership": [
"partnership",
"channel sales",
"partnership management",
"alliances",
"channel management",
"reseller",
],
"patient care": [
"patient care",
"patient management",
"bedside manner",
"patient assessment",
"patient monitoring",
],
"payroll": [
"payroll",
"payroll processing",
"payroll management",
"payroll administration",
],
"pci dss": ["pci dss", "pci", "payment card industry"],
"penetration testing": [
"penetration testing",
"pen testing",
"pentesting",
"ethical hacking",
"red team",
"blue team",
"purple team",
],
"performance management": [
"performance management",
"performance review",
"appraisal",
"kpi management",
"okr",
],
"perl": ["perl"],
"pharmacy": [
"pharmacy",
"dispensing",
"pharmaceutical",
"pharmacology",
"medication management",
],
"photography": ["photography", "photo editing", "lightroom", "photo retouching"],
"php": ["php"],
"physiotherapy": [
"physiotherapy",
"physical therapy",
"rehabilitation",
"musculoskeletal",
],
"pinia": ["pinia"],
"postgresql": ["postgresql", "postgres", "supabase"],
"postman": ["postman", "api testing"],
"power bi": ["power bi", "powerbi", "power bi desktop", "dax"],
"powershell": ["powershell", "power shell", "ps script", "windows scripting"],
"pr": [
"pr",
"public relations",
"media relations",
"press release",
"communications",
],
"primavera": [
"primavera",
"oracle primavera",
"p6",
"primavera p6",
"project scheduling",
],
"prisma": ["prisma", "prisma orm"],
"problem solving": [
"problem solving",
"problem-solving",
"critical thinking",
"analytical thinking",
"troubleshooting",
],
"procurement": [
"procurement",
"purchasing",
"vendor management",
"sourcing",
"tendering",
"rfq",
"rfp",
],
"program management": [
"program management",
"programme management",
"portfolio management",
"pmo",
"project portfolio",
],
"project management": [
"project management",
"pmp",
"prince2",
"agile project management",
"project planning",
"waterfall",
"project delivery",
],
"prometheus": ["prometheus", "prometheus monitoring", "alertmanager"],
"property management": [
"property management",
"pms",
"property management system",
"opera pms",
"real estate management",
],
"prototyping": [
"prototyping",
"rapid prototyping",
"interactive prototype",
"high-fidelity prototype",
],
"public speaking": [
"public speaking",
"presentation skills",
"presenting",
"facilitation",
"keynote",
],
"puppet": ["puppet", "puppet enterprise"],
"pytest": ["pytest", "py.test"],
"python": ["python"],
"pytorch": ["pytorch", "torch"],
"qlik": ["qlik", "qlikview", "qliksense"],
"quality control": [
"quality control",
"qc",
"quality assurance",
"qa",
"iso",
"quality management",
"iso 9001",
"total quality management",
],
"quickbooks": ["quickbooks", "quick books"],
"r": ["r programming", "r language", "rstudio"],
"rabbitmq": ["rabbitmq", "rabbit mq", "amqp", "message queue"],
"radiology": [
"radiology",
"imaging",
"mri",
"ct scan",
"x-ray",
"radiography",
"ultrasound",
],
"rails": ["rails", "ruby on rails", "ror"],
"react": ["react", "react.js", "reactjs", "react hooks", "react native"],
"real estate": [
"real estate",
"property valuation",
"mortgage",
"property development",
"estate agency",
"letting",
],
"reconciliation": [
"reconciliation",
"bank reconciliation",
"account reconciliation",
],
"recruitment": [
"recruitment",
"recruiting",
"talent acquisition",
"hiring",
"staffing",
"talent sourcing",
"headhunting",
],
"redis": ["redis", "redis cache", "redis cluster"],
"redux": ["redux", "redux toolkit", "react redux", "rtk query"],
"reporting": [
"reporting",
"report writing",
"report generation",
"management reporting",
"progress reports",
],
"research": [
"research",
"academic research",
"literature review",
"research methodology",
"data collection",
],
"rest api": [
"rest api",
"restful api",
"restful apis",
"restful",
"api development",
"api design",
"web services",
"api integration",
],
"retail": [
"retail",
"retail sales",
"merchandising",
"point of sale",
"pos",
"fmcg",
],
"revenue management": [
"revenue management",
"yield management",
"revenue optimization",
"pricing strategy",
],
"revenue operations": [
"revenue operations",
"revops",
"sales operations",
"go-to-market",
"gtm",
],
"revit": [
"revit",
"bim",
"building information modeling",
"autodesk revit",
"building information modelling",
],
"risk management": [
"risk management",
"risk assessment",
"enterprise risk management",
"erm",
"risk framework",
"risk mitigation",
],
"robot framework": ["robot framework", "robotframework"],
"ruby": ["ruby"],
"rust": ["rust", "rust lang", "systems programming"],
"sage": ["sage", "sage accounting", "sage 50", "sage 200", "sage intacct"],
"sales": [
"sales",
"selling",
"business development",
"b2b sales",
"b2c sales",
"direct sales",
"inside sales",
"field sales",
],
"saml": [
"saml",
"saml 2.0",
"sso",
"single sign-on",
"single sign on",
"federated identity",
],
"sap": ["sap", "sap finance", "sap s/4hana", "sap fico", "sap hana"],
"scala": ["scala"],
"scheduling": [
"scheduling",
"calendar management",
"appointment setting",
"diary management",
"timetabling",
],
"scikit-learn": ["scikit-learn", "sklearn", "scikit learn"],
"selenium": ["selenium", "selenium webdriver", "selenium grid"],
"seo": [
"seo",
"search engine optimization",
"search engine optimisation",
"on-page seo",
"off-page seo",
"technical seo",
],
"sequelize": ["sequelize"],
"siem": [
"siem",
"security information and event management",
"soc analyst",
"security operations center",
"soc",
],
"six sigma": ["six sigma", "6 sigma", "dmaic", "black belt", "green belt"],
"sketch": ["sketch", "sketch app"],
"sla management": [
"sla",
"sla management",
"service level agreement",
"service level",
"response time",
],
"snowflake": ["snowflake", "snowflake data warehouse"],
"soap": ["soap", "soap api", "soap web service", "wsdl", "web service"],
"soc 2": ["soc 2", "soc2", "aicpa soc"],
"social media": [
"social media",
"social media management",
"instagram",
"facebook marketing",
"tiktok",
"linkedin marketing",
"twitter",
],
"solidworks": ["solidworks", "solid works", "parametric design"],
"special education": [
"special education",
"special needs",
"sen",
"inclusive education",
"learning disabilities",
],
"splunk": ["splunk", "splunk siem"],
"spring boot": [
"spring boot",
"spring",
"spring framework",
"spring mvc",
"spring security",
"spring cloud",
],
"sql": ["sql", "pl/sql", "t-sql", "structured query language"],
"sqlalchemy": ["sqlalchemy", "sql alchemy"],
"sqlite": ["sqlite"],
"sre": [
"sre",
"site reliability engineering",
"site reliability",
"reliability engineering",
],
"ssl tls": [
"ssl",
"tls",
"ssl/tls",
"certificates",
"pki",
"public key infrastructure",
],
"stakeholder management": [
"stakeholder management",
"stakeholder engagement",
"stakeholder communication",
"stakeholder relations",
],
"statistics": ["statistics", "statistical analysis", "spss", "stata", "minitab"],
"stem": ["stem", "science technology engineering mathematics", "stem education"],
"storage": [
"storage",
"nas",
"san",
"object storage",
"block storage",
"aws s3",
"azure blob",
"gcs",
],
"streaming": [
"real-time streaming",
"stream processing",
"event-driven architecture",
"cdc",
"change data capture",
],
"student counseling": [
"student counseling",
"student counselling",
"academic advising",
"student welfare",
"pastoral care",
],
"succession planning": [
"succession planning",
"talent pipeline",
"leadership development",
],
"supply chain": [
"supply chain",
"supply chain management",
"scm",
"end-to-end supply chain",
],
"surgical": [
"surgical",
"theatre",
"perioperative",
"scrub nurse",
"surgical technician",
],
"surveying": [
"surveying",
"land surveying",
"total station",
"gps surveying",
"topographic",
],
"svelte": ["svelte", "sveltekit"],
"swift": ["swift", "swiftui", "swift ui"],
"symfony": ["symfony"],
"tableau": ["tableau", "tableau desktop"],
"tailwind css": ["tailwind", "tailwind css"],
"talent management": [
"talent management",
"talent strategy",
"high potential",
"talent review",
],
"taxation": [
"taxation",
"tax",
"tax compliance",
"tax preparation",
"tax returns",
"vat",
"corporation tax",
"withholding tax",
],
"tdd": [
"tdd",
"test driven development",
"test-driven development",
"unit testing",
"integration testing",
"end-to-end testing",
],
"teaching": [
"teaching",
"instruction",
"lecturing",
"tutoring",
"classroom teaching",
"pedagogy",
],
"teamwork": [
"teamwork",
"team work",
"team player",
"collaboration",
"cross-functional",
],
"technical writing": [
"technical writing",
"technical documentation",
"user manuals",
"sop writing",
"standard operating procedures",
],
"telemedicine": [
"telemedicine",
"telehealth",
"remote patient monitoring",
"virtual clinic",
],
"tensorflow": ["tensorflow", "tf", "keras"],
"terraform": [
"terraform",
"infrastructure as code",
"iac",
"hashicorp terraform",
"pulumi",
],
"testng": ["testng", "test ng"],
"threat intelligence": [
"threat intelligence",
"threat hunting",
"cyber threat",
"ioc",
"indicators of compromise",
],
"ticketing systems": [
"ticketing",
"zendesk",
"freshdesk",
"jira service desk",
"servicenow",
"remedy",
"ivanti",
],
"time management": [
"time management",
"prioritization",
"prioritisation",
"multitasking",
"deadline management",
],
"trade finance": [
"trade finance",
"letter of credit",
"documentary credit",
"trade operations",
],
"training and development": [
"learning and development",
"l&d",
"employee training",
"talent development",
"workforce training",
"capability building",
],
"typeorm": ["typeorm", "type orm"],
"typescript": ["typescript", "ts"],
"ui design": ["ui design", "user interface design", "interface design"],
"ux design": [
"ux design",
"user experience design",
"ux research",
"usability",
"ux writing",
],
"vagrant": ["vagrant"],
"video editing": [
"video editing",
"video production",
"final cut pro",
"davinci resolve",
],
"virtual assistant": [
"virtual assistant",
"executive assistant",
"pa",
"personal assistant",
"administrative assistant",
],
"virtualization": [
"virtualization",
"virtualisation",
"virtual machine",
"vm management",
"kvm",
"proxmox",
],
"vmware": ["vmware", "vsphere", "vcenter", "esxi", "vsan", "vmware workstation"],
"vue": ["vue", "vue.js", "vuejs", "vue 3", "nuxt", "nuxt.js"],
"vuex": ["vuex"],
"vulnerability management": [
"vulnerability management",
"vulnerability assessment",
"vulnerability scanning",
"nessus",
"qualys",
],
"warehouse": [
"warehouse",
"stockroom",
"store room",
"stores",
"warehouse operations",
],
"warehouse management": [
"warehouse management system",
"wms",
"inventory system",
"warehouse operations",
],
"wealth management": [
"wealth management",
"asset management",
"portfolio management",
"investment management",
"fund management",
],
"windows server": [
"windows server",
"iis",
"windows administration",
"active directory",
],
"wireframing": ["wireframe", "wireframing", "low-fidelity"],
"workforce planning": [
"workforce planning",
"headcount planning",
"org design",
"organisational design",
],
"xero": ["xero"],
"xunit": ["xunit", "x unit"],
"zero trust": ["zero trust", "zero trust network", "ztna"],
"zustand": ["zustand"],
}
DEFAULT_CATEGORY_SKILLS: Dict[str, Set[str]] = {
"Admin & Office": {
"data entry",
"erp",
"customer service",
"scheduling",
"communication",
"microsoft powerpoint",
"warehouse",
"time management",
"inventory management",
"reporting",
"virtual assistant",
"microsoft office",
"google workspace",
"minute taking",
"attention to detail",
"teamwork",
"filing",
"adaptability",
"microsoft excel",
"microsoft word",
},
"Construction & Engineering": {
"electrical engineering",
"mechanical engineering",
"health and safety",
"surveying",
"primavera",
"civil 3d",
"autocad",
"revit",
"civil engineering",
"ms project",
"solidworks",
},
"Customer Support": {
"help desk",
"reporting",
"sla management",
"customer satisfaction",
"crm",
"communication",
"customer service",
"live chat",
"call centre",
"ticketing systems",
},
"Cybersecurity": {
"ssl tls",
"bash",
"iso 27001",
"networking",
"zero trust",
"pci dss",
"devsecops",
"python",
"nist",
"penetration testing",
"siem",
"soc 2",
"vulnerability management",
"threat intelligence",
"cloud security",
"linux",
"owasp",
"gdpr",
},
"Data & Analytics": {
"scikit-learn",
"power bi",
"data analysis",
"elasticsearch",
"python",
"r",
"pytorch",
"sql",
"tensorflow",
"tableau",
"data science",
"nlp",
"excel analytics",
"machine learning",
"statistics",
"data visualization",
"computer vision",
"qlik",
"microsoft excel",
"looker",
"generative ai",
},
"Data Engineering": {
"data modeling",
"aws",
"azure",
"redis",
"kafka",
"python",
"data pipeline",
"sql",
"data governance",
"dbt",
"postgresql",
"streaming",
"apache hadoop",
"databricks",
"gcp",
"apache spark",
"snowflake",
"apache airflow",
"mongodb",
"elasticsearch",
},
"Design & Creative": {
"canva",
"adobe xd",
"video editing",
"sketch",
"adobe illustrator",
"prototyping",
"figma",
"adobe photoshop",
"photography",
"ui design",
"wireframing",
"adobe after effects",
"miro",
"brand design",
"adobe premiere",
"ux design",
"invision",
"graphic design",
"3d design",
},
"Education & Training": {
"distance learning",
"mentoring",
"communication",
"curriculum development",
"research",
"lms",
"special education",
"e-learning",
"assessment",
"stem",
"teaching",
"classroom management",
"student counseling",
},
"Finance & Accounting": {
"credit analysis",
"derivatives",
"core banking",
"taxation",
"cash flow",
"financial compliance",
"wealth management",
"xero",
"quickbooks",
"accounting",
"risk management",
"financial analysis",
"cfa",
"sage",
"aml",
"payroll",
"acca",
"kyc",
"ifrs",
"sap",
"auditing",
"accounts payable",
"gaap",
"budgeting",
"bloomberg",
"microsoft excel",
"accounts receivable",
"trade finance",
"actuarial",
"reconciliation",
},
"Healthcare": {
"first aid",
"infection control",
"icd coding",
"mental health",
"radiology",
"physiotherapy",
"nursing",
"medical records",
"ehr systems",
"laboratory",
"clinical trials",
"pharmacy",
"telemedicine",
"nutrition",
"clinical skills",
"midwifery",
"patient care",
"surgical",
"hl7",
},
"Hospitality & Property": {
"property management",
"housekeeping",
"real estate",
"customer service",
"communication",
"hospitality management",
"event management",
"food safety",
"revenue management",
},
"Human Resources": {
"employee relations",
"microsoft excel",
"communication",
"workforce planning",
"recruitment",
"succession planning",
"talent management",
"hris",
"job analysis",
"onboarding",
"training and development",
"performance management",
"culture",
"hr compliance",
"compensation",
"dei",
},
"Infrastructure & Cloud": {
"bash",
"datadog",
"splunk",
"ansible",
"backup recovery",
"ci/cd",
"networking",
"sre",
"linux",
"windows server",
"itil",
"opentelemetry",
"high availability",
"hyper-v",
"powershell",
"storage",
"dynatrace",
"kubernetes",
"chef",
"prometheus",
"new relic",
"vagrant",
"vmware",
"cloudwatch",
"aws",
"azure",
"grafana",
"terraform",
"puppet",
"git",
"cisco",
"gcp",
"docker",
"virtualization",
"active directory",
"elk stack",
},
"Legal": {
"legal compliance",
"conveyancing",
"legal research",
"intellectual property",
"legal writing",
"litigation",
"mergers acquisitions",
"arbitration",
"corporate law",
"due diligence",
"contract drafting",
"employment law",
"legal tech",
},
"Marketing & Growth": {
"cro",
"content writing",
"affiliate marketing",
"crm",
"digital marketing",
"marketing automation",
"google ads",
"email marketing",
"google analytics",
"brand management",
"seo",
"social media",
"facebook ads",
"pr",
},
"Operations & Logistics": {
"3pl",
"supply chain",
"erp",
"demand planning",
"reporting",
"health and safety",
"customs",
"warehouse management",
"lean",
"logistics",
"procurement",
"six sigma",
"quality control",
"inventory management",
"project management",
},
"Sales & Business Dev": {
"revenue operations",
"business analysis",
"customer service",
"crm",
"communication",
"lead generation",
"negotiation",
"account management",
"partnership",
"sales",
"customer success",
"retail",
},
"Software Engineering": {
"bash",
"zustand",
"firestore",
"redux",
"flask",
"clickhouse",
"angular",
"css",
"jira",
"entity framework",
"dotnet",
"ngrx",
"sql",
"mqtt",
"swift",
"agile",
"ionic",
"neo4j",
"confluence",
"nlp",
"nunit",
"testng",
"junit",
"context api",
"kubernetes",
"postgresql",
"cypress",
"dynamodb",
"symfony",
"jmeter",
"elasticsearch",
"generative ai",
"jetpack compose",
"f#",
"ios",
"grpc",
"message broker",
"aws",
"scikit-learn",
"cassandra",
"kafka",
"prisma",
"python",
"graphql",
"rust",
"go",
"xunit",
"html",
"terraform",
"typescript",
"ruby",
"selenium",
"hibernate",
"machine learning",
"spring boot",
"gcp",
"oauth",
"mocha",
"computer vision",
"postman",
"microservices",
"rabbitmq",
"sqlalchemy",
"c",
"nestjs",
"ci/cd",
"ansible",
"influxdb",
"scala",
"android",
"flutter",
"redis",
"java",
"rails",
"fastapi",
"rest api",
"powershell",
"django",
"c++",
"jasmine",
"mobx",
"pytest",
"react",
"vuex",
"typeorm",
"laravel",
"php",
"elixir",
"dart",
"next.js",
"groovy",
"oracle db",
"mysql",
"openapi",
"robot framework",
"azure",
"javascript",
"kotlin",
"jwt",
"jest",
"svelte",
"pytorch",
"tdd",
"tensorflow",
"cucumber",
"pinia",
"mariadb",
"node.js",
"c#",
"saml",
"haskell",
"git",
"sequelize",
"tailwind css",
"mongodb",
"kanban",
"soap",
"sqlite",
"vue",
"docker",
},
}
SOFT_SKILL_KEYS: Set[str] = {
"public speaking",
"customer service",
"scheduling",
"communication",
"problem solving",
"leadership",
"negotiation",
"time management",
"emotional intelligence",
"stakeholder management",
"onboarding",
"decision making",
"reporting",
"customer satisfaction",
"mentoring",
"training and development",
"attention to detail",
"teamwork",
"filing",
"adaptability",
"conflict resolution",
}
SENIORITY_PATTERNS: Dict[str, List[str]] = {
"intern": [
"intern",
"internship",
"trainee",
"industrial attachment",
"attachment student",
"graduate trainee",
"pupil",
"cadet",
],
"junior": [
"junior",
"entry level",
"entry-level",
"assistant",
"fresh graduate",
"graduate",
"associate",
"junior officer",
"junior analyst",
"junior developer",
"junior engineer",
],
"mid": [
"mid",
"middle",
"officer",
"specialist",
"developer",
"designer",
"engineer",
"analyst",
"coordinator",
"technician",
"executive",
"consultant",
"representative",
],
"senior": [
"senior",
"lead",
"manager",
"head",
"principal",
"architect",
"director",
"vp",
"vice president",
"chief",
"cto",
"ceo",
"coo",
"cfo",
"superintendent",
"supervisor",
"managing",
"staff engineer",
"distinguished",
"fellow",
],
}
SENIORITY_ORDER: Dict[str, int] = {"intern": 0, "junior": 1, "mid": 2, "senior": 3}
_TECH_LOC_BLACKLIST: Set[str] = {
"splunk",
"nestjs",
"ansible",
"airflow",
"scala",
"android",
"nginx",
"node",
"postgres",
"flask",
"angular",
"flutter",
"golang",
"redis",
"bitbucket",
"mongo",
"jira",
"java",
"rails",
"spring",
"sql",
"swift",
"gitlab",
"ionic",
"django",
"confluence",
"oracle",
"react",
"kibana",
"kubernetes",
"github",
"laravel",
"dart",
"symfony",
"mysql",
"spark",
"express",
"aws",
"azure",
"kotlin",
"kafka",
"python",
"istio",
"figma",
"rust",
"svelte",
"grafana",
"terraform",
"apache",
"ruby",
"windows",
"nextjs",
"helm",
"boot",
"puppet",
"hadoop",
"gcp",
"jenkins",
"vue",
"linux",
"docker",
"slack",
}
TITLE_ROLE_WORDS: List[str] = [
"developer",
"engineer",
"programmer",
"architect",
"devops",
"officer",
"designer",
"assistant",
"accountant",
"manager",
"analyst",
"specialist",
"coordinator",
"director",
"executive",
"lead",
"scientist",
"strategist",
"advisor",
"trainer",
"consultant",
"researcher",
"technician",
"supervisor",
"associate",
"administrator",
"nurse",
"doctor",
"physician",
"pharmacist",
"therapist",
"clinician",
"midwife",
"radiographer",
"physiotherapist",
"auditor",
"bookkeeper",
"controller",
"treasurer",
"actuary",
"recruiter",
"salesperson",
"representative",
"agent",
"broker",
"teacher",
"instructor",
"lecturer",
"professor",
"tutor",
"admin",
"intern",
"student",
"cashier",
"marketing",
"procurement",
"buyer",
"dispatcher",
"operator",
"superintendent",
]
TITLE_PHRASE_RE = re.compile(
r"\b((?:(?:full[\s\-]?stack|front[\s\-]?end|back[\s\-]?end|senior|junior|lead|"
r"mid[\s\-]?level|entry[\s\-]?level|chief|head\s+of|staff|principal)?\s+)?(?:[\w\-]+\s){0,3}"
r"(?:developer|engineer|designer|analyst|manager|specialist|consultant|architect|"
r"programmer|researcher|officer|director|executive|scientist|trainer|accountant|"
r"nurse|doctor|pharmacist|therapist|teacher|instructor|lecturer|recruiter|"
r"coordinator|supervisor|administrator|technician|advisor|auditor|bookkeeper|"
r"physiotherapist|midwife|clinician|representative|salesperson|buyer|dispatcher|"
r"actuary|superintendent|operator))\b",
re.IGNORECASE,
)
# ---------------------------------------------------------------------------
# Normalisation helpers
# ---------------------------------------------------------------------------
def _normalize(text: str) -> str:
text = (text or "").lower()
text = text.replace("/", " ")
text = text.replace("_", " ")
text = re.sub(r"[^a-z0-9+.#\s-]", " ", text)
return re.sub(r"\s+", " ", text).strip()
_ALIAS_REGEX_CACHE = {}
def _has_alias(text_norm: str, alias: str) -> bool:
alias_norm = _normalize(alias)
if not alias_norm:
return False
pattern = _ALIAS_REGEX_CACHE.get(alias_norm)
if pattern is None:
pattern = re.compile(rf"(?<![a-z0-9]){re.escape(alias_norm)}(?![a-z0-9])")
_ALIAS_REGEX_CACHE[alias_norm] = pattern
return pattern.search(text_norm) is not None
# ---------------------------------------------------------------------------
# Skill extraction
# ---------------------------------------------------------------------------
_SKILL_PATTERN = None
_ALIAS_TO_CANONICAL = {}
def _get_skill_pattern():
global _SKILL_PATTERN, _ALIAS_TO_CANONICAL
if _SKILL_PATTERN is not None:
return _SKILL_PATTERN, _ALIAS_TO_CANONICAL
_ALIAS_TO_CANONICAL = {}
alias_list = []
for canon, ali_list in DEFAULT_SKILL_ALIASES.items():
for a in ali_list:
a_norm = _normalize(a)
if a_norm:
if a_norm not in _ALIAS_TO_CANONICAL:
_ALIAS_TO_CANONICAL[a_norm] = canon
alias_list.append(a_norm)
alias_list.sort(key=len, reverse=True)
pattern_str = "|".join(map(re.escape, alias_list))
_SKILL_PATTERN = re.compile(rf"(?<![a-z0-9])({pattern_str})(?![a-z0-9])")
return _SKILL_PATTERN, _ALIAS_TO_CANONICAL
def _extract_skills(
text: str,
aliases: Optional[Dict[str, List[str]]] = None,
) -> Set[str]:
text_norm = _normalize(text)
if aliases is None or aliases is DEFAULT_SKILL_ALIASES:
pattern, alias_map = _get_skill_pattern()
return {alias_map[match.group(1)] for match in pattern.finditer(text_norm)}
return {
canon
for canon, ali_list in aliases.items()
if any(_has_alias(text_norm, a) for a in ali_list)
}
# ---------------------------------------------------------------------------
# Category inference
# ---------------------------------------------------------------------------
_CATEGORY_TITLE_KEYWORDS: Dict[str, List[str]] = {
"Admin & Office": [
"admin",
"office",
"stock",
"inventory",
"warehouse",
"cashier",
"receptionist",
"clerk",
],
"Software Engineering": [
"developer",
"engineer",
"programmer",
"frontend",
"backend",
"full stack",
"mobile",
"fullstack",
"software",
"devops",
],
"Infrastructure & Cloud": [
"infrastructure",
"cloud",
"sre",
"network",
"system administrator",
"sysadmin",
"it administrator",
"cloud engineer",
],
"Cybersecurity": [
"security",
"cybersecurity",
"cyber",
"infosec",
"soc analyst",
"penetration tester",
"ethical hacker",
],
"Design & Creative": [
"designer",
"ui",
"ux",
"creative",
"graphic",
"visual",
"illustrator",
"photographer",
],
"Marketing & Growth": [
"marketing",
"seo",
"content",
"social media",
"growth",
"digital",
"brand",
],
"Data & Analytics": [
"data analyst",
"analytics",
"machine learning",
"ai",
"scientist",
"bi analyst",
"business intelligence",
],
"Data Engineering": [
"data engineer",
"etl",
"data pipeline",
"data architect",
"analytics engineer",
],
"Finance & Accounting": [
"accountant",
"finance",
"accounting",
"auditor",
"bookkeeper",
"treasurer",
"payroll",
"tax",
"financial",
],
"Human Resources": [
"hr",
"human resource",
"recruitment",
"recruiter",
"talent",
"people",
"hris",
],
"Sales & Business Dev": [
"sales",
"business development",
"account manager",
"sales executive",
"representative",
],
"Healthcare": [
"nurse",
"doctor",
"clinical",
"medical",
"pharmacy",
"patient",
"health",
"therapist",
"midwife",
],
"Legal": [
"legal",
"lawyer",
"attorney",
"advocate",
"counsel",
"paralegal",
"solicitor",
],
"Education & Training": [
"teacher",
"lecturer",
"tutor",
"instructor",
"educator",
"academic",
"trainer",
"professor",
],
"Operations & Logistics": [
"logistics",
"supply chain",
"procurement",
"operations",
"fleet",
"warehouse manager",
"quality",
],
"Customer Support": [
"customer support",
"help desk",
"call centre",
"contact centre",
"customer care",
],
"Construction & Engineering": [
"civil",
"structural",
"mechanical",
"electrical",
"quantity surveyor",
"site engineer",
"construction",
],
"Hospitality & Property": [
"hotel",
"hospitality",
"property",
"real estate",
"housekeeping",
"f&b",
"restaurant",
],
}
def _infer_category(
text: str,
skills: Set[str],
category_skills: Optional[Dict[str, Set[str]]] = None,
) -> str:
if category_skills is None:
category_skills = DEFAULT_CATEGORY_SKILLS
text_norm = _normalize(text)
best, best_score = "Other", 0.0
for cat, cat_skills in category_skills.items():
skill_overlap = len(skills & cat_skills)
kw_hits = sum(
1 for kw in _CATEGORY_TITLE_KEYWORDS.get(cat, []) if kw in text_norm
)
score = skill_overlap * 2.0 + kw_hits * 1.25
if score > best_score:
best_score, best = score, cat
return best
# ---------------------------------------------------------------------------
# Years-of-experience extraction
# ---------------------------------------------------------------------------
def _extract_years(text: str) -> int:
text_norm = _normalize(text)
patterns = [
r"(\d+)\+?\s*(?:years|year|yrs|yr)\s*(?:of)?\s*(?:experience|exp)",
r"experience\s*(?:of)?\s*(\d+)\+?\s*(?:years|year|yrs|yr)",
r"(\d+)\+?\s*(?:years|year|yrs|yr)\s+(?:in\s+)?(?:the\s+)?(?:industry|field|software|development|practice|profession)",
]
years = [int(m.group(1)) for p in patterns for m in re.finditer(p, text_norm)]
return max(years) if years else 0
# ---------------------------------------------------------------------------
# Years from parsed experience date ranges
# ---------------------------------------------------------------------------
_MONTH_SHORT: Dict[str, int] = {
"jan": 1,
"feb": 2,
"mar": 3,
"apr": 4,
"may": 5,
"jun": 6,
"jul": 7,
"aug": 8,
"sep": 9,
"oct": 10,
"nov": 11,
"dec": 12,
}
def _parse_date_token(token: str) -> Optional[_dt]:
token = token.strip().lower()
if re.match(r"present|current|now|ongoing|till\s*date|to\s*date", token):
return _dt.now()
m = re.match(r"([a-z]+)\s+(\d{4})", token)
if m:
month = _MONTH_SHORT.get(m.group(1)[:3], 1)
try:
return _dt(int(m.group(2)), month, 1)
except ValueError:
return None
m = re.match(r"(\d{4})", token)
if m:
return _dt(int(m.group(1)), 6, 1)
return None
def compute_years_from_experience(entries: List[Dict[str, Any]]) -> int:
"""Sum date-range durations across all experience entries → total years."""
total_months = 0
for entry in entries:
period = (entry.get("period") or "").strip()
if not period:
continue
parts = re.split(r"\s*[-–—]\s*", period, maxsplit=1)
if len(parts) != 2:
continue
start = _parse_date_token(parts[0])
end = _parse_date_token(parts[1])
if start and end and end > start:
months = (end.year - start.year) * 12 + (end.month - start.month)
total_months += max(0, months)
return total_months // 12
# ---------------------------------------------------------------------------
# Seniority detection
# ---------------------------------------------------------------------------
def _detect_seniority(text: str, *, is_cv: bool = False) -> str:
"""Detect seniority: title zone first, then body (guards against
false positives from advice/boilerplate text in job descriptions)."""
title_zone = text[:300]
title_norm = _normalize(title_zone)
text_norm = _normalize(text)
order = (
["senior", "junior", "intern", "mid"]
if is_cv
else ["senior", "intern", "junior", "mid"]
)
for level in order:
if any(_has_alias(title_norm, term) for term in SENIORITY_PATTERNS[level]):
return level
for level in order:
if level == "senior" and not is_cv:
continue # already checked title zone; skip body re-check to avoid boilerplate inflation
if any(_has_alias(text_norm, term) for term in SENIORITY_PATTERNS[level]):
return level
years = _extract_years(text)
if years >= 7:
return "senior"
if years >= 3:
return "mid"
return "junior" if is_cv else "mid"
def refine_seniority_with_years(detected: str, years: int) -> str:
if detected == "mid":
if years >= 7:
return "senior"
if years >= 3:
return "mid"
if 0 < years < 2:
return "junior"
return detected
if detected == "intern" and years >= 2:
if years >= 7:
return "senior"
if years >= 3:
return "mid"
return "junior"
return detected
# ---------------------------------------------------------------------------
# CV title extractor
# ---------------------------------------------------------------------------
def _extract_cv_title(cv_text: str, clean_line_fn=None) -> str:
def _simple_clean(raw: str) -> str:
line = re.sub(r"^#+\s*", "", raw)
line = re.sub(r"\*{1,3}|\_{1,2}|`", "", line)
return re.sub(r"\s+", " ", line).strip()
clean = clean_line_fn or _simple_clean
clean_lines = [clean(x) for x in cv_text.splitlines() if x.strip()]
_role_wb_re = re.compile(
r"\b(" + "|".join(re.escape(w) for w in TITLE_ROLE_WORDS) + r")\b",
re.IGNORECASE,
)
for line in clean_lines[1:30]:
stripped = line.strip("| \t")
if (
len(stripped) <= 80
and bool(_role_wb_re.search(stripped))
and "|" not in stripped
and not re.search(r"[@http]", stripped)
and not re.search(
r"\d{4}\s*[-–]\s*(?:\d{4}|present)", stripped, re.IGNORECASE
)
):
return stripped.title()
_SECTION_HDR_RE = re.compile(
r"^(career\s*summary|professional\s*summary|executive\s*summary|"
r"personal\s*statement|career\s*objective|professional\s*profile|"
r"personal\s*profile|about\s*me|work\s*experience|"
r"professional\s*experience|technical\s*skills?|core\s*skills?|"
r"key\s*skills?|educational?\s*background|academic\s*background|"
r"employment\s*history|career\s*history|skills?\s*&?\s*tools?|"
r"skills?|summary|profile|experience|education|projects?|awards?|"
r"achievements?|contact|links?|certifications?)$",
re.IGNORECASE,
)
non_header = [
ln
for ln in clean_lines[:40]
if not _SECTION_HDR_RE.match(re.sub(r"[^a-zA-Z\s&]", " ", ln).strip())
]
head = ". ".join(non_header)[:1000]
m = TITLE_PHRASE_RE.search(head)
if m:
return m.group(1).strip().title()
return "Unknown"