kernelmind/Resume-Screener-API
0
1"""2ScreenerEngine - The Core Logic for Resume Screening AI.3 4This module implements the "2-Layer System" architecture:5Layer 1: Deterministic Information Extraction (Gemini + Pydantic)6Layer 2: Constraint & Similarity Scoring (FAISS + Logic)7 8Tech Stack:9- LangChain: For orchestration and prompt management10- Google Gemini: For reasoning and extraction11- FAISS: For vector similarity search12- Pydantic: For robust data validation13"""14 15import os16from typing import List, Optional17import math18 19from pydantic import BaseModel, Field20from langchain_google_genai import ChatGoogleGenerativeAI, GoogleGenerativeAIEmbeddings21from langchain.prompts import PromptTemplate22from langchain_core.output_parsers import PydanticOutputParser23from langchain_community.vectorstores import FAISS24from langchain.docstore.document import Document25 26# --- Layer 1: Data Models (Pydantic) ---27 28class ExtractedResumeData(BaseModel):29 """Structured data extracted from a resume text."""30 anonymized_id: str = Field(description="A unique hash or ID for the candidate, removing PII like name/email.")31 total_years_experience: float = Field(description="Total professional years of experience found in the resume.")32 degree_level: str = Field(description="Highest degree obtained: 'Bachelor', 'Master', 'PhD', or 'None'.")33 skills_list: List[str] = Field(description="List of technical and soft skills extracted.")34 summary: str = Field(description="A 2-sentence summary of the candidate's profile.")35 36class JobRequirements(BaseModel):37 """Structured requirements extracted from a Job Description."""38 required_years_experience: float = Field(description="Minimum years of experience required.")39 required_degree: str = Field(description="Minimum degree required: 'Bachelor', 'Master', 'PhD', or 'None'.")40 key_skills: List[str] = Field(description="List of must-have skills for the role.")41 42class ScreeningResult(BaseModel):43 """Final output of the screening process."""44 final_score: float = Field(description="Final match score (0-100).")45 is_qualified: bool = Field(description="Whether the candidate met hard constraints.")46 extracted_data: ExtractedResumeData47 explanation: str = Field(description="Reasoning for the score.")48 49# --- The Screener Engine Class ---50 51class ScreenerEngine:52 def __init__(self, google_api_key: Optional[str] = None):53 """Initialize the engine with Gemini and Embeddings."""54 self.api_key = google_api_key or os.getenv("GEMINI_API_KEY")55 if not self.api_key:56 raise ValueError("GEMINI_API_KEY not found in environment.")57 58 # Initialize LLM (Gemini 1.5 Flash is efficient for this)59 self.llm = ChatGoogleGenerativeAI(60 model="gemini-1.5-flash",61 google_api_key=self.api_key,62 temperature=0.0, # Deterministic output63 convert_system_message_to_human=True64 )65 66 # Initialize Embeddings67 self.embeddings = GoogleGenerativeAIEmbeddings(68 model="models/embedding-001",69 google_api_key=self.api_key70 )71 72 # Parsers73 self.resume_parser = PydanticOutputParser(pydantic_object=ExtractedResumeData)74 self.jd_parser = PydanticOutputParser(pydantic_object=JobRequirements)75 76 def _extract_resume_data(self, resume_text: str) -> ExtractedResumeData:77 """Layer 1: Extract structured data from resume using Gemini."""78 prompt = PromptTemplate(79 template="""80 You are an expert Resume Parser. Your goal is to extract structured data for fair hiring.81 82 INSTRUCTIONS:83 1. Anonymize the candidate (generate a hash ID like 'CAND-123' if not present).84 2. Calculate total years of experience precisely based on work history dates.85 3. Normalize degree to: 'Bachelor', 'Master', 'PhD', or 'None'.86 4. Extract all technical skills.87 5. Provide a brief 2-sentence summary.88 89 RESUME TEXT:90 {resume_text}91 92 OUTPUT FORMAT:93 {format_instructions}94 """,95 input_variables=["resume_text"],96 partial_variables={"format_instructions": self.resume_parser.get_format_instructions()}97 )98 99 chain = prompt | self.llm | self.resume_parser100 return chain.invoke({"resume_text": resume_text})101 102 def _extract_jd_requirements(self, jd_text: str) -> JobRequirements:103 """Extract hard constraints from JD."""104 prompt = PromptTemplate(105 template="""106 Extract the minimum requirements from this Job Description.107 108 JOB DESCRIPTION:109 {jd_text}110 111 OUTPUT FORMAT:112 {format_instructions}113 """,114 input_variables=["jd_text"],115 partial_variables={"format_instructions": self.jd_parser.get_format_instructions()}116 )117 118 chain = prompt | self.llm | self.jd_parser119 return chain.invoke({"jd_text": jd_text})120 121 def _calculate_vector_similarity(self, resume_text: str, jd_text: str) -> float:122 """Compute Cosine Similarity using FAISS."""123 # Create vector store with just the JD124 vector_store = FAISS.from_texts([jd_text], self.embeddings)125 126 # Search for the Resume in the JD space (conceptually checking semantic overlap)127 # FAISS returns L2 distance by default, but we can standard usage for similarity128 # Or simpler: embed both and dot product.129 130 resume_vec = self.embeddings.embed_query(resume_text)131 jd_vec = self.embeddings.embed_query(jd_text)132 133 # Manual Cosine Similarity134 dot_product = sum(a*b for a, b in zip(resume_vec, jd_vec))135 norm_a = math.sqrt(sum(a*a for a in resume_vec))136 norm_b = math.sqrt(sum(b*b for b in jd_vec))137 138 if norm_a == 0 or norm_b == 0:139 return 0.0140 141 similarity = dot_product / (norm_a * norm_b)142 return max(0.0, min(1.0, similarity)) # Clamp 0-1143 144 def screen_resume(self, resume_text: str, jd_text: str) -> ScreeningResult:145 """146 Main Entry Point: The 2-Layer Screening Process.147 """148 # --- Layer 1: Information Extraction ---149 resume_data = self._extract_resume_data(resume_text)150 jd_constraints = self._extract_jd_requirements(jd_text)151 152 # --- Layer 2: Constraint & Similarity Scorer ---153 154 score_penalty = 1.0 # Multiplier (1.0 = no penalty)155 explanation_prefix = ""156 is_qualified = True157 158 # 1. Constraint Check (Experience)159 if resume_data.total_years_experience < jd_constraints.required_years_experience:160 score_penalty *= 0.4 # Hard penalty: Max possible score becomes 40%161 is_qualified = False162 explanation_prefix = f"Does not meet experience requirement ({resume_data.total_years_experience} vs {jd_constraints.required_years_experience} years). "163 164 # 2. Constraint Check (Degree) - simplified logic165 degree_hierarchy = {"None": 0, "Bachelor": 1, "Master": 2, "PhD": 3}166 cand_deg_val = degree_hierarchy.get(resume_data.degree_level, 0)167 req_deg_val = degree_hierarchy.get(jd_constraints.required_degree, 0)168 169 if cand_deg_val < req_deg_val:170 score_penalty *= 0.8 # Mild penalty for degree mismatch171 explanation_prefix += f"Degree level ({resume_data.degree_level}) is below required ({jd_constraints.required_degree}). "172 173 # 3. Vector Similarity (Content Match)174 similarity_score = self._calculate_vector_similarity(resume_text, jd_text)175 176 # 4. Final Score Calculation177 # Base score from similarity (0-100) * Penalty Multiplier178 final_raw_score = similarity_score * 100 * score_penalty179 final_score = round(final_raw_score, 1)180 181 # 5. Generate Explanation via LLM (Rubric-based)182 explanation_prompt = PromptTemplate(183 template="""184 Explain this candidate's score ({score}/100) for the role.185 Constraint Issues: {constraints}186 Likely fit: {qualified}187 188 Resume Summary: {summary}189 Required Skills: {skills}190 191 Write a professional 2-sentence justification for the hiring manager.192 """,193 input_variables=["score", "constraints", "qualified", "summary", "skills"]194 )195 196 explanation_chain = explanation_prompt | self.llm197 explanation = explanation_chain.invoke({198 "score": final_score,199 "constraints": explanation_prefix if explanation_prefix else "None",200 "qualified": "Yes" if is_qualified else "No",201 "summary": resume_data.summary,202 "skills": ", ".join(jd_constraints.key_skills)203 }).content204 205 return ScreeningResult(206 final_score=final_score,207 is_qualified=is_qualified,208 extracted_data=resume_data,209 explanation=explanation210 )211 212# --- Usage Example (if run directly) ---213if __name__ == "__main__":214 # Mock Data215 resume_txt = "Jane Doe. Software Engineer with 3 years in Python and AWS. Bachelors in CS."216 jd_txt = "Senior Python Dev. 5+ years experience required. Masters preferred."217 218 try:219 engine = ScreenerEngine()220 result = engine.screen_resume(resume_txt, jd_txt)221 print(result.model_dump_json(indent=2))222 except Exception as e:223 print(f"Error: {e}")224 