Team Ai
Apppublic

kernelmind/Resume-Screener-API

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes
screener_engine.py224 linesDownload Raw Back to core
1"""2ScreenerEngine - The Core Logic for Resume Screening AI.3 4This module implements the "2-Layer System" architecture:5Layer 1: Deterministic Information Extraction (Gemini + Pydantic)6Layer 2: Constraint & Similarity Scoring (FAISS + Logic)7 8Tech Stack:9- LangChain: For orchestration and prompt management10- Google Gemini: For reasoning and extraction11- FAISS: For vector similarity search12- Pydantic: For robust data validation13"""14 15import os16from typing import List, Optional17import math18 19from pydantic import BaseModel, Field20from langchain_google_genai import ChatGoogleGenerativeAI, GoogleGenerativeAIEmbeddings21from langchain.prompts import PromptTemplate22from langchain_core.output_parsers import PydanticOutputParser23from langchain_community.vectorstores import FAISS24from langchain.docstore.document import Document25 26# --- Layer 1: Data Models (Pydantic) ---27 28class ExtractedResumeData(BaseModel):29    """Structured data extracted from a resume text."""30    anonymized_id: str = Field(description="A unique hash or ID for the candidate, removing PII like name/email.")31    total_years_experience: float = Field(description="Total professional years of experience found in the resume.")32    degree_level: str = Field(description="Highest degree obtained: 'Bachelor', 'Master', 'PhD', or 'None'.")33    skills_list: List[str] = Field(description="List of technical and soft skills extracted.")34    summary: str = Field(description="A 2-sentence summary of the candidate's profile.")35 36class JobRequirements(BaseModel):37    """Structured requirements extracted from a Job Description."""38    required_years_experience: float = Field(description="Minimum years of experience required.")39    required_degree: str = Field(description="Minimum degree required: 'Bachelor', 'Master', 'PhD', or 'None'.")40    key_skills: List[str] = Field(description="List of must-have skills for the role.")41 42class ScreeningResult(BaseModel):43    """Final output of the screening process."""44    final_score: float = Field(description="Final match score (0-100).")45    is_qualified: bool = Field(description="Whether the candidate met hard constraints.")46    extracted_data: ExtractedResumeData47    explanation: str = Field(description="Reasoning for the score.")48 49# --- The Screener Engine Class ---50 51class ScreenerEngine:52    def __init__(self, google_api_key: Optional[str] = None):53        """Initialize the engine with Gemini and Embeddings."""54        self.api_key = google_api_key or os.getenv("GEMINI_API_KEY")55        if not self.api_key:56            raise ValueError("GEMINI_API_KEY not found in environment.")57 58        # Initialize LLM (Gemini 1.5 Flash is efficient for this)59        self.llm = ChatGoogleGenerativeAI(60            model="gemini-1.5-flash",61            google_api_key=self.api_key,62            temperature=0.0, # Deterministic output63            convert_system_message_to_human=True64        )65 66        # Initialize Embeddings67        self.embeddings = GoogleGenerativeAIEmbeddings(68            model="models/embedding-001",69            google_api_key=self.api_key70        )71 72        # Parsers73        self.resume_parser = PydanticOutputParser(pydantic_object=ExtractedResumeData)74        self.jd_parser = PydanticOutputParser(pydantic_object=JobRequirements)75 76    def _extract_resume_data(self, resume_text: str) -> ExtractedResumeData:77        """Layer 1: Extract structured data from resume using Gemini."""78        prompt = PromptTemplate(79            template="""80            You are an expert Resume Parser. Your goal is to extract structured data for fair hiring.81            82            INSTRUCTIONS:83            1. Anonymize the candidate (generate a hash ID like 'CAND-123' if not present).84            2. Calculate total years of experience precisely based on work history dates.85            3. Normalize degree to: 'Bachelor', 'Master', 'PhD', or 'None'.86            4. Extract all technical skills.87            5. Provide a brief 2-sentence summary.88            89            RESUME TEXT:90            {resume_text}91            92            OUTPUT FORMAT:93            {format_instructions}94            """,95            input_variables=["resume_text"],96            partial_variables={"format_instructions": self.resume_parser.get_format_instructions()}97        )98 99        chain = prompt | self.llm | self.resume_parser100        return chain.invoke({"resume_text": resume_text})101 102    def _extract_jd_requirements(self, jd_text: str) -> JobRequirements:103        """Extract hard constraints from JD."""104        prompt = PromptTemplate(105            template="""106            Extract the minimum requirements from this Job Description.107            108            JOB DESCRIPTION:109            {jd_text}110            111            OUTPUT FORMAT:112            {format_instructions}113            """,114            input_variables=["jd_text"],115            partial_variables={"format_instructions": self.jd_parser.get_format_instructions()}116        )117        118        chain = prompt | self.llm | self.jd_parser119        return chain.invoke({"jd_text": jd_text})120 121    def _calculate_vector_similarity(self, resume_text: str, jd_text: str) -> float:122        """Compute Cosine Similarity using FAISS."""123        # Create vector store with just the JD124        vector_store = FAISS.from_texts([jd_text], self.embeddings)125        126        # Search for the Resume in the JD space (conceptually checking semantic overlap)127        # FAISS returns L2 distance by default, but we can standard usage for similarity128        # Or simpler: embed both and dot product.129        130        resume_vec = self.embeddings.embed_query(resume_text)131        jd_vec = self.embeddings.embed_query(jd_text)132        133        # Manual Cosine Similarity134        dot_product = sum(a*b for a, b in zip(resume_vec, jd_vec))135        norm_a = math.sqrt(sum(a*a for a in resume_vec))136        norm_b = math.sqrt(sum(b*b for b in jd_vec))137        138        if norm_a == 0 or norm_b == 0:139            return 0.0140            141        similarity = dot_product / (norm_a * norm_b)142        return max(0.0, min(1.0, similarity)) # Clamp 0-1143 144    def screen_resume(self, resume_text: str, jd_text: str) -> ScreeningResult:145        """146        Main Entry Point: The 2-Layer Screening Process.147        """148        # --- Layer 1: Information Extraction ---149        resume_data = self._extract_resume_data(resume_text)150        jd_constraints = self._extract_jd_requirements(jd_text)151        152        # --- Layer 2: Constraint & Similarity Scorer ---153        154        score_penalty = 1.0 # Multiplier (1.0 = no penalty)155        explanation_prefix = ""156        is_qualified = True157        158        # 1. Constraint Check (Experience)159        if resume_data.total_years_experience < jd_constraints.required_years_experience:160            score_penalty *= 0.4 # Hard penalty: Max possible score becomes 40%161            is_qualified = False162            explanation_prefix = f"Does not meet experience requirement ({resume_data.total_years_experience} vs {jd_constraints.required_years_experience} years). "163 164        # 2. Constraint Check (Degree) - simplified logic165        degree_hierarchy = {"None": 0, "Bachelor": 1, "Master": 2, "PhD": 3}166        cand_deg_val = degree_hierarchy.get(resume_data.degree_level, 0)167        req_deg_val = degree_hierarchy.get(jd_constraints.required_degree, 0)168        169        if cand_deg_val < req_deg_val:170            score_penalty *= 0.8 # Mild penalty for degree mismatch171            explanation_prefix += f"Degree level ({resume_data.degree_level}) is below required ({jd_constraints.required_degree}). "172 173        # 3. Vector Similarity (Content Match)174        similarity_score = self._calculate_vector_similarity(resume_text, jd_text)175        176        # 4. Final Score Calculation177        # Base score from similarity (0-100) * Penalty Multiplier178        final_raw_score = similarity_score * 100 * score_penalty179        final_score = round(final_raw_score, 1)180        181        # 5. Generate Explanation via LLM (Rubric-based)182        explanation_prompt = PromptTemplate(183            template="""184            Explain this candidate's score ({score}/100) for the role.185            Constraint Issues: {constraints}186            Likely fit: {qualified}187            188            Resume Summary: {summary}189            Required Skills: {skills}190            191            Write a professional 2-sentence justification for the hiring manager.192            """,193            input_variables=["score", "constraints", "qualified", "summary", "skills"]194        )195        196        explanation_chain = explanation_prompt | self.llm197        explanation = explanation_chain.invoke({198            "score": final_score,199            "constraints": explanation_prefix if explanation_prefix else "None",200            "qualified": "Yes" if is_qualified else "No",201            "summary": resume_data.summary,202            "skills": ", ".join(jd_constraints.key_skills)203        }).content204 205        return ScreeningResult(206            final_score=final_score,207            is_qualified=is_qualified,208            extracted_data=resume_data,209            explanation=explanation210        )211 212# --- Usage Example (if run directly) ---213if __name__ == "__main__":214    # Mock Data215    resume_txt = "Jane Doe. Software Engineer with 3 years in Python and AWS. Bachelors in CS."216    jd_txt = "Senior Python Dev. 5+ years experience required. Masters preferred."217    218    try:219        engine = ScreenerEngine()220        result = engine.screen_resume(resume_txt, jd_txt)221        print(result.model_dump_json(indent=2))222    except Exception as e:223        print(f"Error: {e}")224