Team Ai
Modelpublic

Prathmesh0001/interview-system

sourceHugging Facemitupdated 8mo agoView on Hugging Face
0likes
question_generator.py549 linesDownload Raw Back to root
1"""
2Question Generator Module
3Generates tailored interview questions based on job description and resume
4"""
5from dotenv import load_dotenv
6load_dotenv()
7import os
8from typing import List, Dict
9import json
10
11
12class QuestionGenerator:
13    """Generate interview questions using AI based on JD and resume"""
14    
15    def __init__(self, api_key: str = None, provider: str = "openai"):
16        """
17        Initialize question generator
18        
19        Args:
20            api_key: API key for AI service
21            provider: 'openai' or 'anthropic'
22        """
23        self.api_key = api_key or os.getenv('OPENAI_API_KEY')
24        self.provider = provider
25        
26        # Check if it's actually a Groq key disguised as an OpenAI key
27        is_groq = self.api_key and self.api_key.startswith("gsk_")
28
29        if provider == "openai" or is_groq:
30            try:
31                import openai
32                # If it's a Groq key, we change the base_url
33                base_url = "https://api.groq.com/openai/v1" if is_groq else None
34                self.client = openai.OpenAI(api_key=self.api_key, base_url=base_url)
35                
36                # Use Groq's free model if using Groq, otherwise default OpenAI
37                self.model_name = "llama3-8b-8192" if is_groq else "gpt-4o-mini"
38                self.available = True
39                print(f"Using {'Groq' if is_groq else 'OpenAI'} provider")
40            except ImportError:
41                print("OpenAI library not available. Install with: pip install openai")
42                self.available = False
43        elif provider == "anthropic":
44            try:
45                import anthropic
46                self.client = anthropic.Anthropic(api_key=self.api_key)
47                self.available = True
48            except ImportError:
49                print("Anthropic library not available. Install with: pip install anthropic")
50                self.available = False
51    
52    def generate_questions(self, job_description: str, resume: str, 
53                          num_questions: int = 5) -> List[Dict[str, str]]:
54        """
55        Generate interview questions tailored to JD and resume
56        
57        Args:
58            job_description: Job description text
59            resume: Resume text
60            num_questions: Number of questions to generate
61            
62        Returns:
63            List of question dictionaries with question, category, and difficulty
64        """
65        if not self.available or not self.api_key:
66            return self._generate_fallback_questions(job_description, num_questions)
67        
68        try:
69            prompt = self._create_prompt(job_description, resume, num_questions)
70            
71            if self.provider == "openai" or self.api_key.startswith("gsk_"):
72                response = self.client.chat.completions.create(
73                    model=self.model_name, # Dynamically uses llama3 for Groq
74                    messages=[
75                        {"role": "system", "content": "You are an expert technical interviewer who creates insightful, role-specific interview questions."},
76                        {"role": "user", "content": prompt}
77                    ],
78                    temperature=0.7
79                )
80                questions_text = response.choices[0].message.content
81                return self._parse_questions(questions_text)
82            else:  # anthropic
83                message = self.client.messages.create(
84                    model="claude-3-sonnet-20240229",
85                    max_tokens=2000,
86                    messages=[
87                        {"role": "user", "content": prompt}
88                    ]
89                )
90                questions_text = message.content[0].text
91            
92            return self._parse_questions(questions_text)
93            
94        except Exception as e:
95            print(f"Error generating questions with AI: {e}")
96            return self._generate_fallback_questions(job_description, num_questions)
97    
98    def _create_prompt(self, job_description: str, resume: str, num_questions: int) -> str:
99        """Create prompt for AI question generation"""
100        return f"""Based on the following job description and candidate resume, generate {num_questions} tailored interview questions.
101
102JOB DESCRIPTION:
103{job_description[:1500]}
104
105CANDIDATE RESUME:
106{resume[:1500]}
107
108Generate {num_questions} interview questions that:
1091. Are specific to the role and the candidate's background
1102. Test both technical skills and behavioral competencies
1113. Vary in difficulty from basic to advanced
1124. Cover different aspects of the role
113
114Format each question as JSON with the following structure:
115{{
116    "question": "The interview question",
117    "category": "technical/behavioral/situational",
118    "difficulty": "basic/intermediate/advanced",
119    "focus_area": "specific skill or competency being tested"
120}}
121
122Return ONLY a JSON array of {num_questions} questions, no additional text."""
123    
124    def _parse_questions(self, questions_text: str) -> List[Dict[str, str]]:
125        """Parse AI-generated questions from text"""
126        try:
127            # Try to extract JSON from the response
128            start_idx = questions_text.find('[')
129            end_idx = questions_text.rfind(']') + 1
130            
131            if start_idx != -1 and end_idx > start_idx:
132                json_str = questions_text[start_idx:end_idx]
133                questions = json.loads(json_str)
134                return questions
135            else:
136                raise ValueError("No JSON array found in response")
137                
138        except Exception as e:
139            print(f"Error parsing questions: {e}")
140            # Fallback: create questions from text lines
141            lines = [line.strip() for line in questions_text.split('\n') if line.strip()]
142            questions = []
143            for i, line in enumerate(lines[:5]):
144                if '?' in line or any(line.startswith(q) for q in ['Tell me', 'Describe', 'Explain', 'How']):
145                    questions.append({
146                        'question': line.strip('0123456789.-) '),
147                        'category': 'general',
148                        'difficulty': 'intermediate',
149                        'focus_area': 'general assessment'
150                    })
151            return questions if questions else self._generate_fallback_questions("", 5)
152    
153    def _generate_behavioral_questions(self, num_questions: int = 5) -> List[Dict[str, str]]:
154        """Generate standard behavioral questions"""
155        behavioral_pool = [
156            {
157                'question': "Tell me about yourself and walk me through your background.",
158                'category': 'behavioral',
159                'difficulty': 'basic',
160                'focus_area': 'self-introduction'
161            },
162            {
163                'question': "Describe a challenging situation you faced and how you handled it.",
164                'category': 'behavioral',
165                'difficulty': 'intermediate',
166                'focus_area': 'problem-solving'
167            },
168            {
169                'question': "Tell me about a time when you had to work under pressure or meet a tight deadline.",
170                'category': 'behavioral',
171                'difficulty': 'intermediate',
172                'focus_area': 'time management'
173            },
174            {
175                'question': "Describe a situation where you had to collaborate with a difficult team member.",
176                'category': 'behavioral',
177                'difficulty': 'intermediate',
178                'focus_area': 'teamwork'
179            },
180            {
181                'question': "Tell me about a time when you failed at something. How did you handle it?",
182                'category': 'behavioral',
183                'difficulty': 'advanced',
184                'focus_area': 'resilience'
185            },
186            {
187                'question': "Describe a situation where you had to learn something new quickly.",
188                'category': 'behavioral',
189                'difficulty': 'intermediate',
190                'focus_area': 'adaptability'
191            },
192            {
193                'question': "Tell me about a time when you took initiative on a project.",
194                'category': 'behavioral',
195                'difficulty': 'intermediate',
196                'focus_area': 'leadership'
197            },
198            {
199                'question': "Describe a situation where you had to make a difficult decision.",
200                'category': 'behavioral',
201                'difficulty': 'advanced',
202                'focus_area': 'decision-making'
203            }
204        ]
205        return behavioral_pool[:num_questions]
206    
207    def _generate_technical_questions(self, job_description: str, num_questions: int = 5) -> List[Dict[str, str]]:
208        """Generate technical questions based on job description"""
209        jd_lower = job_description.lower()
210        
211        technical_questions = []
212        
213        # Python-related
214        if 'python' in jd_lower:
215            technical_questions.append({
216                'question': "Explain the difference between lists and tuples in Python. When would you use each?",
217                'category': 'technical',
218                'difficulty': 'intermediate',
219                'focus_area': 'Python fundamentals'
220            })
221            technical_questions.append({
222                'question': "How do you handle exceptions in Python? Give an example of try-except usage.",
223                'category': 'technical',
224                'difficulty': 'intermediate',
225                'focus_area': 'Python error handling'
226            })
227        
228        # Machine Learning
229        if 'machine learning' in jd_lower or 'ml' in jd_lower:
230            technical_questions.append({
231                'question': "What is the difference between supervised and unsupervised learning? Provide examples of each.",
232                'category': 'technical',
233                'difficulty': 'intermediate',
234                'focus_area': 'ML concepts'
235            })
236            technical_questions.append({
237                'question': "Explain overfitting in machine learning. How would you prevent it?",
238                'category': 'technical',
239                'difficulty': 'advanced',
240                'focus_area': 'ML model optimization'
241            })
242        
243        # Data Science
244        if 'data' in jd_lower:
245            technical_questions.append({
246                'question': "How would you handle missing data in a dataset?",
247                'category': 'technical',
248                'difficulty': 'intermediate',
249                'focus_area': 'data preprocessing'
250            })
251        
252        # Deep Learning
253        if 'deep learning' in jd_lower or 'neural network' in jd_lower:
254            technical_questions.append({
255                'question': "Explain how backpropagation works in neural networks.",
256                'category': 'technical',
257                'difficulty': 'advanced',
258                'focus_area': 'deep learning'
259            })
260        
261        # Web Development
262        if 'web' in jd_lower or 'api' in jd_lower:
263            technical_questions.append({
264                'question': "What is the difference between GET and POST requests in HTTP?",
265                'category': 'technical',
266                'difficulty': 'basic',
267                'focus_area': 'web development'
268            })
269        
270        # Database
271        if 'sql' in jd_lower or 'database' in jd_lower:
272            technical_questions.append({
273                'question': "Explain the difference between SQL and NoSQL databases. When would you use each?",
274                'category': 'technical',
275                'difficulty': 'intermediate',
276                'focus_area': 'database'
277            })
278        
279        # Default technical questions if no matches
280        if len(technical_questions) < num_questions:
281            default_technical = [
282                {
283                    'question': "Describe your approach to debugging a complex technical issue.",
284                    'category': 'technical',
285                    'difficulty': 'intermediate',
286                    'focus_area': 'problem-solving'
287                },
288                {
289                    'question': "How do you ensure code quality in your projects?",
290                    'category': 'technical',
291                    'difficulty': 'intermediate',
292                    'focus_area': 'best practices'
293                },
294                {
295                    'question': "Explain a technical concept you recently learned and how you applied it.",
296                    'category': 'technical',
297                    'difficulty': 'intermediate',
298                    'focus_area': 'continuous learning'
299                }
300            ]
301            technical_questions.extend(default_technical)
302        
303        return technical_questions[:num_questions]
304    
305    def generate_resume_specific_questions(self, resume: str, job_description: str, 
306                                          num_questions: int = 10) -> List[Dict[str, str]]:
307        """
308        Generate highly specific questions based on resume content
309        These are the most likely to be asked in real interviews
310        """
311        if not self.available or not self.api_key:
312            print("โš ๏ธ AI not available. Using fallback resume questions...")
313            return self._generate_fallback_resume_questions(resume, job_description, num_questions)
314        
315        try:
316            prompt = f"""You are an expert technical interviewer. Based on the candidate's resume and the job description, generate {num_questions} HIGHLY SPECIFIC interview questions that:
317
3181. Are directly related to projects, technologies, or experiences mentioned in the resume
3192. Test depth of knowledge about skills they claim to have
3203. Ask about specific accomplishments or roles mentioned
3214. Would naturally be asked by a hiring manager reviewing this resume
3225. Connect resume experience to job requirements
323
324RESUME:
325{resume[:2000]}
326
327JOB DESCRIPTION:
328{job_description[:1500]}
329
330Generate questions that dig deep into:
331- Specific projects mentioned (ask about implementation details, challenges, results)
332- Technologies and tools listed (ask about usage, experience level, best practices)
333- Roles and responsibilities (ask about specific scenarios and decisions)
334- Achievements mentioned (ask for details, metrics, process)
335
336Format as JSON array with this structure:
337[
338    {{
339        "question": "Specific question about resume content",
340        "category": "resume_based",
341        "difficulty": "intermediate/advanced",
342        "focus_area": "specific skill/project from resume"
343    }}
344]
345
346Return ONLY the JSON array of {num_questions} questions."""
347
348            if self.provider == "openai" or self.api_key.startswith("gsk_"):
349                response = self.client.chat.completions.create(
350                    model=self.model_name,  # <--- Use the dynamic variable here
351                    messages=[
352                        {"role": "system", "content": "You are an expert interviewer who asks precise, resume-specific questions that would realistically be asked in interviews."},
353                        {"role": "user", "content": prompt}
354                    ],
355                    temperature=0.8
356                )
357                questions_text = response.choices[0].message.content
358            else:  # anthropic
359                message = self.client.messages.create(
360                    model="claude-3-sonnet-20240229",
361                    max_tokens=3000,
362                    messages=[
363                        {"role": "user", "content": prompt}
364                    ]
365                )
366                questions_text = message.content[0].text
367            
368            questions = self._parse_questions(questions_text)
369            
370            # Ensure all questions are marked as resume_based
371            for q in questions:
372                q['category'] = 'resume_based'
373            
374            return questions
375            
376        except Exception as e:
377            print(f"Error generating resume-specific questions: {e}")
378            return self._generate_fallback_resume_questions(resume, job_description, num_questions)
379    
380    def _generate_fallback_resume_questions(self, resume: str, job_description: str, 
381                                           num_questions: int) -> List[Dict[str, str]]:
382        """Generate resume-specific questions when AI is unavailable"""
383        resume_lower = resume.lower()
384        questions = []
385        
386        # Extract likely technologies/skills from resume
387        technologies = {
388            'python': 'Python',
389            'java': 'Java',
390            'javascript': 'JavaScript',
391            'machine learning': 'Machine Learning',
392            'deep learning': 'Deep Learning',
393            'tensorflow': 'TensorFlow',
394            'pytorch': 'PyTorch',
395            'react': 'React',
396            'django': 'Django',
397            'flask': 'Flask',
398            'sql': 'SQL',
399            'aws': 'AWS',
400            'docker': 'Docker'
401        }
402        
403        found_tech = [tech_name for tech_key, tech_name in technologies.items() if tech_key in resume_lower]
404        
405        # Generate questions based on found technologies
406        for tech in found_tech[:5]:
407            questions.append({
408                'question': f"I see you have experience with {tech}. Can you walk me through a specific project where you used {tech} and the challenges you faced?",
409                'category': 'resume_based',
410                'difficulty': 'intermediate',
411                'focus_area': f'{tech} experience'
412            })
413        
414        # Generic resume-based questions
415        generic_resume_questions = [
416            {
417                'question': "Looking at your resume, which project are you most proud of and why?",
418                'category': 'resume_based',
419                'difficulty': 'intermediate',
420                'focus_area': 'project discussion'
421            },
422            {
423                'question': "Can you elaborate on your most recent role and what your day-to-day responsibilities were?",
424                'category': 'resume_based',
425                'difficulty': 'basic',
426                'focus_area': 'work experience'
427            },
428            {
429                'question': "You mentioned [skill] in your resume. How long have you been working with it and at what scale?",
430                'category': 'resume_based',
431                'difficulty': 'intermediate',
432                'focus_area': 'skill verification'
433            },
434            {
435                'question': "Tell me about the technical architecture of one of the projects listed on your resume.",
436                'category': 'resume_based',
437                'difficulty': 'advanced',
438                'focus_area': 'technical depth'
439            },
440            {
441                'question': "What was the biggest technical challenge you faced in your previous role and how did you solve it?",
442                'category': 'resume_based',
443                'difficulty': 'advanced',
444                'focus_area': 'problem-solving'
445            }
446        ]
447        
448        questions.extend(generic_resume_questions)
449        
450        return questions[:num_questions]
451
452    def _generate_fallback_questions(self, job_description: str, 
453                                    num_questions: int) -> List[Dict[str, str]]:
454        """Generate fallback questions when AI is not available"""
455        
456        # Analyze JD for technical terms
457        jd_lower = job_description.lower()
458        
459        fallback_questions = [
460            {
461                'question': "Tell me about yourself and why you're interested in this position.",
462                'category': 'behavioral',
463                'difficulty': 'basic',
464                'focus_area': 'introduction and motivation'
465            },
466            {
467                'question': "Describe a challenging project you've worked on and how you overcame obstacles.",
468                'category': 'behavioral',
469                'difficulty': 'intermediate',
470                'focus_area': 'problem-solving and resilience'
471            },
472            {
473                'question': "What are your strongest technical skills and how have you applied them in your previous roles?",
474                'category': 'technical',
475                'difficulty': 'intermediate',
476                'focus_area': 'technical competency'
477            },
478            {
479                'question': "How do you stay updated with the latest trends and technologies in your field?",
480                'category': 'behavioral',
481                'difficulty': 'basic',
482                'focus_area': 'continuous learning'
483            },
484            {
485                'question': "Describe a situation where you had to work with a difficult team member. How did you handle it?",
486                'category': 'behavioral',
487                'difficulty': 'intermediate',
488                'focus_area': 'teamwork and conflict resolution'
489            },
490            {
491                'question': "Walk me through your approach to debugging a complex technical issue.",
492                'category': 'technical',
493                'difficulty': 'advanced',
494                'focus_area': 'analytical thinking'
495            },
496            {
497                'question': "What do you consider your greatest professional achievement and why?",
498                'category': 'behavioral',
499                'difficulty': 'intermediate',
500                'focus_area': 'self-awareness and impact'
501            }
502        ]
503        
504        # Add role-specific questions based on keywords
505        if 'python' in jd_lower or 'machine learning' in jd_lower:
506            fallback_questions.append({
507                'question': "Explain how you would approach building a machine learning model for a classification problem.",
508                'category': 'technical',
509                'difficulty': 'advanced',
510                'focus_area': 'machine learning expertise'
511            })
512        
513        if 'leadership' in jd_lower or 'manage' in jd_lower:
514            fallback_questions.append({
515                'question': "Describe your experience leading a team and how you ensure team productivity and morale.",
516                'category': 'behavioral',
517                'difficulty': 'advanced',
518                'focus_area': 'leadership and management'
519            })
520        
521        return fallback_questions[:num_questions]
522
523
524if __name__ == "__main__":
525    # Test the question generator
526    generator = QuestionGenerator()
527    
528    sample_jd = """
529    Senior Software Engineer - AI/ML
530    
531    We are looking for an experienced software engineer with expertise in 
532    Python, machine learning, and cloud technologies. The ideal candidate 
533    will have 5+ years of experience building scalable ML systems.
534    """
535    
536    sample_resume = """
537    John Doe
538    Software Engineer with 6 years of experience in Python development,
539    machine learning, and cloud architecture. Expert in TensorFlow and AWS.
540    """
541    
542    print("Generating interview questions...")
543    questions = generator.generate_questions(sample_jd, sample_resume, num_questions=5)
544    
545    print(f"\nGenerated {len(questions)} questions:")
546    for i, q in enumerate(questions, 1):
547        print(f"\n{i}. {q['question']}")
548        print(f"   Category: {q['category']} | Difficulty: {q['difficulty']}")
549        print(f"   Focus: {q['focus_area']}")