Team Ai
Apppublic

Prathmesh0001/interview-system

sourceHugging Faceupdated 8mo agoView on Hugging Face
0likes
answer_analyzer.py446 linesDownload Raw Back to root
1"""
2Enhanced Answer Analyzer Module
3Analyzes interview answers, facial expressions, and speech features
4"""
5
6import nltk
7from nltk.tokenize import word_tokenize, sent_tokenize
8from nltk.corpus import stopwords
9from textblob import TextBlob
10from vaderSentiment.vaderSentiment import SentimentIntensityAnalyzer
11from typing import Dict, List
12import re
13
14# Download required NLTK data
15try:
16    nltk.data.find('tokenizers/punkt')
17except LookupError:
18    nltk.download('punkt', quiet=True)
19
20try:
21    nltk.data.find('corpora/stopwords')
22except LookupError:
23    nltk.download('stopwords', quiet=True)
24
25
26class AnswerAnalyzer:
27    """Analyze interview answers with speech and facial features"""
28    
29    def __init__(self):
30        """Initialize answer analyzer"""
31        self.sentiment_analyzer = SentimentIntensityAnalyzer()
32        self.stop_words = set(stopwords.words('english'))
33        
34        # Filler words to detect
35        self.filler_words = [
36            'um', 'uh', 'like', 'you know', 'sort of', 'kind of',
37            'i mean', 'actually', 'basically', 'literally', 'seriously',
38            'honestly', 'right', 'okay', 'so', 'well', 'yeah'
39        ]
40    
41    def analyze_answer(self, answer: str, question: str, 
42                       video_data: Dict = None,
43                       audio_duration: float = 0) -> Dict:
44        """
45        Comprehensive analysis of interview answer with A/V features
46        
47        Args:
48            answer: Candidate's answer text
49            question: Interview question
50            video_data: Optional facial expression data
51            audio_duration: Duration of answer in seconds
52            
53        Returns:
54            Dictionary with comprehensive analysis
55        """
56        if not answer or not answer.strip():
57            return self._get_empty_analysis()
58        
59        # Text analysis
60        text_analysis = {
61            'text_metrics': self._analyze_text_metrics(answer),
62            'content_quality': self._analyze_content_quality(answer, question),
63            'sentiment': self._analyze_sentiment(answer),
64            'relevance': self._analyze_relevance(answer, question),
65            'clarity': self._analyze_clarity(answer),
66            'professionalism': self._analyze_professionalism(answer)
67        }
68        
69        # Speech analysis
70        speech_analysis = self._analyze_speech_features(answer, audio_duration)
71        
72        # Facial expression analysis
73        facial_analysis = self._analyze_facial_expressions(video_data) if video_data else {}
74        
75        # Combine all analyses
76        analysis = {
77            **text_analysis,
78            'speech_features': speech_analysis,
79            'facial_expressions': facial_analysis,
80            'overall_score': 0
81        }
82        
83        # Calculate overall score
84        analysis['overall_score'] = self._calculate_overall_score(analysis)
85        
86        return analysis
87    
88    def _analyze_speech_features(self, text: str, duration: float) -> Dict:
89        """Analyze speech patterns and features"""
90        words = word_tokenize(text.lower())
91        text_lower = text.lower()
92        
93        # Detect filler words
94        filler_count = sum(text_lower.count(filler) for filler in self.filler_words)
95        
96        # Calculate speaking rate (words per minute)
97        speaking_rate = 0
98        if duration > 0:
99            speaking_rate = (len(words) / duration) * 60
100        
101        # Detect repetitions
102        word_freq = {}
103        for word in words:
104            if word.isalpha() and len(word) > 3:
105                word_freq[word] = word_freq.get(word, 0) + 1
106        repetitions = sum(1 for count in word_freq.values() if count > 2)
107        
108        # Detect long pauses (indicated by multiple punctuation)
109        pause_indicators = text.count('...') + text.count('..') 
110        
111        # Estimate stuttering (repeated characters or words)
112        stutter_pattern = r'\b(\w+)\s+\1\b'
113        stutters = len(re.findall(stutter_pattern, text_lower))
114        
115        # Calculate fluency score
116        fluency_score = 100
117        fluency_score -= min(filler_count * 5, 30)  # -5 per filler word, max -30
118        fluency_score -= min(stutters * 10, 20)     # -10 per stutter, max -20
119        fluency_score -= min(pause_indicators * 5, 15)  # -5 per pause, max -15
120        
121        # Adjust for speaking rate
122        if speaking_rate > 0:
123            if speaking_rate < 100:  # Too slow
124                fluency_score -= 10
125            elif speaking_rate > 200:  # Too fast
126                fluency_score -= 10
127        
128        fluency_score = max(0, min(100, fluency_score))
129        
130        return {
131            'filler_word_count': filler_count,
132            'speaking_rate': round(speaking_rate, 1),
133            'repetitions': repetitions,
134            'stuttering_instances': stutters,
135            'pause_indicators': pause_indicators,
136            'fluency_score': fluency_score,
137            'speaking_pace': self._get_speaking_pace(speaking_rate)
138        }
139    
140    def _get_speaking_pace(self, rate: float) -> str:
141        """Categorize speaking pace"""
142        if rate == 0:
143            return 'unknown'
144        elif rate < 100:
145            return 'too slow'
146        elif rate < 130:
147            return 'slow'
148        elif rate < 160:
149            return 'optimal'
150        elif rate < 190:
151            return 'fast'
152        else:
153            return 'too fast'
154    
155    def _analyze_facial_expressions(self, video_data: Dict) -> Dict:
156        """Analyze facial expressions from video data"""
157        if not video_data:
158            return {
159                'confidence_level': 70,
160                'nervousness': 'moderate',
161                'eye_contact': 'good',
162                'expressions': 'neutral'
163            }
164        
165        # Extract metrics from video_data
166        eye_contact_pct = video_data.get('eye_contact_percentage', 60)
167        dominant_emotion = video_data.get('dominant_emotion', 'neutral')
168        engagement = video_data.get('engagement_score', 70)
169        
170        # Determine nervousness level
171        nervousness = 'low'
172        if engagement < 50:
173            nervousness = 'high'
174        elif engagement < 70:
175            nervousness = 'moderate'
176        
177        # Determine confidence from engagement and emotion
178        confidence = engagement
179        if dominant_emotion in ['happy', 'focused']:
180            confidence += 10
181        elif dominant_emotion in ['sad', 'worried']:
182            confidence -= 10
183        confidence = max(0, min(100, confidence))
184        
185        # Eye contact quality
186        eye_contact_quality = 'excellent' if eye_contact_pct > 70 else \
187                             'good' if eye_contact_pct > 50 else \
188                             'needs improvement'
189        
190        return {
191            'confidence_level': round(confidence, 1),
192            'nervousness': nervousness,
193            'eye_contact': eye_contact_quality,
194            'eye_contact_percentage': round(eye_contact_pct, 1),
195            'dominant_emotion': dominant_emotion,
196            'engagement_score': round(engagement, 1)
197        }
198    
199    def _analyze_text_metrics(self, text: str) -> Dict:
200        """Analyze basic text metrics"""
201        words = word_tokenize(text.lower())
202        sentences = sent_tokenize(text)
203        content_words = [w for w in words if w.isalpha() and w not in self.stop_words]
204        
205        avg_word_length = sum(len(w) for w in content_words) / len(content_words) if content_words else 0
206        avg_sentence_length = len(words) / len(sentences) if sentences else 0
207        
208        return {
209            'word_count': len(words),
210            'sentence_count': len(sentences),
211            'content_word_count': len(content_words),
212            'avg_word_length': round(avg_word_length, 2),
213            'avg_sentence_length': round(avg_sentence_length, 2),
214            'unique_words': len(set(content_words)),
215            'vocabulary_richness': round(len(set(content_words)) / len(content_words), 3) if content_words else 0
216        }
217    
218    def _analyze_content_quality(self, text: str, question: str) -> Dict:
219        """Analyze content quality and specificity"""
220        text_lower = text.lower()
221        
222        # Check for examples and specificity
223        has_numbers = bool(re.search(r'\d+', text))
224        has_examples = any(phrase in text_lower for phrase in 
225                          ['for example', 'for instance', 'such as', 'like when', 'specifically'])
226        has_quantification = any(word in text_lower for word in 
227                                ['increased', 'decreased', 'improved', 'reduced', 'achieved', 
228                                 'percent', '%', 'times', 'doubled', 'tripled'])
229        
230        # Check for STAR method indicators
231        has_situation = any(word in text_lower for word in ['situation', 'context', 'background'])
232        has_task = any(word in text_lower for word in ['task', 'goal', 'objective', 'challenge'])
233        has_action = any(word in text_lower for word in ['action', 'did', 'implemented', 'developed'])
234        has_result = any(word in text_lower for word in ['result', 'outcome', 'impact', 'achievement'])
235        
236        # Calculate quality score
237        quality_score = 50  # Base score
238        if has_numbers: quality_score += 10
239        if has_examples: quality_score += 15
240        if has_quantification: quality_score += 15
241        if has_situation: quality_score += 5
242        if has_task: quality_score += 5
243        if has_action: quality_score += 5
244        if has_result: quality_score += 10
245        
246        quality_score = min(100, quality_score)
247        
248        return {
249            'quality_score': quality_score,
250            'has_specific_examples': has_examples,
251            'has_quantifiable_results': has_quantification,
252            'has_numbers': has_numbers,
253            'uses_star_method': has_situation and has_task and has_action and has_result
254        }
255    
256    def _analyze_sentiment(self, text: str) -> Dict:
257        """Analyze sentiment and confidence"""
258        scores = self.sentiment_analyzer.polarity_scores(text)
259        blob = TextBlob(text)
260        
261        # Calculate confidence level based on sentiment
262        confidence = 50 + (scores['pos'] * 50) - (scores['neg'] * 30)
263        confidence = max(0, min(100, confidence))
264        
265        return {
266            'polarity': round(blob.sentiment.polarity, 3),
267            'subjectivity': round(blob.sentiment.subjectivity, 3),
268            'positive_score': round(scores['pos'], 3),
269            'negative_score': round(scores['neg'], 3),
270            'neutral_score': round(scores['neu'], 3),
271            'confidence_level': round(confidence, 1),
272            'enthusiasm_score': round(scores['pos'] * 100, 1)
273        }
274    
275    def _analyze_relevance(self, answer: str, question: str) -> Dict:
276        """Analyze answer relevance to question"""
277        answer_words = set(word_tokenize(answer.lower()))
278        question_words = set(word_tokenize(question.lower()))
279        
280        # Remove stop words
281        answer_words = {w for w in answer_words if w.isalpha() and w not in self.stop_words}
282        question_words = {w for w in question_words if w.isalpha() and w not in self.stop_words}
283        
284        # Calculate overlap
285        overlap = answer_words.intersection(question_words)
286        relevance = (len(overlap) / len(question_words)) * 100 if question_words else 50
287        relevance = min(100, relevance * 1.5)  # Boost the score slightly
288        
289        return {
290            'relevance_score': round(relevance, 1),
291            'keyword_overlap': len(overlap),
292            'directly_addresses_question': len(overlap) >= len(question_words) * 0.3
293        }
294    
295    def _analyze_clarity(self, text: str) -> Dict:
296        """Analyze clarity and structure"""
297        text_lower = text.lower()
298        
299        # Check for transition words
300        transitions = ['first', 'second', 'then', 'next', 'finally', 'additionally', 
301                      'moreover', 'however', 'therefore', 'consequently']
302        transition_count = sum(1 for t in transitions if t in text_lower)
303        
304        # Calculate clarity score
305        clarity_score = 60  # Base score
306        clarity_score += min(transition_count * 8, 24)  # Up to +24 for transitions
307        
308        # Penalize if too many short sentences or too few
309        sentences = sent_tokenize(text)
310        avg_sentence_length = len(word_tokenize(text)) / len(sentences) if sentences else 0
311        if 10 < avg_sentence_length < 25:
312            clarity_score += 16
313        elif avg_sentence_length < 5:
314            clarity_score -= 10
315        
316        clarity_score = max(0, min(100, clarity_score))
317        
318        return {
319            'clarity_score': clarity_score,
320            'has_structure': transition_count > 0,
321            'transition_words': transition_count
322        }
323    
324    def _analyze_professionalism(self, text: str) -> Dict:
325        """Analyze professionalism"""
326        text_lower = text.lower()
327        
328        # Professional phrases
329        professional_phrases = [
330            'experience', 'responsible for', 'achieved', 'developed', 
331            'implemented', 'collaborated', 'managed', 'led', 'initiated'
332        ]
333        professional_count = sum(1 for phrase in professional_phrases if phrase in text_lower)
334        
335        # Casual words
336        casual_words = ['gonna', 'wanna', 'kinda', 'sorta', 'yeah', 'stuff', 'things', 'like']
337        casual_count = sum(1 for word in casual_words if word in text_lower)
338        
339        # Calculate score
340        prof_score = 70  # Base
341        prof_score += professional_count * 5
342        prof_score -= casual_count * 10
343        prof_score = max(0, min(100, prof_score))
344        
345        return {
346            'professionalism_score': prof_score,
347            'professional_language': professional_count,
348            'casual_language': casual_count
349        }
350    
351    def _calculate_overall_score(self, analysis: Dict) -> float:
352        """Calculate weighted overall score"""
353        weights = {
354            'content_quality': 0.25,
355            'relevance': 0.20,
356            'clarity': 0.15,
357            'sentiment': 0.10,
358            'professionalism': 0.10,
359            'speech': 0.15,
360            'facial': 0.05
361        }
362        
363        scores = {
364            'content_quality': analysis['content_quality']['quality_score'],
365            'relevance': analysis['relevance']['relevance_score'],
366            'clarity': analysis['clarity']['clarity_score'],
367            'sentiment': analysis['sentiment']['confidence_level'],
368            'professionalism': analysis['professionalism']['professionalism_score'],
369            'speech': analysis.get('speech_features', {}).get('fluency_score', 70),
370            'facial': analysis.get('facial_expressions', {}).get('confidence_level', 70)
371        }
372        
373        overall = sum(scores[key] * weights[key] for key in weights.keys())
374        return round(overall, 1)
375    
376    def get_feedback(self, analysis: Dict) -> List[str]:
377        """Generate specific, actionable feedback"""
378        feedback = []
379        
380        # Content feedback
381        content = analysis['content_quality']
382        if not content.get('has_specific_examples'):
383            feedback.append("Add specific examples from your experience to illustrate your points")
384        if not content.get('has_quantifiable_results'):
385            feedback.append("Include measurable results (percentages, numbers, metrics) to demonstrate impact")
386        if content['quality_score'] < 60:
387            feedback.append("Provide more detailed responses with concrete details and outcomes")
388        
389        # Speech feedback
390        if 'speech_features' in analysis:
391            speech = analysis['speech_features']
392            if speech['filler_word_count'] > 3:
393                feedback.append(f"Reduce filler words (found {speech['filler_word_count']} instances of 'um', 'uh', 'like')")
394            if speech['stuttering_instances'] > 0:
395                feedback.append("Practice your answers to reduce stuttering and improve fluency")
396            if speech['speaking_pace'] in ['too slow', 'too fast']:
397                feedback.append(f"Adjust your speaking pace (currently {speech['speaking_pace']}) to 130-160 words/minute")
398        
399        # Facial expression feedback
400        if 'facial_expressions' in analysis:
401            facial = analysis['facial_expressions']
402            if facial.get('eye_contact_percentage', 100) < 50:
403                feedback.append("Maintain better eye contact with the camera (look directly at it)")
404            if facial.get('nervousness') == 'high':
405                feedback.append("Try to relax and appear more confident - take deep breaths before answering")
406            if facial.get('engagement_score', 100) < 60:
407                feedback.append("Show more enthusiasm and engagement through your facial expressions")
408        
409        # Relevance feedback
410        if analysis['relevance']['relevance_score'] < 60:
411            feedback.append("Ensure your answer directly addresses what was asked in the question")
412        
413        # Clarity feedback
414        if analysis['clarity']['clarity_score'] < 60:
415            feedback.append("Structure your answer better using transitions (first, then, finally)")
416        
417        # Professionalism feedback
418        if analysis['professionalism']['professionalism_score'] < 60:
419            feedback.append("Use more professional language and avoid casual expressions")
420        
421        # Length feedback
422        word_count = analysis['text_metrics']['word_count']
423        if word_count < 30:
424            feedback.append("Provide longer, more detailed answers (aim for 50-150 words)")
425        elif word_count > 200:
426            feedback.append("Keep answers more concise and focused on key points")
427        
428        # If no issues, give encouragement
429        if not feedback:
430            feedback.append("Excellent answer! Keep up the great work!")
431        
432        return feedback[:5]  # Return top 5 feedback points
433    
434    def _get_empty_analysis(self) -> Dict:
435        """Return empty analysis"""
436        return {
437            'text_metrics': {'word_count': 0},
438            'content_quality': {'quality_score': 0},
439            'sentiment': {'confidence_level': 0},
440            'relevance': {'relevance_score': 0},
441            'clarity': {'clarity_score': 0},
442            'professionalism': {'professionalism_score': 0},
443            'speech_features': {'fluency_score': 0},
444            'facial_expressions': {'confidence_level': 0},
445            'overall_score': 0
446        }