Prathmesh0001/interview-system
0
1"""
2Enhanced Answer Analyzer Module
3Analyzes interview answers, facial expressions, and speech features
4"""
5
6import nltk
7from nltk.tokenize import word_tokenize, sent_tokenize
8from nltk.corpus import stopwords
9from textblob import TextBlob
10from vaderSentiment.vaderSentiment import SentimentIntensityAnalyzer
11from typing import Dict, List
12import re
13
14# Download required NLTK data
15try:
16 nltk.data.find('tokenizers/punkt')
17except LookupError:
18 nltk.download('punkt', quiet=True)
19
20try:
21 nltk.data.find('corpora/stopwords')
22except LookupError:
23 nltk.download('stopwords', quiet=True)
24
25
26class AnswerAnalyzer:
27 """Analyze interview answers with speech and facial features"""
28
29 def __init__(self):
30 """Initialize answer analyzer"""
31 self.sentiment_analyzer = SentimentIntensityAnalyzer()
32 self.stop_words = set(stopwords.words('english'))
33
34 # Filler words to detect
35 self.filler_words = [
36 'um', 'uh', 'like', 'you know', 'sort of', 'kind of',
37 'i mean', 'actually', 'basically', 'literally', 'seriously',
38 'honestly', 'right', 'okay', 'so', 'well', 'yeah'
39 ]
40
41 def analyze_answer(self, answer: str, question: str,
42 video_data: Dict = None,
43 audio_duration: float = 0) -> Dict:
44 """
45 Comprehensive analysis of interview answer with A/V features
46
47 Args:
48 answer: Candidate's answer text
49 question: Interview question
50 video_data: Optional facial expression data
51 audio_duration: Duration of answer in seconds
52
53 Returns:
54 Dictionary with comprehensive analysis
55 """
56 if not answer or not answer.strip():
57 return self._get_empty_analysis()
58
59 # Text analysis
60 text_analysis = {
61 'text_metrics': self._analyze_text_metrics(answer),
62 'content_quality': self._analyze_content_quality(answer, question),
63 'sentiment': self._analyze_sentiment(answer),
64 'relevance': self._analyze_relevance(answer, question),
65 'clarity': self._analyze_clarity(answer),
66 'professionalism': self._analyze_professionalism(answer)
67 }
68
69 # Speech analysis
70 speech_analysis = self._analyze_speech_features(answer, audio_duration)
71
72 # Facial expression analysis
73 facial_analysis = self._analyze_facial_expressions(video_data) if video_data else {}
74
75 # Combine all analyses
76 analysis = {
77 **text_analysis,
78 'speech_features': speech_analysis,
79 'facial_expressions': facial_analysis,
80 'overall_score': 0
81 }
82
83 # Calculate overall score
84 analysis['overall_score'] = self._calculate_overall_score(analysis)
85
86 return analysis
87
88 def _analyze_speech_features(self, text: str, duration: float) -> Dict:
89 """Analyze speech patterns and features"""
90 words = word_tokenize(text.lower())
91 text_lower = text.lower()
92
93 # Detect filler words
94 filler_count = sum(text_lower.count(filler) for filler in self.filler_words)
95
96 # Calculate speaking rate (words per minute)
97 speaking_rate = 0
98 if duration > 0:
99 speaking_rate = (len(words) / duration) * 60
100
101 # Detect repetitions
102 word_freq = {}
103 for word in words:
104 if word.isalpha() and len(word) > 3:
105 word_freq[word] = word_freq.get(word, 0) + 1
106 repetitions = sum(1 for count in word_freq.values() if count > 2)
107
108 # Detect long pauses (indicated by multiple punctuation)
109 pause_indicators = text.count('...') + text.count('..')
110
111 # Estimate stuttering (repeated characters or words)
112 stutter_pattern = r'\b(\w+)\s+\1\b'
113 stutters = len(re.findall(stutter_pattern, text_lower))
114
115 # Calculate fluency score
116 fluency_score = 100
117 fluency_score -= min(filler_count * 5, 30) # -5 per filler word, max -30
118 fluency_score -= min(stutters * 10, 20) # -10 per stutter, max -20
119 fluency_score -= min(pause_indicators * 5, 15) # -5 per pause, max -15
120
121 # Adjust for speaking rate
122 if speaking_rate > 0:
123 if speaking_rate < 100: # Too slow
124 fluency_score -= 10
125 elif speaking_rate > 200: # Too fast
126 fluency_score -= 10
127
128 fluency_score = max(0, min(100, fluency_score))
129
130 return {
131 'filler_word_count': filler_count,
132 'speaking_rate': round(speaking_rate, 1),
133 'repetitions': repetitions,
134 'stuttering_instances': stutters,
135 'pause_indicators': pause_indicators,
136 'fluency_score': fluency_score,
137 'speaking_pace': self._get_speaking_pace(speaking_rate)
138 }
139
140 def _get_speaking_pace(self, rate: float) -> str:
141 """Categorize speaking pace"""
142 if rate == 0:
143 return 'unknown'
144 elif rate < 100:
145 return 'too slow'
146 elif rate < 130:
147 return 'slow'
148 elif rate < 160:
149 return 'optimal'
150 elif rate < 190:
151 return 'fast'
152 else:
153 return 'too fast'
154
155 def _analyze_facial_expressions(self, video_data: Dict) -> Dict:
156 """Analyze facial expressions from video data"""
157 if not video_data:
158 return {
159 'confidence_level': 70,
160 'nervousness': 'moderate',
161 'eye_contact': 'good',
162 'expressions': 'neutral'
163 }
164
165 # Extract metrics from video_data
166 eye_contact_pct = video_data.get('eye_contact_percentage', 60)
167 dominant_emotion = video_data.get('dominant_emotion', 'neutral')
168 engagement = video_data.get('engagement_score', 70)
169
170 # Determine nervousness level
171 nervousness = 'low'
172 if engagement < 50:
173 nervousness = 'high'
174 elif engagement < 70:
175 nervousness = 'moderate'
176
177 # Determine confidence from engagement and emotion
178 confidence = engagement
179 if dominant_emotion in ['happy', 'focused']:
180 confidence += 10
181 elif dominant_emotion in ['sad', 'worried']:
182 confidence -= 10
183 confidence = max(0, min(100, confidence))
184
185 # Eye contact quality
186 eye_contact_quality = 'excellent' if eye_contact_pct > 70 else \
187 'good' if eye_contact_pct > 50 else \
188 'needs improvement'
189
190 return {
191 'confidence_level': round(confidence, 1),
192 'nervousness': nervousness,
193 'eye_contact': eye_contact_quality,
194 'eye_contact_percentage': round(eye_contact_pct, 1),
195 'dominant_emotion': dominant_emotion,
196 'engagement_score': round(engagement, 1)
197 }
198
199 def _analyze_text_metrics(self, text: str) -> Dict:
200 """Analyze basic text metrics"""
201 words = word_tokenize(text.lower())
202 sentences = sent_tokenize(text)
203 content_words = [w for w in words if w.isalpha() and w not in self.stop_words]
204
205 avg_word_length = sum(len(w) for w in content_words) / len(content_words) if content_words else 0
206 avg_sentence_length = len(words) / len(sentences) if sentences else 0
207
208 return {
209 'word_count': len(words),
210 'sentence_count': len(sentences),
211 'content_word_count': len(content_words),
212 'avg_word_length': round(avg_word_length, 2),
213 'avg_sentence_length': round(avg_sentence_length, 2),
214 'unique_words': len(set(content_words)),
215 'vocabulary_richness': round(len(set(content_words)) / len(content_words), 3) if content_words else 0
216 }
217
218 def _analyze_content_quality(self, text: str, question: str) -> Dict:
219 """Analyze content quality and specificity"""
220 text_lower = text.lower()
221
222 # Check for examples and specificity
223 has_numbers = bool(re.search(r'\d+', text))
224 has_examples = any(phrase in text_lower for phrase in
225 ['for example', 'for instance', 'such as', 'like when', 'specifically'])
226 has_quantification = any(word in text_lower for word in
227 ['increased', 'decreased', 'improved', 'reduced', 'achieved',
228 'percent', '%', 'times', 'doubled', 'tripled'])
229
230 # Check for STAR method indicators
231 has_situation = any(word in text_lower for word in ['situation', 'context', 'background'])
232 has_task = any(word in text_lower for word in ['task', 'goal', 'objective', 'challenge'])
233 has_action = any(word in text_lower for word in ['action', 'did', 'implemented', 'developed'])
234 has_result = any(word in text_lower for word in ['result', 'outcome', 'impact', 'achievement'])
235
236 # Calculate quality score
237 quality_score = 50 # Base score
238 if has_numbers: quality_score += 10
239 if has_examples: quality_score += 15
240 if has_quantification: quality_score += 15
241 if has_situation: quality_score += 5
242 if has_task: quality_score += 5
243 if has_action: quality_score += 5
244 if has_result: quality_score += 10
245
246 quality_score = min(100, quality_score)
247
248 return {
249 'quality_score': quality_score,
250 'has_specific_examples': has_examples,
251 'has_quantifiable_results': has_quantification,
252 'has_numbers': has_numbers,
253 'uses_star_method': has_situation and has_task and has_action and has_result
254 }
255
256 def _analyze_sentiment(self, text: str) -> Dict:
257 """Analyze sentiment and confidence"""
258 scores = self.sentiment_analyzer.polarity_scores(text)
259 blob = TextBlob(text)
260
261 # Calculate confidence level based on sentiment
262 confidence = 50 + (scores['pos'] * 50) - (scores['neg'] * 30)
263 confidence = max(0, min(100, confidence))
264
265 return {
266 'polarity': round(blob.sentiment.polarity, 3),
267 'subjectivity': round(blob.sentiment.subjectivity, 3),
268 'positive_score': round(scores['pos'], 3),
269 'negative_score': round(scores['neg'], 3),
270 'neutral_score': round(scores['neu'], 3),
271 'confidence_level': round(confidence, 1),
272 'enthusiasm_score': round(scores['pos'] * 100, 1)
273 }
274
275 def _analyze_relevance(self, answer: str, question: str) -> Dict:
276 """Analyze answer relevance to question"""
277 answer_words = set(word_tokenize(answer.lower()))
278 question_words = set(word_tokenize(question.lower()))
279
280 # Remove stop words
281 answer_words = {w for w in answer_words if w.isalpha() and w not in self.stop_words}
282 question_words = {w for w in question_words if w.isalpha() and w not in self.stop_words}
283
284 # Calculate overlap
285 overlap = answer_words.intersection(question_words)
286 relevance = (len(overlap) / len(question_words)) * 100 if question_words else 50
287 relevance = min(100, relevance * 1.5) # Boost the score slightly
288
289 return {
290 'relevance_score': round(relevance, 1),
291 'keyword_overlap': len(overlap),
292 'directly_addresses_question': len(overlap) >= len(question_words) * 0.3
293 }
294
295 def _analyze_clarity(self, text: str) -> Dict:
296 """Analyze clarity and structure"""
297 text_lower = text.lower()
298
299 # Check for transition words
300 transitions = ['first', 'second', 'then', 'next', 'finally', 'additionally',
301 'moreover', 'however', 'therefore', 'consequently']
302 transition_count = sum(1 for t in transitions if t in text_lower)
303
304 # Calculate clarity score
305 clarity_score = 60 # Base score
306 clarity_score += min(transition_count * 8, 24) # Up to +24 for transitions
307
308 # Penalize if too many short sentences or too few
309 sentences = sent_tokenize(text)
310 avg_sentence_length = len(word_tokenize(text)) / len(sentences) if sentences else 0
311 if 10 < avg_sentence_length < 25:
312 clarity_score += 16
313 elif avg_sentence_length < 5:
314 clarity_score -= 10
315
316 clarity_score = max(0, min(100, clarity_score))
317
318 return {
319 'clarity_score': clarity_score,
320 'has_structure': transition_count > 0,
321 'transition_words': transition_count
322 }
323
324 def _analyze_professionalism(self, text: str) -> Dict:
325 """Analyze professionalism"""
326 text_lower = text.lower()
327
328 # Professional phrases
329 professional_phrases = [
330 'experience', 'responsible for', 'achieved', 'developed',
331 'implemented', 'collaborated', 'managed', 'led', 'initiated'
332 ]
333 professional_count = sum(1 for phrase in professional_phrases if phrase in text_lower)
334
335 # Casual words
336 casual_words = ['gonna', 'wanna', 'kinda', 'sorta', 'yeah', 'stuff', 'things', 'like']
337 casual_count = sum(1 for word in casual_words if word in text_lower)
338
339 # Calculate score
340 prof_score = 70 # Base
341 prof_score += professional_count * 5
342 prof_score -= casual_count * 10
343 prof_score = max(0, min(100, prof_score))
344
345 return {
346 'professionalism_score': prof_score,
347 'professional_language': professional_count,
348 'casual_language': casual_count
349 }
350
351 def _calculate_overall_score(self, analysis: Dict) -> float:
352 """Calculate weighted overall score"""
353 weights = {
354 'content_quality': 0.25,
355 'relevance': 0.20,
356 'clarity': 0.15,
357 'sentiment': 0.10,
358 'professionalism': 0.10,
359 'speech': 0.15,
360 'facial': 0.05
361 }
362
363 scores = {
364 'content_quality': analysis['content_quality']['quality_score'],
365 'relevance': analysis['relevance']['relevance_score'],
366 'clarity': analysis['clarity']['clarity_score'],
367 'sentiment': analysis['sentiment']['confidence_level'],
368 'professionalism': analysis['professionalism']['professionalism_score'],
369 'speech': analysis.get('speech_features', {}).get('fluency_score', 70),
370 'facial': analysis.get('facial_expressions', {}).get('confidence_level', 70)
371 }
372
373 overall = sum(scores[key] * weights[key] for key in weights.keys())
374 return round(overall, 1)
375
376 def get_feedback(self, analysis: Dict) -> List[str]:
377 """Generate specific, actionable feedback"""
378 feedback = []
379
380 # Content feedback
381 content = analysis['content_quality']
382 if not content.get('has_specific_examples'):
383 feedback.append("Add specific examples from your experience to illustrate your points")
384 if not content.get('has_quantifiable_results'):
385 feedback.append("Include measurable results (percentages, numbers, metrics) to demonstrate impact")
386 if content['quality_score'] < 60:
387 feedback.append("Provide more detailed responses with concrete details and outcomes")
388
389 # Speech feedback
390 if 'speech_features' in analysis:
391 speech = analysis['speech_features']
392 if speech['filler_word_count'] > 3:
393 feedback.append(f"Reduce filler words (found {speech['filler_word_count']} instances of 'um', 'uh', 'like')")
394 if speech['stuttering_instances'] > 0:
395 feedback.append("Practice your answers to reduce stuttering and improve fluency")
396 if speech['speaking_pace'] in ['too slow', 'too fast']:
397 feedback.append(f"Adjust your speaking pace (currently {speech['speaking_pace']}) to 130-160 words/minute")
398
399 # Facial expression feedback
400 if 'facial_expressions' in analysis:
401 facial = analysis['facial_expressions']
402 if facial.get('eye_contact_percentage', 100) < 50:
403 feedback.append("Maintain better eye contact with the camera (look directly at it)")
404 if facial.get('nervousness') == 'high':
405 feedback.append("Try to relax and appear more confident - take deep breaths before answering")
406 if facial.get('engagement_score', 100) < 60:
407 feedback.append("Show more enthusiasm and engagement through your facial expressions")
408
409 # Relevance feedback
410 if analysis['relevance']['relevance_score'] < 60:
411 feedback.append("Ensure your answer directly addresses what was asked in the question")
412
413 # Clarity feedback
414 if analysis['clarity']['clarity_score'] < 60:
415 feedback.append("Structure your answer better using transitions (first, then, finally)")
416
417 # Professionalism feedback
418 if analysis['professionalism']['professionalism_score'] < 60:
419 feedback.append("Use more professional language and avoid casual expressions")
420
421 # Length feedback
422 word_count = analysis['text_metrics']['word_count']
423 if word_count < 30:
424 feedback.append("Provide longer, more detailed answers (aim for 50-150 words)")
425 elif word_count > 200:
426 feedback.append("Keep answers more concise and focused on key points")
427
428 # If no issues, give encouragement
429 if not feedback:
430 feedback.append("Excellent answer! Keep up the great work!")
431
432 return feedback[:5] # Return top 5 feedback points
433
434 def _get_empty_analysis(self) -> Dict:
435 """Return empty analysis"""
436 return {
437 'text_metrics': {'word_count': 0},
438 'content_quality': {'quality_score': 0},
439 'sentiment': {'confidence_level': 0},
440 'relevance': {'relevance_score': 0},
441 'clarity': {'clarity_score': 0},
442 'professionalism': {'professionalism_score': 0},
443 'speech_features': {'fluency_score': 0},
444 'facial_expressions': {'confidence_level': 0},
445 'overall_score': 0
446 }