Team Ai
Apppublic

prav2020/Computer_vision

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes
1import cv2
2import mediapipe as mp
3import time
4import numpy as np
5import subprocess
6import threading
7
8# Initialize MediaPipe
9mp_hands = mp.solutions.hands
10mp_drawing = mp.solutions.drawing_utils
11
12hands = mp_hands.Hands(max_num_hands=1, min_detection_confidence=0.7)
13
14# Voice control variables
15prev_count = -1
16last_spoken_time = 0
17cooldown_period = 2.0  # 2 seconds cooldown
18is_speaking = False
19
20def speak_with_subprocess(count):
21    """Use subprocess to call Windows speech - MOST RELIABLE METHOD"""
22    global is_speaking
23    
24    try:
25        if count == 0:
26            text = "Zero fingers"
27        elif count == 1:
28            text = "One finger"
29        else:
30            text = f"{count} fingers"
31        
32        print(f"๐Ÿ”Š Speaking: {text}")
33        
34        # Use Windows PowerShell to speak (most reliable)
35        subprocess.run([
36            'powershell', 
37            '-Command', 
38            f'Add-Type -AssemblyName System.Speech; $speak = New-Object System.Speech.Synthesis.SpeechSynthesizer; $speak.Speak("{text}")'
39        ], check=True, timeout=10)
40        
41    except subprocess.TimeoutExpired:
42        print("Speech timeout")
43    except Exception as e:
44        print(f"Speech error: {e}")
45    finally:
46        is_speaking = False
47
48def count_fingers(hand_landmarks):
49    """
50    Count open fingers based on landmark positions
51    Thumb handled differently from other fingers
52    """
53    landmarks = hand_landmarks.landmark
54    fingers = []
55
56    # Thumb: compare tip and MCP (landmark 4 vs 2)
57    if landmarks[4].x < landmarks[3].x:  # For right hand
58        fingers.append(1)
59    else:
60        fingers.append(0)
61
62    # Other 4 fingers: tip y < pip y โ†’ finger open
63    tips = [8, 12, 16, 20]
64    pips = [6, 10, 14, 18]
65
66    for tip, pip in zip(tips, pips):
67        if landmarks[tip].y < landmarks[pip].y:
68            fingers.append(1)
69        else:
70            fingers.append(0)
71
72    return sum(fingers)
73
74# Start video capture
75cap = cv2.VideoCapture(0)
76
77print("๐Ÿš€ Finger Counter with Windows Speech")
78print("๐ŸŽฏ Show your hand to the camera")
79print("๐Ÿ”Š Using Windows built-in speech (most reliable)")
80print("โน๏ธ Press 'q' to quit")
81
82try:
83    while cap.isOpened():
84        ret, frame = cap.read()
85        if not ret:
86            break
87
88        # Flip for natural interaction
89        frame = cv2.flip(frame, 1)
90        rgb_frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
91
92        # Process hand landmarks
93        results = hands.process(rgb_frame)
94
95        # Create black mask
96        mask = np.zeros_like(frame)
97        finger_count = 0
98        hand_detected = False
99
100        if results.multi_hand_landmarks:
101            hand_detected = True
102            for hand_landmarks in results.multi_hand_landmarks:
103                # Draw landmarks only on mask
104                mp_drawing.draw_landmarks(
105                    mask,
106                    hand_landmarks,
107                    mp_hands.HAND_CONNECTIONS,
108                    mp_drawing.DrawingSpec(color=(0,255,0), thickness=2, circle_radius=3),
109                    mp_drawing.DrawingSpec(color=(255,0,0), thickness=2)
110                )
111
112                # Count fingers
113                finger_count = count_fingers(hand_landmarks)
114        
115        # Display finger count on mask
116        cv2.putText(mask, f'Fingers: {finger_count}', (50, 50),
117                    cv2.FONT_HERSHEY_SIMPLEX, 1, (0, 255, 255), 2)
118        
119        # Voice output logic
120        current_time = time.time()
121        time_since_last_speech = current_time - last_spoken_time
122        
123        if (hand_detected and 
124            finger_count != prev_count and 
125            time_since_last_speech >= cooldown_period and
126            not is_speaking):
127            
128            # Start speech in a separate thread
129            is_speaking = True
130            speech_thread = threading.Thread(target=speak_with_subprocess, args=(finger_count,))
131            speech_thread.daemon = True
132            speech_thread.start()
133            
134            # Update tracking variables
135            prev_count = finger_count
136            last_spoken_time = current_time
137        
138        # Display status information
139        if hand_detected:
140            if is_speaking:
141                status_text = 'Speaking...'
142                status_color = (0, 255, 0)  # Green
143            elif time_since_last_speech < cooldown_period:
144                cooldown_remaining = cooldown_period - time_since_last_speech
145                status_text = f'Cooldown: {cooldown_remaining:.1f}s'
146                status_color = (255, 255, 255)  # White
147            else:
148                status_text = 'Voice ready'
149                status_color = (0, 255, 0)  # Green
150        else:
151            status_text = 'No hand detected'
152            status_color = (0, 0, 255)  # Red
153            prev_count = -1  # Reset when no hand
154        
155        cv2.putText(mask, status_text, (50, 90),
156                    cv2.FONT_HERSHEY_SIMPLEX, 0.6, status_color, 2)
157        
158        cv2.putText(mask, "Press 'Q' to quit", (50, 130),
159                    cv2.FONT_HERSHEY_SIMPLEX, 0.5, (200, 200, 200), 1)
160
161        cv2.imshow("Hand Mask + Finger Count", mask)
162
163        if cv2.waitKey(1) & 0xFF == ord('q'):
164            break
165
166except Exception as e:
167    print(f"Error: {e}")
168
169finally:
170    # Cleanup
171    cap.release()
172    cv2.destroyAllWindows()
173    print("โœ… Application closed successfully")