Repository navigation
Expand file tree
/
Copy pathapp.py
More file actions
194 lines (166 loc) · 7.14 KB
/
Copy pathapp.py
File metadata and controls
194 lines (166 loc) · 7.14 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
# app.py
from flask import Flask, render_template, Response, jsonify, send_from_directory
import cv2
import numpy as np
import mediapipe as mp
from flask_cors import CORS
import tensorflow as tf
import os
app = Flask(__name__)
# Enable CORS with more specific settings
CORS(app, resources={r"/*": {"origins": "*"}})
# Initialize mediapipe
mp_holistic = mp.solutions.holistic
mp_drawing = mp.solutions.drawing_utils
mp_face_mesh = mp.solutions.face_mesh
# Load the trained model
model = tf.keras.models.load_model('Models/GRU_L2Reg_final.keras')
# Actions array - update to match the model's training data
actions = np.array(['Hello', 'Hill', 'Deaf', 'Thanks','You'])
# Detection variables
sequence = []
sentence = []
predictions = []
threshold = 0.5
def mediapipe_detection(image, model):
image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)
image.flags.writeable = False
results = model.process(image)
image.flags.writeable = True
image = cv2.cvtColor(image, cv2.COLOR_RGB2BGR)
return image, results
def draw_styled_landmarks(image, results):
# Draw face connections
if results.face_landmarks:
mp_drawing.draw_landmarks(
image,
results.face_landmarks,
mp_face_mesh.FACEMESH_CONTOURS,
mp_drawing.DrawingSpec(color=(80,110,10), thickness=1, circle_radius=1),
mp_drawing.DrawingSpec(color=(80,256,121), thickness=1, circle_radius=1)
)
# Draw pose connections
if results.pose_landmarks:
mp_drawing.draw_landmarks(
image,
results.pose_landmarks,
mp_holistic.POSE_CONNECTIONS,
mp_drawing.DrawingSpec(color=(80,22,10), thickness=2, circle_radius=4),
mp_drawing.DrawingSpec(color=(80,44,121), thickness=2, circle_radius=2)
)
# Draw hand connections
if results.left_hand_landmarks:
mp_drawing.draw_landmarks(
image,
results.left_hand_landmarks,
mp_holistic.HAND_CONNECTIONS,
mp_drawing.DrawingSpec(color=(121,22,76), thickness=2, circle_radius=4),
mp_drawing.DrawingSpec(color=(121,44,250), thickness=2, circle_radius=2)
)
if results.right_hand_landmarks:
mp_drawing.draw_landmarks(
image,
results.right_hand_landmarks,
mp_holistic.HAND_CONNECTIONS,
mp_drawing.DrawingSpec(color=(245,117,66), thickness=2, circle_radius=4),
mp_drawing.DrawingSpec(color=(245,66,230), thickness=2, circle_radius=2)
)
def extract_keypoints(results):
pose = np.array([[res.x, res.y, res.z, res.visibility] for res in results.pose_landmarks.landmark]).flatten() if results.pose_landmarks else np.zeros(33*4)
face = np.array([[res.x, res.y, res.z] for res in results.face_landmarks.landmark]).flatten() if results.face_landmarks else np.zeros(468*3)
lh = np.array([[res.x, res.y, res.z] for res in results.left_hand_landmarks.landmark]).flatten() if results.left_hand_landmarks else np.zeros(21*3)
rh = np.array([[res.x, res.y, res.z] for res in results.right_hand_landmarks.landmark]).flatten() if results.right_hand_landmarks else np.zeros(21*3)
return np.concatenate([pose, face, lh, rh])
def prob_viz(res, actions, input_frame, colors):
# Return the input frame unmodified to remove probability visualization
return input_frame
def generate_frames():
# Make colors match the number of actions
colors = [(245, 117, 16), (117, 245, 16), (16, 117, 245), (245, 16, 117),(117, 16, 245) , (205,92,92), (255, 0, 0), (0, 255, 0), (0, 0, 255), (255, 255, 0)]
sequence = []
global sentence # Make sentence global so we can access it from other routes
sentence = []
predictions = []
last_prediction = None # Keep track of the last prediction
cap = cv2.VideoCapture(0)
# Set mediapipe model
with mp_holistic.Holistic(min_detection_confidence=0.5, min_tracking_confidence=0.5) as holistic:
while True:
# Read feed
ret, frame = cap.read()
if not ret:
break
# Make detections
image, results = mediapipe_detection(frame, holistic)
# Draw landmarks - keep this to show face points
draw_styled_landmarks(image, results)
# Prediction logic
keypoints = extract_keypoints(results)
sequence.append(keypoints)
sequence = sequence[-30:]
if len(sequence) == 30:
res = model.predict(np.expand_dims(sequence, axis=0))[0]
predicted_action_index = np.argmax(res)
predictions.append(predicted_action_index)
predicted_action = actions[predicted_action_index]
# Update sentence in backend without displaying it on video
if np.unique(predictions[-10:])[0] == predicted_action_index:
if res[predicted_action_index] > threshold:
if len(sentence) > 0:
if predicted_action != sentence[-1]:
sentence.append(predicted_action)
else:
sentence.append(predicted_action)
if len(sentence) > 5:
sentence = sentence[-5:]
# Log the prediction but don't display it on the video
if predicted_action != last_prediction:
print(f"Predicted Action: {predicted_action}")
last_prediction = predicted_action
# Convert to jpg format
ret, buffer = cv2.imencode('.jpg', image)
frame = buffer.tobytes()
yield (b'--frame\r\n'
b'Content-Type: image/jpeg\r\n\r\n' + frame + b'\r\n')
# Root path to serve a simple HTML page
@app.route('/')
def index():
return """
<html>
<head>
<title>Indian Sign Language Detection</title>
</head>
<body>
<h1>Indian Sign Language Detection</h1>
<img src="/video_feed" width="640" height="480" />
<div id="sentence"></div>
</body>
</html>
"""
# Create a route to access detected sentences
@app.route('/get_sentence')
def get_sentence():
global sentence
return jsonify({
'sentence': ' '.join(sentence),
'words': sentence
})
@app.route('/video_feed')
def video_feed():
# Set proper headers for streaming content
return Response(generate_frames(),
mimetype='multipart/x-mixed-replace; boundary=frame',
headers={
'Cache-Control': 'no-cache, no-store, must-revalidate',
'Pragma': 'no-cache',
'Expires': '0'
})
# Serve static files from a directory
@app.route('/static/<path:path>')
def serve_static(path):
return send_from_directory('static', path)
if __name__ == "__main__":
# Ensure the static directory exists
os.makedirs('static', exist_ok=True)
# Run the app making it accessible from any device on the network
app.run(debug=True, host='0.0.0.0', port=5000)