forked from Interalab/Robotics
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathVoiceControl.py
More file actions
130 lines (111 loc) · 4.08 KB
/
Copy pathVoiceControl.py
File metadata and controls
130 lines (111 loc) · 4.08 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
import speech_recognition as sr
from google import genai
from elevenlabs.client import ElevenLabs
import serial
import time
import subprocess
import os
# ================= 配置区 =================
SERIAL_PORT = "/dev/cu.usbmodem1101"
BAUD_RATE = 9600
# --- CHANGE 1: Track last_command like your gesture code ---
last_command = None
try:
ser = serial.Serial(SERIAL_PORT, BAUD_RATE, timeout=1)
time.sleep(2)
if ser.is_open:
ser.write(b'0') # Ensure Manual Mode on start
print(f"[Status] Connected to Robot on {SERIAL_PORT}")
except Exception as e:
print(f"[Serial Error] Could not open serial port: {e}")
ser = None
gemini_client = genai.Client(api_key="")
MODEL_ID = "gemini-3-flash-preview"
ELEVENLABS_KEY = ""
el_client = ElevenLabs(api_key=ELEVENLABS_KEY)
COMMAND_MAP = {
"forward": "w",
"backward": "s",
"left": "a",
"right": "d",
"stop": "x",
"auto": "1",
"manual": "0"
}
def send_robot_command(text):
"""Refined to match the stability of your gesture code"""
global last_command
if ser is None or not ser.is_open:
return
text = text.lower()
cmd_to_send = None
# Find the matching command
for word, cmd_char in COMMAND_MAP.items():
if word in text:
cmd_to_send = cmd_char
break
# --- CHANGE 2: Deduplication Logic ---
if cmd_to_send and cmd_to_send != last_command:
# If it's a direction, force manual mode '0' first to override 'competitionRun'
if cmd_to_send in ['w', 's', 'a', 'd']:
ser.write(b'0')
time.sleep(0.01)
ser.write(cmd_to_send.encode())
ser.flush() # Ensure it leaves the Python buffer immediately
print(f"[Robot] >>> SENT: {cmd_to_send.upper()}")
last_command = cmd_to_send
elif not cmd_to_send:
print(f"[Robot] No command in: '{text}'")
def text_to_speech(text):
# (Keep your original TTS function exactly as it was)
print("[Status] Converting text to speech...")
try:
response = el_client.text_to_speech.convert(
voice_id="cgSgspJ2msm6clMCkdW9",
text=text,
model_id="eleven_multilingual_v2",
output_format="mp3_44100_128"
)
audio_bytes = b"".join(response)
with open("reply.mp3", "wb") as f:
f.write(audio_bytes)
print("[Status] Playing audio...")
subprocess.run(["afplay", "reply.mp3"])
except Exception as e:
print(f"[TTS Error] {e}")
def listen_and_talk():
recognizer = sr.Recognizer()
with sr.Microphone() as source:
print("\n[Status] Listening...")
recognizer.adjust_for_ambient_noise(source, duration=0.5)
try:
audio = recognizer.listen(source, timeout=5, phrase_time_limit=5)
user_text = recognizer.recognize_google(audio, language='en-US')
print(f"-> You said: {user_text}")
# --- CHANGE 3: Action occurs BEFORE the Gemini/TTS delay ---
send_robot_command(user_text)
print("[Status] Gemini is thinking...")
response = gemini_client.models.generate_content(
model=MODEL_ID,
contents=f"The user said: '{user_text}'. Short response for robot control.",
config={'system_instruction': 'Respond naturally and briefly.'}
)
reply_text = response.text
print(f"\n[Gemini]: {reply_text}")
text_to_speech(reply_text)
except sr.WaitTimeoutError:
print("[Mic] No speech detected.")
except sr.UnknownValueError:
print("[Mic] Could not understand.")
except Exception as e:
print(f"[Error] {e}")
if __name__ == "__main__":
print(f"--- Robot Voice Controller ({MODEL_ID}) ---")
while True:
user_input = input("\nPress Enter to speak, or 'q' to quit: ")
if user_input.lower() == 'q':
if ser:
ser.write(b'x')
ser.close()
break
listen_and_talk()