265 lines
8.4 KiB
Markdown
265 lines
8.4 KiB
Markdown
# Server implementation notes
|
||
|
||
The STT server listens for TCP connections from Robovoice, captures audio
|
||
from a microphone when `on` is received, runs speech recognition (Moonshine),
|
||
and sends transcript messages back over the same connection.
|
||
|
||
## Architecture
|
||
|
||
```
|
||
┌── on/off (TCP, newline-delimited JSON)
|
||
Robovoice ──────────────►│
|
||
│ STT Server
|
||
Robovoice ◄──────────────┤
|
||
└── partial/final (TCP, newline-delimited JSON)
|
||
```
|
||
|
||
The server:
|
||
1. Listens on a TCP port (e.g. 5210)
|
||
2. Accepts a connection from Robovoice
|
||
3. Reads lines: waits for `{"event":"on"}`
|
||
4. Records audio from the microphone
|
||
5. Waits for `{"event":"off"}` (or a timeout)
|
||
6. Runs STT on the captured audio
|
||
7. Sends `{"final":true,"text":"..."}`\n back over the connection
|
||
|
||
## Framing
|
||
|
||
Every message is a single JSON object on one line, terminated by `\n`. No
|
||
length prefix, no binary framing. Use `readline()` / `StreamReader.ReadLineAsync()`.
|
||
|
||
## Python server with Moonshine
|
||
|
||
[Moonshine](https://github.com/usefulsensors/moonshine) is a lightweight ASR
|
||
model by Useful Sensors. Install with `pip install moonshine`.
|
||
|
||
```python
|
||
import socket
|
||
import json
|
||
import numpy as np
|
||
import sounddevice as sd
|
||
import moonshine
|
||
|
||
LISTEN_PORT = 5210
|
||
SAMPLE_RATE = 16000
|
||
|
||
model = moonshine.MoonshineModel(model="moonshine/base")
|
||
|
||
server = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
|
||
server.bind(("0.0.0.0", LISTEN_PORT))
|
||
server.listen(1)
|
||
|
||
print(f"STT server listening on :{LISTEN_PORT}")
|
||
|
||
while True:
|
||
conn, addr = server.accept()
|
||
print(f"Client connected: {addr}")
|
||
|
||
buf = ""
|
||
with conn:
|
||
while True:
|
||
data = conn.recv(4096).decode("utf-8")
|
||
if not data:
|
||
break
|
||
buf += data
|
||
|
||
while "\n" in buf:
|
||
line, buf = buf.split("\n", 1)
|
||
msg = json.loads(line)
|
||
|
||
if msg.get("event") == "on":
|
||
print("PTT on — recording")
|
||
audio_chunks = []
|
||
|
||
# Record until "off" or timeout
|
||
conn.settimeout(0.1)
|
||
while True:
|
||
try:
|
||
data2 = conn.recv(4096).decode("utf-8")
|
||
if not data2:
|
||
break
|
||
buf += data2
|
||
while "\n" in buf:
|
||
line2, buf = buf.split("\n", 1)
|
||
msg2 = json.loads(line2)
|
||
if msg2.get("event") == "off":
|
||
break
|
||
except socket.timeout:
|
||
pass
|
||
|
||
chunk = sd.rec(int(SAMPLE_RATE * 0.1),
|
||
samplerate=SAMPLE_RATE,
|
||
channels=1, dtype="float32")
|
||
sd.wait()
|
||
audio_chunks.append(chunk.flatten())
|
||
|
||
conn.settimeout(None)
|
||
|
||
if not audio_chunks:
|
||
continue
|
||
|
||
audio = np.concatenate(audio_chunks)
|
||
print(f"Captured {len(audio)/SAMPLE_RATE:.1f}s")
|
||
|
||
text = moonshine.transcribe(model, audio).strip()
|
||
|
||
if text:
|
||
print(f"Transcript: {text}")
|
||
reply = json.dumps({"final": True, "text": text})
|
||
conn.sendall((reply + "\n").encode("utf-8"))
|
||
else:
|
||
print("Empty transcript")
|
||
```
|
||
|
||
## Python server with streaming partials
|
||
|
||
For lower latency, send partial results while still recording:
|
||
|
||
```python
|
||
import socket
|
||
import json
|
||
import numpy as np
|
||
import sounddevice as sd
|
||
import moonshine
|
||
|
||
LISTEN_PORT = 5210
|
||
SAMPLE_RATE = 16000
|
||
CHUNK_DURATION = 0.5
|
||
|
||
model = moonshine.MoonshineModel(model="moonshine/base")
|
||
|
||
server = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
|
||
server.bind(("0.0.0.0", LISTEN_PORT))
|
||
server.listen(1)
|
||
|
||
print(f"STT server listening on :{LISTEN_PORT}")
|
||
|
||
while True:
|
||
conn, addr = server.accept()
|
||
print(f"Client connected: {addr}")
|
||
buf = ""
|
||
|
||
with conn:
|
||
while True:
|
||
data = conn.recv(4096).decode("utf-8")
|
||
if not data:
|
||
break
|
||
buf += data
|
||
|
||
while "\n" in buf:
|
||
line, buf = buf.split("\n", 1)
|
||
msg = json.loads(line)
|
||
|
||
if msg.get("event") != "on":
|
||
continue
|
||
|
||
print("PTT on — recording")
|
||
audio_chunks = []
|
||
|
||
while True:
|
||
try:
|
||
conn.settimeout(CHUNK_DURATION)
|
||
data2 = conn.recv(4096).decode("utf-8")
|
||
if not data2:
|
||
break
|
||
buf += data2
|
||
while "\n" in buf:
|
||
line2, buf = buf.split("\n", 1)
|
||
msg2 = json.loads(line2)
|
||
if msg2.get("event") == "off":
|
||
break
|
||
except socket.timeout:
|
||
pass
|
||
|
||
chunk = sd.rec(int(SAMPLE_RATE * CHUNK_DURATION),
|
||
samplerate=SAMPLE_RATE,
|
||
channels=1, dtype="float32")
|
||
sd.wait()
|
||
audio_chunks.append(chunk.flatten())
|
||
|
||
# Send partial every few chunks
|
||
if len(audio_chunks) % 4 == 0:
|
||
partial_audio = np.concatenate(audio_chunks)
|
||
partial_text = moonshine.transcribe(model, partial_audio).strip()
|
||
if partial_text:
|
||
reply = json.dumps({"final": False, "text": partial_text})
|
||
conn.sendall((reply + "\n").encode("utf-8"))
|
||
|
||
conn.settimeout(None)
|
||
|
||
if not audio_chunks:
|
||
continue
|
||
|
||
audio = np.concatenate(audio_chunks)
|
||
text = moonshine.transcribe(model, audio).strip()
|
||
|
||
if text:
|
||
print(f"Final: {text}")
|
||
reply = json.dumps({"final": True, "text": text})
|
||
conn.sendall((reply + "\n").encode("utf-8"))
|
||
```
|
||
|
||
## C# server skeleton
|
||
|
||
```csharp
|
||
using System.Net;
|
||
using System.Net.Sockets;
|
||
using System.Text.Json;
|
||
|
||
var listener = new TcpListener(IPAddress.Any, 5210);
|
||
listener.Start();
|
||
|
||
Console.WriteLine("STT server listening on :5210");
|
||
|
||
while (true)
|
||
{
|
||
var client = listener.AcceptTcpClient();
|
||
Console.WriteLine($"Client connected: {client.Client.RemoteEndPoint}");
|
||
|
||
using var stream = client.GetStream();
|
||
using var reader = new StreamReader(stream, Encoding.UTF8);
|
||
using var writer = new StreamWriter(stream, Encoding.UTF8) { AutoFlush = true };
|
||
|
||
string? line;
|
||
while ((line = reader.ReadLine()) is not null)
|
||
{
|
||
var msg = JsonSerializer.Deserialize<Dictionary<string, string>>(line);
|
||
if (msg?["event"] != "on")
|
||
continue;
|
||
|
||
Console.WriteLine("PTT on — recording");
|
||
// Capture audio...
|
||
|
||
// Read until "off"
|
||
while ((line = reader.ReadLine()) is not null)
|
||
{
|
||
msg = JsonSerializer.Deserialize<Dictionary<string, string>>(line);
|
||
if (msg?["event"] == "off")
|
||
break;
|
||
}
|
||
|
||
// Run STT...
|
||
string text = "recognized text here";
|
||
|
||
var reply = JsonSerializer.Serialize(new { final = true, text });
|
||
writer.WriteLine(reply);
|
||
}
|
||
}
|
||
```
|
||
|
||
## Tips
|
||
|
||
- **One connection per client:** Robovoice maintains a single persistent TCP
|
||
connection. The server should handle one client at a time (or track
|
||
multiple if needed).
|
||
- **Timeout:** implement a recording timeout in case the `off` message is
|
||
delayed or the client disconnects. 10–30 seconds is reasonable.
|
||
- **Partials:** optional but improve UX — Robovoice logs them so the user
|
||
sees live feedback. Only `final` triggers TTS.
|
||
- **Encoding:** always UTF-8. Every line is a UTF-8 JSON object terminated
|
||
by `\n`.
|
||
- **Reconnection:** Robovoice auto-reconnects every 3 seconds if the
|
||
connection drops. The server just needs to accept new connections.
|
||
- **Moonshine models:** `moonshine/base` (faster, less accurate) or
|
||
`moonshine/tiny` (fastest). Choose based on your hardware.
|