Add private realtime Talk for OpenClaw

This commit is contained in:
Mikei386
2026-09-03 22:52:15 +02:00
parent 42ec28c9f6
commit 0a68df22eb
10 changed files with 829 additions and 11 deletions
+75 -2
View File
@@ -3,7 +3,8 @@
STT-Worker – langlebiger Whisper-Transkriptions-Service (CPU-only).
Liest Audio-Dateien (WAV, MP3, OGG, FLAC, WebM/Opus via ffmpeg),
transkribiert sie mit whisper.cpp (whisper-cli) und liefert JSON-Text.
transkribiert sie mit einem dauerhaft geladenen whisper.cpp-Server (mit
whisper-cli als Rückfallweg) und liefert JSON-Text.
Konfiguration über Umgebungsvariablen:
WHISPER_HOST Bind-Adresse (Default: 127.0.0.1)
@@ -28,6 +29,7 @@ import sys
import tempfile
import time
import uuid
import urllib.request
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
# ---------------------------------------------------------------------------
@@ -45,6 +47,7 @@ WHISPER_MODEL = os.environ.get(
)
WHISPER_THREADS = int(os.environ.get("WHISPER_THREADS", "8"))
WHISPER_LANGUAGE = os.environ.get("WHISPER_LANGUAGE", "de")
WHISPER_SERVER_URL = os.environ.get("WHISPER_SERVER_URL", "").rstrip("/")
FFMPEG_BIN = os.environ.get("FFMPEG_BIN", "/usr/bin/ffmpeg")
LOG_LEVEL = os.environ.get("LOG_LEVEL", "INFO")
@@ -139,13 +142,22 @@ def transcribe(
temperature: float | None = None,
) -> dict:
"""
Führt die Transkription mit whisper-cli aus.
Führt die Transkription bevorzugt über den persistenten whisper.cpp-
Server aus. Dadurch wird das Modell nicht pro Aufnahme neu geladen.
Liefert dict mit 'text' und Metadaten.
"""
lang = language or WHISPER_LANGUAGE
if lang == "auto":
lang = "auto"
if WHISPER_SERVER_URL:
return _transcribe_via_server(
audio_path,
language=lang,
prompt=prompt,
temperature=temperature,
)
out_prefix = f"/tmp/stt_{uuid.uuid4().hex[:12]}"
out_json = out_prefix + ".json"
@@ -217,6 +229,65 @@ def transcribe(
return result
def _transcribe_via_server(
audio_path: str,
language: str,
prompt: str | None = None,
temperature: float | None = None,
) -> dict:
"""Sendet eine Aufnahme an den bereits geladenen whisper-server."""
boundary = f"----athena-whisper-{uuid.uuid4().hex}"
chunks: list[bytes] = []
def add_field(name: str, value: str) -> None:
chunks.extend([
f"--{boundary}\r\n".encode(),
f'Content-Disposition: form-data; name="{name}"\r\n\r\n'.encode(),
value.encode("utf-8"),
b"\r\n",
])
with open(audio_path, "rb") as f:
audio = f.read()
chunks.extend([
f"--{boundary}\r\n".encode(),
b'Content-Disposition: form-data; name="file"; filename="audio.wav"\r\n',
b"Content-Type: audio/wav\r\n\r\n",
audio,
b"\r\n",
])
add_field("response_format", "json")
add_field("language", language)
if prompt:
add_field("prompt", prompt)
if temperature is not None:
add_field("temperature", str(temperature))
chunks.append(f"--{boundary}--\r\n".encode())
request = urllib.request.Request(
f"{WHISPER_SERVER_URL}/inference",
data=b"".join(chunks),
headers={"Content-Type": f"multipart/form-data; boundary={boundary}"},
method="POST",
)
t0 = time.monotonic()
with urllib.request.urlopen(request, timeout=300) as response:
payload = json.loads(response.read().decode("utf-8"))
elapsed = time.monotonic() - t0
text = payload.get("text", "") if isinstance(payload, dict) else ""
result = {
"text": text.strip(),
"language": language,
"duration_ms": int(elapsed * 1000),
"engine": "whisper-server",
}
log.info(
"Transkription (persistent): %d ms, %d Zeichen, Sprache=%s",
result["duration_ms"], len(result["text"]), language,
)
return result
# ---------------------------------------------------------------------------
# HTTP-Handler
# ---------------------------------------------------------------------------
@@ -300,6 +371,8 @@ class STTHandler(BaseHTTPRequestHandler):
"whisper_cli_exists": cli_ok,
"threads": WHISPER_THREADS,
"language": WHISPER_LANGUAGE,
"server_url": WHISPER_SERVER_URL or None,
"persistent_model": bool(WHISPER_SERVER_URL),
"ffmpeg": FFMPEG_BIN,
"ffmpeg_exists": os.path.isfile(FFMPEG_BIN),
})