Fix OpenClaw realtime transcript item identity

This commit is contained in:
Mikei386
2026-09-16 14:26:16 +02:00
parent 4239b217eb
commit b9dcadedd5
3 changed files with 46 additions and 2 deletions
+5
View File
@@ -12,6 +12,11 @@ WebRTC; Athena transcribes it; the service asks OpenClaw to run
connection. This is a half-duplex prototype. Barge-in and remote-network TURN
support are not implemented.
The bridge announces each committed user audio item before sending its completed
transcript with the same item ID. OpenClaw requires that sequence to persist the
transcript; omitting the item caused “Realtime transcript refers to an unknown
speech item” in the browser.
Required service environment:
| Name | Purpose |
+4 -1
View File
@@ -263,8 +263,11 @@ class RealtimeSession:
if not text:
self.busy = False
return
item_id = f"item_{uuid.uuid4().hex}"
# OpenClaw tracks transcript items before accepting their text.
self.emit({"type": "input_audio_buffer.committed", "item_id": item_id})
self.emit({"type": "conversation.item.input_audio_transcription.completed",
"item_id": f"item_{uuid.uuid4().hex}", "transcript": text})
"item_id": item_id, "transcript": text})
await self.consult(text)
except asyncio.CancelledError:
self.busy = False
+37 -1
View File
@@ -18,7 +18,7 @@ from av import AudioFrame
from cryptography.hazmat.primitives.asymmetric.ed25519 import Ed25519PrivateKey
from cryptography.hazmat.primitives.serialization import Encoding, PublicFormat
from server import FRAME_SAMPLES, SAMPLE_RATE, create_app, decode_public_key_token, make_wav
from server import FRAME_SAMPLES, SAMPLE_RATE, RealtimeSession, create_app, decode_public_key_token, make_wav
class Microphone(MediaStreamTrack):
@@ -56,6 +56,36 @@ async def serve(app):
class VoiceSmokeTest(unittest.IsolatedAsyncioTestCase):
async def test_transcript_item_announced_before_completion(self):
async def stt(_):
return web.json_response({"text": "Hallo Athena"})
fake = web.Application()
fake.router.add_post("/audio/transcriptions", stt)
runner, base = await serve(fake)
os.environ["ATHENA_API_BASE_URL"] = base
events = []
class Channel:
readyState = "open"
def send(self, message):
events.append(json.loads(message))
try:
async with ClientSession() as http:
session = RealtimeSession(None, http, None)
session.channel = Channel()
await session.transcribe_and_consult(b"\0\0" * SAMPLE_RATE)
committed = next((index, event) for index, event in enumerate(events)
if event.get("type") == "input_audio_buffer.committed")
completed = next((index, event) for index, event in enumerate(events)
if event.get("type") == "conversation.item.input_audio_transcription.completed")
self.assertLess(committed[0], completed[0])
self.assertEqual(committed[1]["item_id"], completed[1]["item_id"])
finally:
await runner.cleanup()
async def test_public_key_session_token(self):
private = Ed25519PrivateKey.generate()
pem = private.public_key().public_bytes(Encoding.PEM, PublicFormat.SubjectPublicKeyInfo)
@@ -163,6 +193,12 @@ class VoiceSmokeTest(unittest.IsolatedAsyncioTestCase):
self.assertTrue(any(event.get("type") ==
"conversation.item.input_audio_transcription.completed"
for event in events))
committed = next((index, event) for index, event in enumerate(events)
if event.get("type") == "input_audio_buffer.committed")
completed = next((index, event) for index, event in enumerate(events)
if event.get("type") == "conversation.item.input_audio_transcription.completed")
self.assertLess(committed[0], completed[0])
self.assertEqual(committed[1]["item_id"], completed[1]["item_id"])
self.assertTrue(any(event.get("type") ==
"response.output_audio_transcript.done" for event in events))
finally: