Fix OpenClaw realtime transcript item identity
This commit is contained in:
@@ -12,6 +12,11 @@ WebRTC; Athena transcribes it; the service asks OpenClaw to run
|
|||||||
connection. This is a half-duplex prototype. Barge-in and remote-network TURN
|
connection. This is a half-duplex prototype. Barge-in and remote-network TURN
|
||||||
support are not implemented.
|
support are not implemented.
|
||||||
|
|
||||||
|
The bridge announces each committed user audio item before sending its completed
|
||||||
|
transcript with the same item ID. OpenClaw requires that sequence to persist the
|
||||||
|
transcript; omitting the item caused “Realtime transcript refers to an unknown
|
||||||
|
speech item” in the browser.
|
||||||
|
|
||||||
Required service environment:
|
Required service environment:
|
||||||
|
|
||||||
| Name | Purpose |
|
| Name | Purpose |
|
||||||
|
|||||||
@@ -263,8 +263,11 @@ class RealtimeSession:
|
|||||||
if not text:
|
if not text:
|
||||||
self.busy = False
|
self.busy = False
|
||||||
return
|
return
|
||||||
|
item_id = f"item_{uuid.uuid4().hex}"
|
||||||
|
# OpenClaw tracks transcript items before accepting their text.
|
||||||
|
self.emit({"type": "input_audio_buffer.committed", "item_id": item_id})
|
||||||
self.emit({"type": "conversation.item.input_audio_transcription.completed",
|
self.emit({"type": "conversation.item.input_audio_transcription.completed",
|
||||||
"item_id": f"item_{uuid.uuid4().hex}", "transcript": text})
|
"item_id": item_id, "transcript": text})
|
||||||
await self.consult(text)
|
await self.consult(text)
|
||||||
except asyncio.CancelledError:
|
except asyncio.CancelledError:
|
||||||
self.busy = False
|
self.busy = False
|
||||||
|
|||||||
@@ -18,7 +18,7 @@ from av import AudioFrame
|
|||||||
from cryptography.hazmat.primitives.asymmetric.ed25519 import Ed25519PrivateKey
|
from cryptography.hazmat.primitives.asymmetric.ed25519 import Ed25519PrivateKey
|
||||||
from cryptography.hazmat.primitives.serialization import Encoding, PublicFormat
|
from cryptography.hazmat.primitives.serialization import Encoding, PublicFormat
|
||||||
|
|
||||||
from server import FRAME_SAMPLES, SAMPLE_RATE, create_app, decode_public_key_token, make_wav
|
from server import FRAME_SAMPLES, SAMPLE_RATE, RealtimeSession, create_app, decode_public_key_token, make_wav
|
||||||
|
|
||||||
|
|
||||||
class Microphone(MediaStreamTrack):
|
class Microphone(MediaStreamTrack):
|
||||||
@@ -56,6 +56,36 @@ async def serve(app):
|
|||||||
|
|
||||||
|
|
||||||
class VoiceSmokeTest(unittest.IsolatedAsyncioTestCase):
|
class VoiceSmokeTest(unittest.IsolatedAsyncioTestCase):
|
||||||
|
async def test_transcript_item_announced_before_completion(self):
|
||||||
|
async def stt(_):
|
||||||
|
return web.json_response({"text": "Hallo Athena"})
|
||||||
|
|
||||||
|
fake = web.Application()
|
||||||
|
fake.router.add_post("/audio/transcriptions", stt)
|
||||||
|
runner, base = await serve(fake)
|
||||||
|
os.environ["ATHENA_API_BASE_URL"] = base
|
||||||
|
events = []
|
||||||
|
|
||||||
|
class Channel:
|
||||||
|
readyState = "open"
|
||||||
|
|
||||||
|
def send(self, message):
|
||||||
|
events.append(json.loads(message))
|
||||||
|
|
||||||
|
try:
|
||||||
|
async with ClientSession() as http:
|
||||||
|
session = RealtimeSession(None, http, None)
|
||||||
|
session.channel = Channel()
|
||||||
|
await session.transcribe_and_consult(b"\0\0" * SAMPLE_RATE)
|
||||||
|
committed = next((index, event) for index, event in enumerate(events)
|
||||||
|
if event.get("type") == "input_audio_buffer.committed")
|
||||||
|
completed = next((index, event) for index, event in enumerate(events)
|
||||||
|
if event.get("type") == "conversation.item.input_audio_transcription.completed")
|
||||||
|
self.assertLess(committed[0], completed[0])
|
||||||
|
self.assertEqual(committed[1]["item_id"], completed[1]["item_id"])
|
||||||
|
finally:
|
||||||
|
await runner.cleanup()
|
||||||
|
|
||||||
async def test_public_key_session_token(self):
|
async def test_public_key_session_token(self):
|
||||||
private = Ed25519PrivateKey.generate()
|
private = Ed25519PrivateKey.generate()
|
||||||
pem = private.public_key().public_bytes(Encoding.PEM, PublicFormat.SubjectPublicKeyInfo)
|
pem = private.public_key().public_bytes(Encoding.PEM, PublicFormat.SubjectPublicKeyInfo)
|
||||||
@@ -163,6 +193,12 @@ class VoiceSmokeTest(unittest.IsolatedAsyncioTestCase):
|
|||||||
self.assertTrue(any(event.get("type") ==
|
self.assertTrue(any(event.get("type") ==
|
||||||
"conversation.item.input_audio_transcription.completed"
|
"conversation.item.input_audio_transcription.completed"
|
||||||
for event in events))
|
for event in events))
|
||||||
|
committed = next((index, event) for index, event in enumerate(events)
|
||||||
|
if event.get("type") == "input_audio_buffer.committed")
|
||||||
|
completed = next((index, event) for index, event in enumerate(events)
|
||||||
|
if event.get("type") == "conversation.item.input_audio_transcription.completed")
|
||||||
|
self.assertLess(committed[0], completed[0])
|
||||||
|
self.assertEqual(committed[1]["item_id"], completed[1]["item_id"])
|
||||||
self.assertTrue(any(event.get("type") ==
|
self.assertTrue(any(event.get("type") ==
|
||||||
"response.output_audio_transcript.done" for event in events))
|
"response.output_audio_transcript.done" for event in events))
|
||||||
finally:
|
finally:
|
||||||
|
|||||||
Reference in New Issue
Block a user