Skip to content

Commit 84e6e99

Browse files
authored
DRU-651 -- Turn chat voice notes into text for the agent (#727)
1 parent 37f79a1 commit 84e6e99

15 files changed

Lines changed: 200 additions & 19 deletions

File tree

‎backend/druks/chat/constants.py‎

Lines changed: 7 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -3,6 +3,7 @@
33
# The dot keeps the name outside NAME_PATTERN, so no registry server can share its vault row.
44
CHAT_KEY_NAME = f"{DRUKS_SERVER_NAME}.chat"
55
CHAT_BRIDGE_PORT = 43123
6+
TRANSCRIPTION_TIMEOUT_SECONDS = 120.0
67
# The header an agent's MCP calls carry to name their conversation.
78
CONVERSATION_HEADER = "X-Druks-Conversation"
89
# Every agent's prompt ends with this, so the agent knows an internal message when it
@@ -18,3 +19,9 @@
1819
"[Internal: Run {run} failed: {failure}. Tell the person in one short line, with no "
1920
"error details, and do not retry it.]"
2021
)
22+
# The line between what a person typed and what they said in the same message.
23+
VOICE_NOTE_MARKER = "[Voice note]"
24+
TRANSCRIPTION_FAILED_MESSAGE = (
25+
"[Internal: Druks could not turn the person's voice note into text. Tell the person "
26+
"in one short line to write it instead.]"
27+
)

‎backend/druks/chat/exceptions.py‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -23,3 +23,7 @@ class ChatBridgeUnavailable(ChatBridgeError):
2323

2424
class ChannelHasNoThreadsError(ChatError):
2525
"""The conversation's channel has no thread to read."""
26+
27+
28+
class TranscriptionError(ChatError):
29+
"""Druks got no transcript for a voice note."""

‎backend/druks/chat/models.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -37,6 +37,8 @@ class Message(Base, Uuid7Pk):
3737
# reply, so the source's copy of it is known as Druks's own.
3838
source_id: Mapped[str | None] = mapped_column(unique=True)
3939
file: Mapped[File | None] = FileField()
40+
# What the person said in the message's voice note.
41+
transcript: Mapped[str] = mapped_column(default="", server_default=text("''"))
4042
created_at: Mapped[datetime] = mapped_column(default=Base.utc_now)
4143
delivered_at: Mapped[datetime | None]
4244

‎backend/druks/chat/schemas.py‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -34,6 +34,7 @@ class MessageResponse(Schema):
3434
tool_calls: list[dict]
3535
is_internal: bool
3636
file: FileSummary | None
37+
transcript: str
3738
created_at: datetime
3839
delivered_at: datetime | None
3940

‎backend/druks/chat/service.py‎

Lines changed: 49 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -38,6 +38,7 @@
3838
from druks.sandbox.layout import get_remote_home, get_work_root
3939
from druks.sandbox.models import SandboxIdentity, SecretRef
4040
from druks.sandbox.templates import get_template_id
41+
from druks.services.exceptions import ServiceNotConnectedError
4142
from druks.workspaces import Workspace
4243

4344
from .bots.constants import ADMIN_PROMPT, ADMIN_TOOLS
@@ -48,11 +49,14 @@
4849
FAILURE_MESSAGE,
4950
INTERNAL_MESSAGES_PROMPT,
5051
RESULT_MESSAGE,
52+
TRANSCRIPTION_FAILED_MESSAGE,
53+
VOICE_NOTE_MARKER,
5154
)
5255
from .enums import MessageRole, MessageState
53-
from .exceptions import ChatBridgeError, ChatHarnessError, ChatSandboxGone
56+
from .exceptions import ChatBridgeError, ChatHarnessError, ChatSandboxGone, TranscriptionError
5457
from .models import Conversation, Message
5558
from .sandbox import CHAT_SANDBOX
59+
from .services import SpeechToText
5660

5761
logger = logging.getLogger(__name__)
5862

@@ -235,8 +239,8 @@ async def send_turn(
235239
config: AgentConfig,
236240
prompt: str,
237241
) -> Message | None:
238-
"""Start the agent and send the pending messages.
239-
Return the turn's message, unless a Stop or pause came first."""
242+
"""Start the agent, transcribe the pending voice notes, and send the pending
243+
messages. Return the turn's message, unless a Stop or pause came first."""
240244
host = bridge.host
241245
status = await bridge.request("status", conversationId=conversation.id)
242246
if status["status"] == "running":
@@ -274,13 +278,32 @@ async def send_turn(
274278
expires_at = Base.utc_now() + timedelta(seconds=SANDBOX_HOST_LEASE_SECONDS)
275279
await sandbox_client.set_expiry(host_id=host.id, expires_at=expires_at)
276280
identity.expires_at = expires_at
281+
# The account's other conversations write this row too. Release it before the
282+
# transcription calls.
283+
await session.commit()
277284
messages = [message]
278285
if conversation.connection:
279286
# The sandbox can take seconds to start, and a person can take the chat over meanwhile.
280287
if await conversation.is_held(session):
281288
return
282289
messages = await conversation.list_pending_messages(session)
283-
delivered_messages = [pending for pending in messages if await pending.mark_delivered(session)]
290+
notes = []
291+
for pending in messages:
292+
file = pending.file
293+
if file and file.content_type.startswith("audio/"):
294+
try:
295+
pending.transcript = await get_transcript(session, file)
296+
except (TranscriptionError, ServiceNotConnectedError) as error:
297+
logger.warning("Chat message %s has no transcript: %s", pending.id, error)
298+
notes.append(
299+
await conversation.create_message(
300+
session, TRANSCRIPTION_FAILED_MESSAGE, is_internal=True
301+
)
302+
)
303+
# The notes are the newest messages, so they close the turn.
304+
delivered_messages = [
305+
pending for pending in (*messages, *notes) if await pending.mark_delivered(session)
306+
]
284307
await session.commit()
285308
if delivered_messages:
286309
await bridge.request(
@@ -295,16 +318,33 @@ async def send_turn(
295318
return
296319

297320

321+
async def get_transcript(session: AsyncSession, file: File) -> str:
322+
"""The words in a voice note, from the Speech To Text card. Druks refuses a note
323+
over the upload cap before the call."""
324+
if file.size > MAX_UPLOAD_BYTES:
325+
raise TranscriptionError(
326+
f"The voice note is {file.size} bytes. The cap is {MAX_UPLOAD_BYTES} bytes."
327+
)
328+
content = get_file_storage().open(file.id)
329+
return await SpeechToText.transcribe(
330+
session, name=file.name, content_type=file.content_type, content=content
331+
)
332+
333+
298334
async def get_turn_content(
299335
host: Host, conversation_root: str, messages: list[Message]
300336
) -> list[dict]:
301-
"""The ACP content blocks the agent reads: each message's text, then its file. An
302-
image travels in the prompt. Audio adds nothing. Any other file goes to the
303-
conversation's folder in the sandbox, and the agent gets a link to it."""
337+
"""The ACP content blocks the agent reads: each message's text, with the words of
338+
its voice note under a marker, then its file. An image travels in the prompt. Audio
339+
adds nothing more. Any other file goes to the conversation's folder in the sandbox,
340+
and the agent gets a link to it."""
304341
content = []
305342
for message in messages:
306-
if message.body:
307-
content.append({"type": "text", "text": message.body})
343+
parts = [message.body]
344+
if message.transcript:
345+
parts += [VOICE_NOTE_MARKER, message.transcript]
346+
if text := "\n".join(part for part in parts if part):
347+
content.append({"type": "text", "text": text})
308348
if file := message.file:
309349
if file.content_type.startswith("image/") and file.size <= MAX_UPLOAD_BYTES:
310350
image = get_file_storage().open(file.id)

‎backend/druks/chat/services.py‎

Lines changed: 53 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,53 @@
1+
import httpx
2+
from pydantic import BaseModel, Field, SecretStr
3+
from sqlalchemy.ext.asyncio import AsyncSession
4+
5+
from druks.secrets.datastructures import Audience
6+
from druks.secrets.models import VaultSecret
7+
from druks.services import Service
8+
from druks.services.exceptions import ServiceNotConnectedError
9+
10+
from .constants import TRANSCRIPTION_TIMEOUT_SECONDS
11+
from .exceptions import TranscriptionError
12+
13+
14+
class SpeechToText(Service):
15+
"""The server Druks sends voice notes to: any server that speaks the OpenAI audio
16+
API, such as OpenAI, Groq, or a local one."""
17+
18+
description = (
19+
"The service Druks sends voice notes to. Any server that speaks the OpenAI audio API."
20+
)
21+
required = False
22+
23+
class Settings(BaseModel):
24+
url: str = Field(
25+
title="Address", description="Base URL, for example https://api.openai.com/v1."
26+
)
27+
key: SecretStr = Field(title="Key")
28+
model: str = Field(title="Model", description="For example whisper-1.")
29+
30+
@classmethod
31+
async def transcribe(
32+
cls, session: AsyncSession, *, name: str, content_type: str, content: bytes
33+
) -> str:
34+
"""The words in an audio file. An empty answer is a failure."""
35+
card = await VaultSecret.lookup(session, cls.secret_kind, Audience.service(cls.slug))
36+
if not card:
37+
raise ServiceNotConnectedError(cls.slug)
38+
try:
39+
async with httpx.AsyncClient(timeout=TRANSCRIPTION_TIMEOUT_SECONDS) as client:
40+
response = await client.post(
41+
f"{card.identity['url'].rstrip('/')}/audio/transcriptions",
42+
# A key pasted with a space would echo through the transport's error.
43+
headers={"Authorization": f"Bearer {card.secrets['key'].strip()}"},
44+
data={"model": card.identity["model"]},
45+
files={"file": (name, content, content_type)},
46+
)
47+
response.raise_for_status()
48+
text = response.json()["text"].strip()
49+
except Exception as error: # noqa: BLE001 — any transport or shape failure is a failed note
50+
raise TranscriptionError(f"{cls.title} gave no transcript: {error}") from error
51+
if not text:
52+
raise TranscriptionError(f"{cls.title} gave an empty transcript.")
53+
return text
Lines changed: 27 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,27 @@
1+
"""Keep the words of a chat message's voice note.
2+
3+
Revision ID: e613b1a81427
4+
Revises: 2f9d7f55982a
5+
Create Date: 2026-09-26
6+
"""
7+
8+
from collections.abc import Sequence
9+
10+
import sqlalchemy as sa
11+
from alembic import op
12+
13+
revision: str = "e613b1a81427"
14+
down_revision: str | Sequence[str] | None = "2f9d7f55982a"
15+
branch_labels: str | Sequence[str] | None = None
16+
depends_on: str | Sequence[str] | None = None
17+
18+
19+
def upgrade() -> None:
20+
op.add_column(
21+
"chat_messages",
22+
sa.Column("transcript", sa.String(), nullable=False, server_default=""),
23+
)
24+
25+
26+
def downgrade() -> None:
27+
op.drop_column("chat_messages", "transcript")

‎backend/tests/test_whatsapp.py‎

Lines changed: 26 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -29,14 +29,20 @@
2929
from druks.chat.channels.whatsapp.constants import WAHA_AUDIENCE
3030
from druks.chat.channels.whatsapp.services import Waha
3131
from druks.chat.channels.whatsapp.webhooks import WahaEvents
32-
from druks.chat.constants import CONVERSATION_HEADER, INTERNAL_MESSAGES_PROMPT
32+
from druks.chat.constants import (
33+
CONVERSATION_HEADER,
34+
INTERNAL_MESSAGES_PROMPT,
35+
TRANSCRIPTION_FAILED_MESSAGE,
36+
VOICE_NOTE_MARKER,
37+
)
3338
from druks.chat.enums import (
3439
BotAccess,
3540
ConversationSource,
3641
MessageRole,
3742
MessageState,
3843
PauseSignal,
3944
)
45+
from druks.chat.exceptions import TranscriptionError
4046
from druks.chat.models import Conversation
4147
from druks.harnesses.claude import ClaudeHarness
4248
from druks.mcp.enums import Toolkit
@@ -348,13 +354,19 @@ async def test_one_turn_answers_every_pending_message_and_knows_its_own_reply(
348354
photo["payload"].update(hasMedia=True, media=waha_media("photo.png", "image/png"))
349355
form = message_event(ANA, "Here is the form", key="M4")
350356
form["payload"].update(hasMedia=True, media=waha_media("form.pdf", "application/pdf"))
351-
note = message_event(ANA, "", key="M5")
357+
note = message_event(ANA, "Call me", key="M5")
352358
note["payload"].update(hasMedia=True, media=waha_media("note.oga", "audio/ogg"))
359+
failed_note = message_event(ANA, "", key="M6")
360+
failed_note["payload"].update(hasMedia=True, media=waha_media("again.oga", "audio/ogg"))
353361
await receive(connection, message_event(ANA, "Hello", key="M1"))
354362
await receive(connection, photo)
355363
await receive(connection, message_event(ANA, "And the second one?", key="M3"))
356364
await receive(connection, form)
357365
await receive(connection, note)
366+
await receive(connection, failed_note)
367+
await receive(connection, message_event(ANA, "Anyone there?", key="M7"))
368+
transcribe = AsyncMock(side_effect=["at six", TranscriptionError("The provider is down.")])
369+
monkeypatch.setattr(service.SpeechToText, "transcribe", transcribe)
358370
[conversation] = await Conversation.list_for_connection(druks_db, connection.id)
359371
config = SimpleNamespace(
360372
harness_class=ClaudeHarness,
@@ -418,14 +430,24 @@ async def copy_arrives_first() -> None:
418430
"name": "form.pdf",
419431
"mimeType": "application/pdf",
420432
},
433+
{"type": "text", "text": f"Call me\n{VOICE_NOTE_MARKER}\nat six"},
434+
{"type": "text", "text": "Anyone there?"},
435+
{"type": "text", "text": TRANSCRIPTION_FAILED_MESSAGE},
436+
]
437+
assert (prompt["timeout"], prompt["messageId"]) == (60, asked[-1].id)
438+
assert [(message.body, message.transcript) for message in asked[4:]] == [
439+
("Call me", "at six"),
440+
("", ""),
441+
("Anyone there?", ""),
442+
(TRANSCRIPTION_FAILED_MESSAGE, ""),
421443
]
422-
assert prompt["timeout"] == 60
444+
assert asked[-1].is_internal
423445
[upload] = host.upload_file.await_args_list
424446
assert (upload.kwargs["local"].is_file(), upload.kwargs["remote"]) == (True, copy)
425447
[start] = [values for method, values in requests if method == "start"]
426448
assert start["headers"] == [{"name": CONVERSATION_HEADER, "value": conversation.id}]
427449
assert start["meta"]["claudeCode"]["options"]["systemPrompt"] == "Be kind."
428-
assert [message.state for message in asked] == [MessageState.REPLIED] * 5
450+
assert [message.state for message in asked] == [MessageState.REPLIED] * 8
429451
assert reply.source_id == "REPLY1"
430452
sends = [body for method, path, body in waha.calls if path == "/api/sendText"]
431453
assert sends == [

‎docs/chat.md‎

Lines changed: 13 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -235,6 +235,15 @@ bulk. Before it sends a reply, Druks takes a new message ID from WAHA and
235235
records it. WAHA's copy of the sent message then carries a known ID, even when
236236
the copy arrives before the send returns.
237237

238+
A voice note becomes text before its turn. Druks sends the audio to the
239+
[Speech To Text card](configuration.md#speech-to-text) and saves the words on
240+
the message. The agent reads them under what the person typed, marked as a
241+
voice note, and answers in text. Nothing goes to the person before the reply.
242+
When Druks cannot transcribe a note, when no card is connected, or when the
243+
note is over 25 MiB, the agent gets an internal message instead. It then tells
244+
the person in one line to write instead. The web page shows the typed text, the
245+
words, and a link to the audio.
246+
238247
Druks also adds **internal messages** to a conversation. Each one comes from a
239248
fixed template. An internal message starts a turn like any message, and it
240249
never goes to WhatsApp. Druks talks to agents, and agents talk to people.
@@ -320,9 +329,10 @@ A file you send to the bot, in a direct message or in a thread you joined,
320329
becomes a Druks file on its message. A message with only a file starts a turn
321330
like any other. An image reaches the agent with its message. Any other file
322331
except audio reaches the agent as a link to a copy in the sandbox. The agent
323-
opens the copy with its tools. An image over 25 MiB goes as a link too. Each
324-
further file in one Slack message gets a message of its own. The agent sends no
325-
files back.
332+
opens the copy with its tools. An image over 25 MiB goes as a link too. An audio
333+
clip becomes text the way a WhatsApp voice note does: see
334+
[WhatsApp](#whatsapp). Each further file in one Slack message gets a message of
335+
its own. The agent sends no files back.
326336

327337
### Rooms
328338

‎docs/configuration.md‎

Lines changed: 11 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -484,6 +484,17 @@ event older than five minutes. With rotation off, the pasted bot token and each
484484
person's Slack token live until someone revokes them. A person connects their
485485
own Slack account through the same app: see [Chat](chat.md#slack).
486486

487+
## Speech to text
488+
489+
**Speech To Text** is the service that Druks sends voice notes to: any server
490+
that speaks the OpenAI audio API, such as OpenAI, Groq, or a local server.
491+
Connect it from **Settings → Connections → Services** with the server's base
492+
URL, for example `https://api.openai.com/v1`, a key, and a model, for example
493+
`whisper-1`. Druks checks none of the values when you save the card, so a wrong
494+
key shows up on the first voice note. Druks sends a note of at most 25 MiB and
495+
refuses a bigger one without a call. Without the card, the agent tells the
496+
person to write instead. See [Chat](chat.md#whatsapp) for what the agent gets.
497+
487498
## Harnesses
488499

489500
Druks registers two subscription providers, `anthropic` and `openai`. Each

0 commit comments

Comments
 (0)