Skip to content

Commit c632874

Browse files
committed
DRU-651 -- Turn chat voice notes into text for the agent
1 parent 14f0695 commit c632874

15 files changed

Lines changed: 195 additions & 18 deletions

File tree

‎backend/druks/chat/constants.py‎

Lines changed: 6 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -18,3 +18,9 @@
1818
"[Internal: Run {run} failed: {failure}. Tell the person in one short line, with no "
1919
"error details, and do not retry it.]"
2020
)
21+
# The line between what a person typed and what they said in the same message.
22+
VOICE_NOTE_MARKER = "[Voice note]"
23+
TRANSCRIPTION_FAILED_MESSAGE = (
24+
"[Internal: Druks could not turn the person's voice note into text. Tell the person "
25+
"in one short line to write it instead.]"
26+
)

‎backend/druks/chat/models.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -37,6 +37,8 @@ class Message(Base, Uuid7Pk):
3737
# reply, so the source's copy of it is known as Druks's own.
3838
source_id: Mapped[str | None] = mapped_column(unique=True)
3939
file: Mapped[File | None] = FileField()
40+
# What the person said in the message's voice note.
41+
transcript: Mapped[str] = mapped_column(default="", server_default=text("''"))
4042
created_at: Mapped[datetime] = mapped_column(default=Base.utc_now)
4143
delivered_at: Mapped[datetime | None]
4244

‎backend/druks/chat/schemas.py‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -34,6 +34,7 @@ class MessageResponse(Schema):
3434
tool_calls: list[dict]
3535
is_internal: bool
3636
file: FileSummary | None
37+
transcript: str
3738
created_at: datetime
3839
delivered_at: datetime | None
3940

‎backend/druks/chat/service.py‎

Lines changed: 49 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -15,6 +15,8 @@
1515
from druks.accounts.enums import AccountKind
1616
from druks.apps.loader import get_app
1717
from druks.apps.registry import channels
18+
from druks.core.exceptions import TranscriptionError
19+
from druks.core.services import SpeechToText
1820
from druks.durable.engine import step_session
1921
from druks.durable.models import Run
2022
from druks.files.constants import MAX_UPLOAD_BYTES
@@ -38,6 +40,7 @@
3840
from druks.sandbox.layout import get_remote_home, get_work_root
3941
from druks.sandbox.models import SandboxIdentity, SecretRef
4042
from druks.sandbox.templates import get_template_id
43+
from druks.services.exceptions import ServiceNotConnectedError
4144
from druks.workspaces import Workspace
4245

4346
from .bots.constants import ADMIN_PROMPT, ADMIN_TOOLS
@@ -48,6 +51,8 @@
4851
FAILURE_MESSAGE,
4952
INTERNAL_MESSAGES_PROMPT,
5053
RESULT_MESSAGE,
54+
TRANSCRIPTION_FAILED_MESSAGE,
55+
VOICE_NOTE_MARKER,
5156
)
5257
from .enums import MessageRole, MessageState
5358
from .exceptions import ChatBridgeError, ChatHarnessError, ChatSandboxGone
@@ -235,8 +240,8 @@ async def send_turn(
235240
config: AgentConfig,
236241
prompt: str,
237242
) -> Message | None:
238-
"""Start the agent and send the pending messages.
239-
Return the turn's message, unless a Stop or pause came first."""
243+
"""Start the agent, transcribe the pending voice notes, and send the pending
244+
messages. Return the turn's message, unless a Stop or pause came first."""
240245
host = bridge.host
241246
status = await bridge.request("status", conversationId=conversation.id)
242247
if status["status"] == "running":
@@ -274,13 +279,32 @@ async def send_turn(
274279
expires_at = Base.utc_now() + timedelta(seconds=SANDBOX_HOST_LEASE_SECONDS)
275280
await sandbox_client.set_expiry(host_id=host.id, expires_at=expires_at)
276281
identity.expires_at = expires_at
282+
# The account's other conversations write this row too. Release it before the
283+
# transcription calls.
284+
await session.commit()
277285
messages = [message]
278286
if conversation.connection:
279287
# The sandbox can take seconds to start, and a person can take the chat over meanwhile.
280288
if await conversation.is_held(session):
281289
return
282290
messages = await conversation.list_pending_messages(session)
283-
delivered_messages = [pending for pending in messages if await pending.mark_delivered(session)]
291+
notes = []
292+
for pending in messages:
293+
file = pending.file
294+
if file and file.content_type.startswith("audio/"):
295+
try:
296+
pending.transcript = await get_transcript(session, file)
297+
except (TranscriptionError, ServiceNotConnectedError) as error:
298+
logger.warning("Chat message %s has no transcript: %s", pending.id, error)
299+
notes.append(
300+
await conversation.create_message(
301+
session, TRANSCRIPTION_FAILED_MESSAGE, is_internal=True
302+
)
303+
)
304+
# The notes are the newest messages, so they close the turn.
305+
delivered_messages = [
306+
pending for pending in (*messages, *notes) if await pending.mark_delivered(session)
307+
]
284308
await session.commit()
285309
if delivered_messages:
286310
await bridge.request(
@@ -295,16 +319,33 @@ async def send_turn(
295319
return
296320

297321

322+
async def get_transcript(session: AsyncSession, file: File) -> str:
323+
"""The words in a voice note, from the Speech To Text card. Druks refuses a note
324+
over the upload cap before the call."""
325+
if file.size > MAX_UPLOAD_BYTES:
326+
raise TranscriptionError(
327+
f"The voice note is {file.size} bytes. The cap is {MAX_UPLOAD_BYTES} bytes."
328+
)
329+
content = await asyncio.to_thread(get_file_storage().open, file.id)
330+
return await SpeechToText.transcribe(
331+
session, name=file.name, content_type=file.content_type, content=content
332+
)
333+
334+
298335
async def get_turn_content(
299336
host: Host, conversation_root: str, messages: list[Message]
300337
) -> list[dict]:
301-
"""The ACP content blocks the agent reads: each message's text, then its file. An
302-
image travels in the prompt. Audio adds nothing. Any other file goes to the
303-
conversation's folder in the sandbox, and the agent gets a link to it."""
338+
"""The ACP content blocks the agent reads: each message's text, with the words of
339+
its voice note under a marker, then its file. An image travels in the prompt. Audio
340+
adds nothing more. Any other file goes to the conversation's folder in the sandbox,
341+
and the agent gets a link to it."""
304342
content = []
305343
for message in messages:
306-
if message.body:
307-
content.append({"type": "text", "text": message.body})
344+
parts = [message.body]
345+
if message.transcript:
346+
parts += [VOICE_NOTE_MARKER, message.transcript]
347+
if text := "\n".join(part for part in parts if part):
348+
content.append({"type": "text", "text": text})
308349
if file := message.file:
309350
if file.content_type.startswith("image/") and file.size <= MAX_UPLOAD_BYTES:
310351
image = await asyncio.to_thread(get_file_storage().open, file.id)

‎backend/druks/core/exceptions.py‎

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
1+
from druks.exceptions import DruksError
2+
3+
4+
class TranscriptionError(DruksError):
5+
"""Druks got no transcript for a voice note."""

‎backend/druks/core/services.py‎

Lines changed: 48 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -6,17 +6,23 @@
66
from githubkit import GitHub
77
from pydantic import BaseModel, Field, SecretStr
88
from slack_sdk.errors import SlackApiError
9+
from sqlalchemy.ext.asyncio import AsyncSession
910

1011
from druks.core.apis.github import GITHUB_AUTHORITY, GitHubClient
1112
from druks.core.apis.linear import LINEAR_GRAPHQL_URL
1213
from druks.core.apis.slack import SLACK_AUTHORITY, SLACK_BOT_SCOPES, SlackClient
14+
from druks.core.exceptions import TranscriptionError
15+
from druks.secrets.datastructures import Audience
1316
from druks.secrets.enums import SecretKind
17+
from druks.secrets.models import VaultSecret
1418
from druks.services import Service, ServiceConnectError
19+
from druks.services.exceptions import ServiceNotConnectedError
1520
from druks.settings import load_settings
1621

1722
logger = logging.getLogger(__name__)
1823

1924
_VERIFY_TIMEOUT = 10.0
25+
_TRANSCRIBE_TIMEOUT = 120.0
2026

2127

2228
class Github(Service):
@@ -265,3 +271,45 @@ def get_manifest(cls) -> dict[str, Any]:
265271
"token_rotation_enabled": False,
266272
},
267273
}
274+
275+
276+
class SpeechToText(Service):
277+
"""The server Druks sends voice notes to: any server that speaks the OpenAI audio
278+
API, such as OpenAI, Groq, or a local one."""
279+
280+
description = (
281+
"The service Druks sends voice notes to. Any server that speaks the OpenAI audio API."
282+
)
283+
required = False
284+
285+
class Settings(BaseModel):
286+
url: str = Field(
287+
title="Address", description="Base URL, for example https://api.openai.com/v1."
288+
)
289+
key: SecretStr = Field(title="Key")
290+
model: str = Field(title="Model", description="For example whisper-1.")
291+
292+
@classmethod
293+
async def transcribe(
294+
cls, session: AsyncSession, *, name: str, content_type: str, content: bytes
295+
) -> str:
296+
"""The words in an audio file. An empty answer is a failure."""
297+
card = await VaultSecret.lookup(session, cls.secret_kind, Audience.service(cls.slug))
298+
if not card:
299+
raise ServiceNotConnectedError(cls.slug)
300+
try:
301+
async with httpx.AsyncClient(timeout=_TRANSCRIBE_TIMEOUT) as client:
302+
response = await client.post(
303+
f"{card.identity['url'].rstrip('/')}/audio/transcriptions",
304+
# A key pasted with a space would echo through the transport's error.
305+
headers={"Authorization": f"Bearer {card.secrets['key'].strip()}"},
306+
data={"model": card.identity["model"]},
307+
files={"file": (name, content, content_type)},
308+
)
309+
response.raise_for_status()
310+
text = response.json()["text"].strip()
311+
except Exception as error: # noqa: BLE001 — any transport or shape failure is a failed note
312+
raise TranscriptionError(f"{cls.title} gave no transcript: {error}") from error
313+
if not text:
314+
raise TranscriptionError(f"{cls.title} gave an empty transcript.")
315+
return text
Lines changed: 27 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,27 @@
1+
"""Keep the words of a chat message's voice note.
2+
3+
Revision ID: e613b1a81427
4+
Revises: 2f9d7f55982a
5+
Create Date: 2026-09-26
6+
"""
7+
8+
from collections.abc import Sequence
9+
10+
import sqlalchemy as sa
11+
from alembic import op
12+
13+
revision: str = "e613b1a81427"
14+
down_revision: str | Sequence[str] | None = "2f9d7f55982a"
15+
branch_labels: str | Sequence[str] | None = None
16+
depends_on: str | Sequence[str] | None = None
17+
18+
19+
def upgrade() -> None:
20+
op.add_column(
21+
"chat_messages",
22+
sa.Column("transcript", sa.String(), nullable=False, server_default=""),
23+
)
24+
25+
26+
def downgrade() -> None:
27+
op.drop_column("chat_messages", "transcript")

‎backend/tests/test_whatsapp.py‎

Lines changed: 26 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -29,7 +29,12 @@
2929
from druks.chat.channels.whatsapp.constants import WAHA_AUDIENCE
3030
from druks.chat.channels.whatsapp.services import Waha
3131
from druks.chat.channels.whatsapp.webhooks import WahaEvents
32-
from druks.chat.constants import CONVERSATION_HEADER, INTERNAL_MESSAGES_PROMPT
32+
from druks.chat.constants import (
33+
CONVERSATION_HEADER,
34+
INTERNAL_MESSAGES_PROMPT,
35+
TRANSCRIPTION_FAILED_MESSAGE,
36+
VOICE_NOTE_MARKER,
37+
)
3338
from druks.chat.enums import (
3439
BotAccess,
3540
ConversationSource,
@@ -38,6 +43,7 @@
3843
PauseSignal,
3944
)
4045
from druks.chat.models import Conversation
46+
from druks.core.exceptions import TranscriptionError
4147
from druks.harnesses.claude import ClaudeHarness
4248
from druks.mcp.enums import Toolkit
4349
from druks.mcp.server import _is_visible, _validate_agent_tools
@@ -348,13 +354,19 @@ async def test_one_turn_answers_every_pending_message_and_knows_its_own_reply(
348354
photo["payload"].update(hasMedia=True, media=waha_media("photo.png", "image/png"))
349355
form = message_event(ANA, "Here is the form", key="M4")
350356
form["payload"].update(hasMedia=True, media=waha_media("form.pdf", "application/pdf"))
351-
note = message_event(ANA, "", key="M5")
357+
note = message_event(ANA, "Call me", key="M5")
352358
note["payload"].update(hasMedia=True, media=waha_media("note.oga", "audio/ogg"))
359+
failed_note = message_event(ANA, "", key="M6")
360+
failed_note["payload"].update(hasMedia=True, media=waha_media("again.oga", "audio/ogg"))
353361
await receive(connection, message_event(ANA, "Hello", key="M1"))
354362
await receive(connection, photo)
355363
await receive(connection, message_event(ANA, "And the second one?", key="M3"))
356364
await receive(connection, form)
357365
await receive(connection, note)
366+
await receive(connection, failed_note)
367+
await receive(connection, message_event(ANA, "Anyone there?", key="M7"))
368+
transcribe = AsyncMock(side_effect=["at six", TranscriptionError("The provider is down.")])
369+
monkeypatch.setattr(service.SpeechToText, "transcribe", transcribe)
358370
[conversation] = await Conversation.list_for_connection(druks_db, connection.id)
359371
config = SimpleNamespace(
360372
harness_class=ClaudeHarness,
@@ -418,14 +430,24 @@ async def copy_arrives_first() -> None:
418430
"name": "form.pdf",
419431
"mimeType": "application/pdf",
420432
},
433+
{"type": "text", "text": f"Call me\n{VOICE_NOTE_MARKER}\nat six"},
434+
{"type": "text", "text": "Anyone there?"},
435+
{"type": "text", "text": TRANSCRIPTION_FAILED_MESSAGE},
436+
]
437+
assert (prompt["timeout"], prompt["messageId"]) == (60, asked[-1].id)
438+
assert [(message.body, message.transcript) for message in asked[4:]] == [
439+
("Call me", "at six"),
440+
("", ""),
441+
("Anyone there?", ""),
442+
(TRANSCRIPTION_FAILED_MESSAGE, ""),
421443
]
422-
assert prompt["timeout"] == 60
444+
assert asked[-1].is_internal
423445
[upload] = host.upload_file.await_args_list
424446
assert (upload.kwargs["local"].is_file(), upload.kwargs["remote"]) == (True, copy)
425447
[start] = [values for method, values in requests if method == "start"]
426448
assert start["headers"] == [{"name": CONVERSATION_HEADER, "value": conversation.id}]
427449
assert start["meta"]["claudeCode"]["options"]["systemPrompt"] == "Be kind."
428-
assert [message.state for message in asked] == [MessageState.REPLIED] * 5
450+
assert [message.state for message in asked] == [MessageState.REPLIED] * 8
429451
assert reply.source_id == "REPLY1"
430452
sends = [body for method, path, body in waha.calls if path == "/api/sendText"]
431453
assert sends == [

‎docs/chat.md‎

Lines changed: 13 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -235,6 +235,15 @@ bulk. Before it sends a reply, Druks takes a new message ID from WAHA and
235235
records it. WAHA's copy of the sent message then carries a known ID, even when
236236
the copy arrives before the send returns.
237237

238+
A voice note becomes text before its turn. Druks sends the audio to the
239+
[Speech To Text card](configuration.md#speech-to-text) and saves the words on
240+
the message. The agent reads them under what the person typed, marked as a
241+
voice note, and answers in text. Nothing goes to the person before the reply.
242+
When Druks cannot transcribe a note, when no card is connected, or when the
243+
note is over 25 MiB, the agent gets an internal message instead. It then tells
244+
the person in one line to write instead. The web page shows the typed text, the
245+
words, and a link to the audio.
246+
238247
Druks also adds **internal messages** to a conversation. Each one comes from a
239248
fixed template. An internal message starts a turn like any message, and it
240249
never goes to WhatsApp. Druks talks to agents, and agents talk to people.
@@ -320,9 +329,10 @@ A file you send to the bot, in a direct message or in a thread you joined,
320329
becomes a Druks file on its message. A message with only a file starts a turn
321330
like any other. An image reaches the agent with its message. Any other file
322331
except audio reaches the agent as a link to a copy in the sandbox. The agent
323-
opens the copy with its tools. An image over 25 MiB goes as a link too. Each
324-
further file in one Slack message gets a message of its own. The agent sends no
325-
files back.
332+
opens the copy with its tools. An image over 25 MiB goes as a link too. An audio
333+
clip becomes text the way a WhatsApp voice note does: see
334+
[WhatsApp](#whatsapp). Each further file in one Slack message gets a message of
335+
its own. The agent sends no files back.
326336

327337
### Rooms
328338

‎docs/configuration.md‎

Lines changed: 11 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -484,6 +484,17 @@ event older than five minutes. With rotation off, the pasted bot token and each
484484
person's Slack token live until someone revokes them. A person connects their
485485
own Slack account through the same app: see [Chat](chat.md#slack).
486486

487+
## Speech to text
488+
489+
**Speech To Text** is the service that Druks sends voice notes to: any server
490+
that speaks the OpenAI audio API, such as OpenAI, Groq, or a local server.
491+
Connect it from **Settings → Connections → Services** with the server's base
492+
URL, for example `https://api.openai.com/v1`, a key, and a model, for example
493+
`whisper-1`. Druks checks none of the values when you save the card, so a wrong
494+
key shows up on the first voice note. Druks sends a note of at most 25 MiB and
495+
refuses a bigger one without a call. Without the card, the agent tells the
496+
person to write instead. See [Chat](chat.md#whatsapp) for what the agent gets.
497+
487498
## Harnesses
488499

489500
Druks registers two subscription providers, `anthropic` and `openai`. Each

0 commit comments

Comments
 (0)