feat: add local speech platform (#1671)

Fixes #1646
This commit is contained in:
Palash Debnath
2026-08-27 19:51:07 +05:30
committed by GitHub
parent 078bad8e0f
commit b92d35ac5d
21 changed files with 1611 additions and 19 deletions
+2
View File
@@ -10,6 +10,7 @@ the frozen-backend fallback mirror it for their toolchains.
**Highlights** **Highlights**
- VoiceStudio now acts as a local speech platform: other apps can trigger its native dictation or connect through versioned HTTP, WebSocket, JSON-RPC, CLI, and MCP transports (#1646)
- Dictation now stays bound to the app where it started and recovers locally from silent recognizer output (#1175) - Dictation now stays bound to the app where it started and recovers locally from silent recognizer output (#1175)
- The backend now answers within a second of launch and narrates its startup step by step (#1550) - The backend now answers within a second of launch and narrates its startup step by step (#1550)
- Reporting a bug from an outdated build now offers the latest release first (#1547) - Reporting a bug from an outdated build now offers the latest release first (#1547)
@@ -22,6 +23,7 @@ the frozen-backend fallback mirror it for their toolchains.
- The backend binds its port immediately and reports startup progress live — `/health` answers 503-with-step and a new `/startup/progress` endpoint lists every step while PyTorch, API routes, and database migrations load in the background, so "starting at step X" is never mistakable for "dead"; the desktop splash narrates each step (#1550) - The backend binds its port immediately and reports startup progress live — `/health` answers 503-with-step and a new `/startup/progress` endpoint lists every step while PyTorch, API routes, and database migrations load in the background, so "starting at step X" is never mistakable for "dead"; the desktop splash narrates each step (#1550)
### Added ### Added
- A bundled Rust loopback sidecar exposes dictation start/stop/toggle, focused-output sessions, discovery, and JSON-RPC; the backend adds versioned streaming events and a dependency-free CLI bridge for Herdr, coding agents, editors, desktop apps, and TUIs (#1646)
- Headless NVIDIA and ROCm machines can now join as worker-only Docker Compose services with no published UI and durable protocol-v2 enrollment; update both machines together before reconnecting (#1638) — thanks @jkrogers9862! - Headless NVIDIA and ROCm machines can now join as worker-only Docker Compose services with no published UI and durable protocol-v2 enrollment; update both machines together before reconnecting (#1638) — thanks @jkrogers9862!
- Linux ARM64 (Asahi Apple Silicon) support for the OmniVoice GGUF engine — a `linux-aarch64` binary built with GGML Vulkan where the toolchain allows it, so Apple GPUs accelerate generation through the open-source Honeykrisp driver instead of falling back to CPU-only (#1641) - Linux ARM64 (Asahi Apple Silicon) support for the OmniVoice GGUF engine — a `linux-aarch64` binary built with GGML Vulkan where the toolchain allows it, so Apple GPUs accelerate generation through the open-source Honeykrisp driver instead of falling back to CPU-only (#1641)
- One-command install on every desktop OS: `curl -fsSL https://voicestudio.sh/install | sh` (macOS/Linux/WSL) or `irm https://voicestudio.sh/install | iex` (Windows) — the URL serves the right script per platform, and Windows gains a source installer (`scripts/install.ps1`) with a 3-OS CI smoke (#1626) - One-command install on every desktop OS: `curl -fsSL https://voicestudio.sh/install | sh` (macOS/Linux/WSL) or `irm https://voicestudio.sh/install | iex` (Windows) — the URL serves the right script per platform, and Windows gains a source installer (`scripts/install.ps1`) with a 3-OS CI smoke (#1626)
+10 -3
View File
@@ -254,7 +254,7 @@ FastAPI backend
<a id="api"></a> <a id="api"></a>
## OpenAI-compatible API ## Local speech platform and OpenAI-compatible API
Point an OpenAI-compatible audio client at the local backend: Point an OpenAI-compatible audio client at the local backend:
@@ -267,6 +267,8 @@ Point an OpenAI-compatible audio client at the local backend:
|---|---| |---|---|
| `POST /v1/audio/speech` | TTS to `mp3`, `opus`, `aac`, `flac`, `wav`, or `pcm`; select a profile with `voice` and an engine with `model` | | `POST /v1/audio/speech` | TTS to `mp3`, `opus`, `aac`, `flac`, `wav`, or `pcm`; select a profile with `voice` and an engine with `model` |
| `POST /v1/audio/transcriptions` | STT to `json`, `text`, `verbose_json`, `srt`, or `vtt` | | `POST /v1/audio/transcriptions` | STT to `json`, `text`, `verbose_json`, `srt`, or `vtt` |
| `WS /v1/audio/transcriptions/stream` | Live PCM/WebM transcription with partial, utterance, and session-final events |
| `GET /.well-known/voicestudio-speech` | Discover HTTP, WebSocket, MCP, and native dictation-control transports |
| `GET /v1/audio/voices` | List local voice profiles and engines | | `GET /v1/audio/voices` | List local voice profiles and engines |
```python ```python
@@ -283,7 +285,12 @@ with client.audio.speech.with_streaming_response.create(
response.stream_to_file("speech.wav") response.stream_to_file("speech.wav")
``` ```
The full API reference is in **Settings → OpenAPI Reference**. For LAN, Tailscale, or proxy access, read [API authentication](docs/api-auth.md) before exposing the backend. The bundled Rust control sidecar also lets Herdr, coding agents, VS Code,
desktop apps, and TUIs trigger the existing system-wide dictation flow or reuse
its safe native insertion. See the [speech platform guide](docs/speech-platform.md).
The full API reference is in **Settings → OpenAPI Reference**. For LAN,
Tailscale, or proxy access, read [API authentication](docs/api-auth.md) before
exposing the backend.
### Agent skills ### Agent skills
@@ -312,7 +319,7 @@ The [notebook](notebooks/OmniVoice_Studio_Colab.ipynb) runs the app and web UI o
| Fix setup | [Troubleshooting](docs/install/troubleshooting.md) · [model downloads](docs/downloading-models.md) · [Hugging Face token](docs/setup/huggingface-token.md) | | Fix setup | [Troubleshooting](docs/install/troubleshooting.md) · [model downloads](docs/downloading-models.md) · [Hugging Face token](docs/setup/huggingface-token.md) |
| Choose an engine | [Engine guides](docs/engines/README.md) · [benchmarks](docs/benchmarks.md) · [expressive speech](docs/expressive-speech.md) | | Choose an engine | [Engine guides](docs/engines/README.md) · [benchmarks](docs/benchmarks.md) · [expressive speech](docs/expressive-speech.md) |
| Tune hardware | [Performance](docs/performance.md) · [remote workers](docs/remote-workers.md) | | Tune hardware | [Performance](docs/performance.md) · [remote workers](docs/remote-workers.md) |
| Build integrations | [API auth](docs/api-auth.md) · [MCP](docs/mcp.md) · [examples](examples/README.md) | | Build integrations | [Speech platform](docs/speech-platform.md) · [API auth](docs/api-auth.md) · [MCP](docs/mcp.md) · [examples](examples/README.md) |
| Build VoiceStudio | [Contributing](.github/CONTRIBUTING.md) · [engine acceptance](docs/engine-acceptance.md) | | Build VoiceStudio | [Contributing](.github/CONTRIBUTING.md) · [engine acceptance](docs/engine-acceptance.md) |
| Track changes | [Changelog](CHANGELOG.md) · [roadmap](docs/ROADMAP.md) · [latest release](https://github.com/debpalash/VoiceStudio/releases/latest) | | Track changes | [Changelog](CHANGELOG.md) · [roadmap](docs/ROADMAP.md) · [latest release](https://github.com/debpalash/VoiceStudio/releases/latest) |
| Remove everything | [Uninstall guide](docs/install/uninstall.md) | | Remove everything | [Uninstall guide](docs/install/uninstall.md) |
+66 -2
View File
@@ -38,11 +38,14 @@ Protocol:
from __future__ import annotations from __future__ import annotations
import asyncio import asyncio
import json
import logging import logging
import math import math
import os import os
import tempfile import tempfile
import time import time
import uuid
from typing import Any
from fastapi import APIRouter, WebSocket, WebSocketDisconnect from fastapi import APIRouter, WebSocket, WebSocketDisconnect
@@ -52,6 +55,9 @@ from services.text_polish import polish_text
router = APIRouter() router = APIRouter()
logger = logging.getLogger("omnivoice.capture_ws") logger = logging.getLogger("omnivoice.capture_ws")
SPEECH_PROTOCOL = "voicestudio.speech.v1"
PLATFORM_STREAM_PATH = "/v1/audio/transcriptions/stream"
# How often (seconds) to run transcription on the accumulated buffer. # How often (seconds) to run transcription on the accumulated buffer.
# Shorter = more responsive but more GPU load. # Shorter = more responsive but more GPU load.
PARTIAL_INTERVAL_S = float(os.environ.get("OMNIVOICE_STREAM_INTERVAL", "2.0")) PARTIAL_INTERVAL_S = float(os.environ.get("OMNIVOICE_STREAM_INTERVAL", "2.0"))
@@ -83,6 +89,39 @@ _AEC_FAR = 0x01 # playback reference frame (feed the echo model only)
SR_MIN, SR_MAX = 8000, 96000 SR_MIN, SR_MAX = 8000, 96000
def _is_end_control(text: str | None) -> bool:
"""Accept the versioned JSON control frame and the legacy ``EOF`` frame."""
if text == "EOF":
return True
if not text:
return False
try:
message = json.loads(text)
except (TypeError, json.JSONDecodeError):
return False
return isinstance(message, dict) and message.get("type") == "input_audio.end"
class _PlatformWebSocket:
"""Add v1 session metadata without changing the legacy WebSocket contract."""
def __init__(self, websocket: WebSocket):
self._websocket = websocket
self.session_id = uuid.uuid4().hex
def __getattr__(self, name: str) -> Any:
return getattr(self._websocket, name)
async def send_json(self, data: Any, mode: str = "text") -> None:
if isinstance(data, dict):
data = dict(data)
data.setdefault("protocol", SPEECH_PROTOCOL)
data.setdefault("session_id", self.session_id)
if data.get("type") == "final":
data.setdefault("final_kind", "summary")
await self._websocket.send_json(data, mode=mode)
def _bounded_sample_rate(query_params) -> int: def _bounded_sample_rate(query_params) -> int:
try: try:
sample_rate = int(query_params.get("sr", "16000")) sample_rate = int(query_params.get("sr", "16000"))
@@ -194,9 +233,24 @@ def _select_sherpa_spec(websocket: WebSocket):
return _usable_spec(mid) if mid else None return _usable_spec(mid) if mid else None
@router.websocket(PLATFORM_STREAM_PATH)
@router.websocket("/ws/transcribe") @router.websocket("/ws/transcribe")
async def ws_transcribe(websocket: WebSocket): async def ws_transcribe(websocket: WebSocket):
"""Stream audio in, get partial + final transcription out.""" """Stream audio in, get partial + final transcription out."""
is_platform_stream = websocket.url.path == PLATFORM_STREAM_PATH
if is_platform_stream:
websocket = _PlatformWebSocket(websocket)
# A browser can reach localhost regardless of the page's own origin.
# Reject ambient cross-site WebSocket handshakes before the loopback-host
# shortcut or accept(), while keeping native clients (no Origin header)
# and configured/same-origin browser UIs working (#1646 review).
origin = websocket.headers.get("origin")
if origin:
from core.csrf import origin_allowed
if not origin_allowed(websocket):
await websocket.close(code=1008, reason="browser origin not allowed")
return
# Loopback origin guard — refuse anything not from 127.0.0.1, ::1, or # Loopback origin guard — refuse anything not from 127.0.0.1, ::1, or
# localhost. Privileged HTTP routers use Depends(require_admin) at router # localhost. Privileged HTTP routers use Depends(require_admin) at router
# level; WebSocket dependency injection differs across FastAPI versions, so we # level; WebSocket dependency injection differs across FastAPI versions, so we
@@ -211,6 +265,16 @@ async def ws_transcribe(websocket: WebSocket):
return return
await websocket.accept() await websocket.accept()
if is_platform_stream:
await websocket.send_json({
"type": "session.started",
"input_format": (
"audio/pcm;encoding=s16le;channels=1"
if _requested_pcm_sample_rate(websocket.query_params) is not None
else "audio/webm;codecs=opus"
),
"sample_rate": _bounded_sample_rate(websocket.query_params),
})
# Live-dictation engine selection. When a sherpa-onnx model is selected # Live-dictation engine selection. When a sherpa-onnx model is selected
# (via ?model= or the dictation.model_id pref) AND sherpa is installed, # (via ?model= or the dictation.model_id pref) AND sherpa is installed,
@@ -333,7 +397,7 @@ async def ws_transcribe(websocket: WebSocket):
total_bytes += len(data) total_bytes += len(data)
last_audio_time = time.monotonic() last_audio_time = time.monotonic()
continue continue
if msg.get("text") == "EOF": if _is_end_control(msg.get("text")):
# Client signals end-of-audio but stays connected for `final`. # Client signals end-of-audio but stays connected for `final`.
running = False running = False
break break
@@ -656,7 +720,7 @@ async def _recv_pcm_frame(websocket: WebSocket, aec):
return "skip", b"" return "skip", b""
return "near", aec.process_near_end(payload) return "near", aec.process_near_end(payload)
return "near", data return "near", data
if msg.get("text") == "EOF": if _is_end_control(msg.get("text")):
return "eof", b"" return "eof", b""
return "skip", b"" return "skip", b""
+160
View File
@@ -0,0 +1,160 @@
"""Discovery contract for VoiceStudio's local speech platform.
Interfaces should discover this document instead of hard-coding whichever
dictation route the desktop happens to use. Endpoint URLs are relative so the
same response works on loopback, a tailnet GPU host, and a reverse proxy.
"""
from __future__ import annotations
import os
from typing import Literal
from fastapi import APIRouter
from pydantic import BaseModel, Field
from core.version import APP_VERSION
router = APIRouter(tags=["Speech Platform"])
SPEECH_PROTOCOL = "voicestudio.speech.v1"
STREAM_PATH = "/v1/audio/transcriptions/stream"
class EndpointCapability(BaseModel):
path: str
transport: Literal["http", "websocket", "mcp-streamable-http", "mcp-stdio"]
method: str | None = None
protocol: str | None = None
class StreamInputCapability(BaseModel):
framing: Literal["binary"] = "binary"
formats: list[str]
default_format: str
sample_rate_query: str = "sr"
end_control: dict[str, str]
class StreamOutputCapability(BaseModel):
framing: Literal["json"] = "json"
events: list[str]
final_kinds: list[str]
class SpeechFeatureCapabilities(BaseModel):
batch_transcription: bool = True
streaming_transcription: bool = True
partial_transcripts: bool = True
utterance_finals: bool = True
session_summary: bool = True
word_timestamps: bool = True
local_refinement: bool = True
acoustic_echo_cancellation: bool = True
native_dictation_control: bool = False
class SpeechAuthCapabilities(BaseModel):
loopback: Literal["none"] = "none"
remote: Literal["bearer"] = "bearer"
header: str = "Authorization: Bearer <OMNIVOICE_API_KEY>"
browser_session_endpoint: str = "/api/auth/session"
websocket_ticket_endpoint: str = "/api/auth/ws-ticket"
websocket_ticket_query_parameter: Literal["ws_ticket"] = "ws_ticket"
class SpeechCapabilities(BaseModel):
schema_: Literal["voicestudio.speech-capabilities"] = Field(
default="voicestudio.speech-capabilities",
serialization_alias="schema",
)
protocol: Literal["voicestudio.speech.v1"] = SPEECH_PROTOCOL
protocol_version: Literal["1.0"] = "1.0"
service: str = "VoiceStudio"
service_version: str = APP_VERSION
local_first: bool = True
endpoints: dict[str, EndpointCapability]
stream_input: StreamInputCapability
stream_output: StreamOutputCapability
features: SpeechFeatureCapabilities
authentication: SpeechAuthCapabilities
def speech_capabilities() -> SpeechCapabilities:
"""Return the stable, side-effect-free integration contract."""
endpoints = {
"capabilities": EndpointCapability(
path="/.well-known/voicestudio-speech",
transport="http",
method="GET",
),
"batch_transcription": EndpointCapability(
path="/v1/audio/transcriptions",
transport="http",
method="POST",
protocol="openai.audio.transcriptions",
),
"streaming_transcription": EndpointCapability(
path=STREAM_PATH,
transport="websocket",
protocol=SPEECH_PROTOCOL,
),
"mcp": EndpointCapability(
path="/mcp",
transport="mcp-streamable-http",
method="POST",
protocol="mcp",
),
"mcp_stdio": EndpointCapability(
path="python -m backend.mcp_shim",
transport="mcp-stdio",
protocol="mcp",
),
}
native_control = False
try:
control_port = int(os.environ.get("VOICESTUDIO_SPEECH_CONTROL_PORT", ""))
except (TypeError, ValueError):
control_port = 0
if 0 < control_port <= 65535:
native_control = True
endpoints["native_dictation_control"] = EndpointCapability(
path=f"http://127.0.0.1:{control_port}/v1/capabilities",
transport="http",
method="GET",
protocol=SPEECH_PROTOCOL,
)
return SpeechCapabilities(
endpoints=endpoints,
stream_input=StreamInputCapability(
formats=[
"audio/pcm;encoding=s16le;channels=1",
"audio/webm;codecs=opus",
],
default_format="audio/webm;codecs=opus",
end_control={"type": "input_audio.end"},
),
stream_output=StreamOutputCapability(
events=["session.started", "status", "partial", "final", "error"],
final_kinds=["utterance", "summary"],
),
features=SpeechFeatureCapabilities(
native_dictation_control=native_control,
),
authentication=SpeechAuthCapabilities(),
)
@router.get(
"/.well-known/voicestudio-speech",
response_model=SpeechCapabilities,
response_model_by_alias=True,
)
@router.get(
"/v1/audio/capabilities",
response_model=SpeechCapabilities,
response_model_by_alias=True,
)
async def get_speech_capabilities() -> SpeechCapabilities:
"""Advertise batch, streaming, and agent-facing speech transports."""
return speech_capabilities()
+2 -1
View File
@@ -666,6 +666,7 @@ def _phase_a_build_inner() -> None:
events, events,
capture, capture,
capture_ws, capture_ws,
speech_platform,
dictation, dictation,
openai_compat, openai_compat,
tts_stream, tts_stream,
@@ -685,7 +686,7 @@ def _phase_a_build_inner() -> None:
system, profiles, exports, generation, dub_core, dub_generate, system, profiles, exports, generation, dub_core, dub_generate,
dub_export, dub_translate, projects, glossary, engines, tools, dub_export, dub_translate, projects, glossary, engines, tools,
stories, setup, gallery, archetypes, describe_voice, community, stories, setup, gallery, archetypes, describe_voice, community,
batch, watermark, events, capture, capture_ws, dictation, batch, watermark, events, capture, capture_ws, speech_platform, dictation,
openai_compat, tts_stream, marketplace, personas, sonitranslate, openai_compat, tts_stream, marketplace, personas, sonitranslate,
audiobook, longform_jobs, pronunciation, settings_router, audiobook, longform_jobs, pronunciation, settings_router,
media_tools_router, auth_router, _mcp_bindings_router, workers_router, media_tools_router, auth_router, _mcp_bindings_router, workers_router,
+3 -1
View File
@@ -33,7 +33,9 @@ WS_TICKET_PREFIX = "ovs_ws_ticket_"
_TOKEN_BYTES = 32 _TOKEN_BYTES = 32
_ENCODED_TOKEN_LENGTH = 43 _ENCODED_TOKEN_LENGTH = 43
_TOKEN_BODY_RE = re.compile(rf"^[A-Za-z0-9_-]{{{_ENCODED_TOKEN_LENGTH}}}$") _TOKEN_BODY_RE = re.compile(rf"^[A-Za-z0-9_-]{{{_ENCODED_TOKEN_LENGTH}}}$")
_ALLOWED_WS_PATHS = frozenset({"/ws/events", "/ws/transcribe"}) _ALLOWED_WS_PATHS = frozenset(
{"/ws/events", "/ws/transcribe", "/v1/audio/transcriptions/stream"}
)
_ADMIN_CAPABILITIES = frozenset({"consume", "admin"}) _ADMIN_CAPABILITIES = frozenset({"consume", "admin"})
_KEY_GENERATION_INFO = b"omnivoice-admin-key-generation-v1" _KEY_GENERATION_INFO = b"omnivoice-admin-key-generation-v1"
+1
View File
@@ -0,0 +1 @@
"""Dependency-free client for VoiceStudio's local speech platform."""
+278
View File
@@ -0,0 +1,278 @@
"""CLI/module bridge for terminals, editor extensions, and agent hooks.
The desktop app must be running for native dictation control. Batch
transcription can also target a standalone or remote VoiceStudio backend.
"""
from __future__ import annotations
import argparse
import ipaddress
import json
import mimetypes
import os
from pathlib import Path
import secrets
import sys
from typing import Any
from urllib import error, request
from urllib.parse import urlsplit
DEFAULT_CONTROL_URL = "http://127.0.0.1:3902"
DEFAULT_ENGINE_URL = "http://127.0.0.1:3900"
class SpeechClientError(RuntimeError):
pass
class _RejectCredentialRedirect(request.HTTPRedirectHandler):
def redirect_request(self, req, fp, code, msg, headers, newurl): # noqa: ARG002
raise SpeechClientError("VoiceStudio refused a credentialed redirect")
def _join_url(base_url: str, path: str) -> str:
return f"{base_url.rstrip('/')}/{path.lstrip('/')}"
def _decode_error(exc: error.HTTPError) -> str:
try:
body = exc.read().decode("utf-8", errors="replace")
except Exception:
body = ""
try:
detail = json.loads(body)
except (TypeError, json.JSONDecodeError):
detail = body.strip()
return f"HTTP {exc.code}: {detail or exc.reason}"
def _is_loopback_host(host: str | None) -> bool:
if not host:
return False
if host.lower() == "localhost":
return True
try:
return ipaddress.ip_address(host).is_loopback
except ValueError:
return False
def _open(req: request.Request, timeout: float = 300.0) -> tuple[bytes, str]:
target = urlsplit(req.full_url)
scheme = target.scheme.lower()
if scheme not in {"http", "https"}:
raise SpeechClientError("VoiceStudio URLs must use http:// or https://")
credentialed = bool(req.get_header("Authorization"))
if credentialed and scheme != "https" and not _is_loopback_host(target.hostname):
raise SpeechClientError("Remote VoiceStudio credentials require https://")
try:
opener = (
request.build_opener(_RejectCredentialRedirect())
if credentialed
else request.build_opener()
)
with opener.open(req, timeout=timeout) as response: # noqa: S310
return response.read(), response.headers.get("Content-Type", "")
except error.HTTPError as exc:
raise SpeechClientError(_decode_error(exc)) from exc
except error.URLError as exc:
raise SpeechClientError(f"VoiceStudio is unavailable: {exc.reason}") from exc
def _json_request(method: str, url: str, payload: Any | None = None) -> Any:
data = None if payload is None else json.dumps(payload).encode("utf-8")
headers = {"Accept": "application/json"}
if data is not None:
headers["Content-Type"] = "application/json"
body, _ = _open(request.Request(url, data=data, headers=headers, method=method), timeout=10.0)
try:
return json.loads(body)
except json.JSONDecodeError as exc:
raise SpeechClientError("VoiceStudio returned invalid JSON") from exc
def _encode_multipart(
*,
filename: str,
audio: bytes,
fields: dict[str, str],
boundary: str | None = None,
) -> tuple[bytes, str]:
boundary = boundary or f"voicestudio-{secrets.token_hex(16)}"
marker = boundary.encode("ascii")
parts: list[bytes] = []
for name, value in fields.items():
parts.extend(
[
b"--" + marker + b"\r\n",
f'Content-Disposition: form-data; name="{name}"\r\n\r\n'.encode(),
value.encode("utf-8"),
b"\r\n",
]
)
safe_filename = Path(filename).name.replace('"', "") or "audio.wav"
content_type = mimetypes.guess_type(safe_filename)[0] or "application/octet-stream"
if Path(safe_filename).suffix.lower() in {".wav", ".wave"}:
content_type = "audio/wav"
parts.extend(
[
b"--" + marker + b"\r\n",
(
'Content-Disposition: form-data; name="file"; '
f'filename="{safe_filename}"\r\n'
).encode(),
f"Content-Type: {content_type}\r\n\r\n".encode(),
audio,
b"\r\n--" + marker + b"--\r\n",
]
)
return b"".join(parts), f"multipart/form-data; boundary={boundary}"
def _control(args: argparse.Namespace, action: str) -> int:
method = "GET" if action in {"status", "capabilities"} else "POST"
path = {
"status": "/v1/status",
"capabilities": "/v1/capabilities",
"start": "/v1/dictation/start",
"stop": "/v1/dictation/stop",
"toggle": "/v1/dictation/toggle",
}[action]
result = _json_request(method, _join_url(args.control_url, path))
print(json.dumps(result, ensure_ascii=False, indent=2))
return 0
def _read_audio(path: str, stdin_filename: str) -> tuple[bytes, str]:
if path == "-":
return sys.stdin.buffer.read(), stdin_filename
audio_path = Path(path)
try:
return audio_path.read_bytes(), audio_path.name
except OSError as exc:
display_name = path.replace("\\", "/").rsplit("/", 1)[-1] or "audio input"
reason = exc.strerror or type(exc).__name__
raise SpeechClientError(f"could not read '{display_name}': {reason}") from exc
def _response_text(body: bytes, content_type: str) -> str:
decoded = body.decode("utf-8", errors="replace")
if "json" not in content_type.lower():
return decoded
try:
payload = json.loads(decoded)
except json.JSONDecodeError:
return decoded
if isinstance(payload, dict) and isinstance(payload.get("text"), str):
return payload["text"]
return decoded
def _transcribe(args: argparse.Namespace) -> int:
audio, filename = _read_audio(args.audio, args.stdin_filename)
fields = {
"model": args.model,
"response_format": args.response_format,
}
if args.language:
fields["language"] = args.language
body, content_type = _encode_multipart(filename=filename, audio=audio, fields=fields)
headers = {"Content-Type": content_type, "Accept": "application/json, text/plain"}
api_key = os.environ.get("OMNIVOICE_API_KEY", "").strip()
if api_key:
headers["Authorization"] = f"Bearer {api_key}"
output_session_id = None
if args.insert:
session = _json_request(
"POST", _join_url(args.control_url, "/v1/output/sessions")
)
output_session_id = session["session_id"]
session_needs_cleanup = output_session_id is not None
try:
response_body, response_type = _open(
request.Request(
_join_url(args.engine_url, "/v1/audio/transcriptions"),
data=body,
headers=headers,
method="POST",
)
)
if output_session_id is not None:
_json_request(
"POST",
_join_url(
args.control_url,
f"/v1/output/sessions/{output_session_id}/insert",
),
{"text": _response_text(response_body, response_type)},
)
session_needs_cleanup = False
finally:
if session_needs_cleanup:
try:
_json_request(
"DELETE",
_join_url(args.control_url, f"/v1/output/sessions/{output_session_id}"),
)
except Exception: # noqa: BLE001
# Best-effort cleanup must not replace the original failure or
# KeyboardInterrupt that brought control into this finally.
pass
sys.stdout.buffer.write(response_body)
if response_body and not response_body.endswith(b"\n"):
sys.stdout.buffer.write(b"\n")
return 0
def _parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(
prog="voicestudio-speech",
description="Control and consume VoiceStudio's local speech platform.",
)
parser.add_argument(
"--control-url",
default=os.environ.get("VOICESTUDIO_SPEECH_URL", DEFAULT_CONTROL_URL),
)
parser.add_argument(
"--engine-url",
default=os.environ.get("VOICESTUDIO_URL", DEFAULT_ENGINE_URL),
)
subparsers = parser.add_subparsers(dest="command", required=True)
for command in ("status", "capabilities", "start", "stop", "toggle"):
subparsers.add_parser(command)
transcribe = subparsers.add_parser("transcribe")
transcribe.add_argument("audio", help="audio file, or - for stdin")
transcribe.add_argument("--stdin-filename", default="audio.wav")
transcribe.add_argument("--model", default="whisper-1")
transcribe.add_argument("--language")
transcribe.add_argument(
"--format",
dest="response_format",
choices=("json", "text", "verbose_json", "srt", "vtt"),
default="text",
)
transcribe.add_argument(
"--insert",
action="store_true",
help="insert the result into the app focused when this command starts",
)
return parser
def main(argv: list[str] | None = None) -> int:
args = _parser().parse_args(argv)
try:
if args.command == "transcribe":
return _transcribe(args)
return _control(args, args.command)
except (SpeechClientError, KeyError) as exc:
print(f"voicestudio-speech: {exc}", file=sys.stderr)
return 2
if __name__ == "__main__":
raise SystemExit(main())
+16 -7
View File
@@ -1,10 +1,14 @@
# Agentic voice: VoiceStudio as a TTS/STT provider # Agentic voice: VoiceStudio as a TTS/STT provider
VoiceStudio exposes an **OpenAI-compatible API**, so any agent framework that VoiceStudio exposes a **local speech platform**—OpenAI-compatible batch audio,
a versioned transcription WebSocket, native dictation control, and MCP—so any agent framework that
speaks to OpenAI's audio endpoints can use your local VoiceStudio for speech — speaks to OpenAI's audio endpoints can use your local VoiceStudio for speech —
in your own cloned voice, with nothing leaving your machine. You bring the in your own cloned voice, with nothing leaving your machine. You bring the
agent runtime; VoiceStudio is the voice. agent runtime; VoiceStudio is the voice.
For dictating directly into Claude Code, Codex, Pi, Antigravity CLI, Herdr, or
another focused prompt, use the [Rust control sidecar](speech-platform.md).
This is "agentic v1": VoiceStudio is a provider, not the orchestrator. You wire This is "agentic v1": VoiceStudio is a provider, not the orchestrator. You wire
your own agent (a support line, a desk assistant, a Discord persona) and point your own agent (a support line, a desk assistant, a Discord persona) and point
its TTS/STT at VoiceStudio. its TTS/STT at VoiceStudio.
@@ -16,13 +20,17 @@ its TTS/STT at VoiceStudio.
## The endpoints ## The endpoints
VoiceStudio serves these on `http://localhost:3900/v1` (or your VoiceStudio's service root is `http://localhost:3900` (or your
[remote backend URL](remote-gpu.md)): [remote backend URL](remote-gpu.md)). OpenAI-compatible clients use
`http://localhost:3900/v1` as their base URL, while discovery stays at the
service root: `http://localhost:3900/.well-known/voicestudio-speech`.
| OpenAI route | VoiceStudio support | | OpenAI route | VoiceStudio support |
|---|---| |---|---|
| `POST /v1/audio/speech` | TTS. `model` = engine id, `voice` = a voice-profile id (your clone) or preset, `response_format` incl. `pcm` and `wav`, `speed`. Default output is 24 kHz. | | `POST /v1/audio/speech` | TTS. `model` = engine id, `voice` = a voice-profile id (your clone) or preset, `response_format` incl. `pcm` and `wav`, `speed`. Default output is 24 kHz. |
| `POST /v1/audio/transcriptions` | STT (Whisper-family). | | `POST /v1/audio/transcriptions` | STT (Whisper-family). |
| `WS /v1/audio/transcriptions/stream` | Live partial/final STT from PCM or WebM. |
| `GET /.well-known/voicestudio-speech` | Machine-readable transport discovery. |
| `GET /v1/audio/voices` | list available voices (VoiceStudio extension). | | `GET /v1/audio/voices` | list available voices (VoiceStudio extension). |
A contract test (`tests/test_agentic_provider_contract.py`) pins this request A contract test (`tests/test_agentic_provider_contract.py`) pins this request
@@ -74,10 +82,11 @@ single local agent, pipecat is lighter.
## Remote backend ## Remote backend
Running VoiceStudio on a [remote GPU box](remote-gpu.md)? Use that backend's URL Running VoiceStudio on a [remote GPU box](remote-gpu.md)? Append `/v1` to that
as `base_url` and pass its `OMNIVOICE_API_KEY` as the `api_key` — the same backend's service-root URL for the OpenAI client's `base_url`, and pass its
bearer the rest of the app uses. Keep it on your tailnet, not the open `OMNIVOICE_API_KEY` as the `api_key` — the same bearer the rest of the app uses.
internet. Keep the unmodified service root for `/.well-known/voicestudio-speech`
discovery, and keep the backend on your tailnet, not the open internet.
## Use your own voice responsibly ## Use your own voice responsibly
+6
View File
@@ -4,6 +4,12 @@ VoiceStudio dictation records from the system-wide shortcut, transcribes
locally, and—where the desktop permits it—inserts the result into the app where locally, and—where the desktop permits it—inserts the result into the app where
the shortcut was pressed. The pill never needs keyboard focus. the shortcut was pressed. The pill never needs keyboard focus.
The same flow is available to other applications through the bundled Rust
control sidecar. Herdr actions, editor extensions, agent hooks, and scripts can
start or stop VoiceStudio's capture over loopback HTTP/JSON-RPC or stream their
own microphone audio to the versioned WebSocket API. See the
[local speech platform](../speech-platform.md) for the protocol and examples.
## Use it ## Use it
1. Choose an installed dictation model in the Model Catalogue. 1. Choose an installed dictation model in the Model Catalogue.
+191
View File
@@ -0,0 +1,191 @@
# Local speech platform
VoiceStudio is both a desktop dictation app and a headless local speech
service. The desktop remains one app: its bundled Rust control sidecar owns
microphone activation, focused-target capture, clipboard safety, and native
insertion; the Python backend keeps ASR models warm and exposes the audio data
plane.
This split lets an integration choose how much it owns:
```text
Herdr / terminal / desktop app ── start, stop, toggle ──> Rust control :3902
├─ captures target
├─ opens VoiceStudio mic
└─ inserts final text
VS Code / custom GUI / remote mic ── PCM or WebM ───────> WS/HTTP :3900
└─ partial/final text
reserve target / insert final ─> Rust control :3902
Claude Code / Codex / Pi / agents ── MCP HTTP/stdio ───> MCP :3900
```
The Rust sidecar is part of the VoiceStudio process, not a second application.
It starts with the desktop app and binds only to `127.0.0.1`.
## Discover capabilities
Desktop/native discovery:
```bash
curl http://127.0.0.1:3902/.well-known/voicestudio-speech
```
Engine/data-plane discovery:
```bash
curl http://127.0.0.1:3900/.well-known/voicestudio-speech
```
Both return `voicestudio.speech.v1`. The desktop document includes absolute
control, batch, streaming, output-session, and MCP endpoints. The backend
document uses relative URLs so it also works behind Tailscale or a reverse
proxy; it advertises native control only when launched by the desktop app.
## Use VoiceStudio capture from any app
These calls use VoiceStudio's existing microphone, model selection, pill,
refinement, and session-bound insertion. The app under the cursor remains the
destination.
```bash
curl -X POST http://127.0.0.1:3902/v1/dictation/start
curl -X POST http://127.0.0.1:3902/v1/dictation/stop
curl -X POST http://127.0.0.1:3902/v1/dictation/toggle
```
JSON-RPC clients use the same actions:
```json
{"jsonrpc":"2.0","id":1,"method":"dictation.toggle"}
```
Send that object to `POST http://127.0.0.1:3902/rpc`. The installed
VoiceStudio executable also accepts `--dictate-start`, `--dictate-stop`, and
`--dictate-toggle`; the single-instance bridge forwards them to the running
app without opening the Studio window.
The dependency-free Python bridge is convenient for hooks and TUIs:
```bash
python -m backend.speech_client status
python -m backend.speech_client toggle
python -m backend.speech_client transcribe recording.wav
python -m backend.speech_client transcribe recording.wav --insert
```
`--insert` captures the focused destination before transcription starts and
uses the same clipboard-preserving native delivery as the global shortcut.
## Bring your own capture interface
An editor extension or GUI can own the microphone and consume live text.
Connect to:
```text
ws://127.0.0.1:3900/v1/audio/transcriptions/stream
```
Send binary WebM/Opus frames by default. For raw signed 16-bit mono PCM, use
`?pcm=1&sr=16000`. Finish without closing the socket by sending:
```json
{"type":"input_audio.end"}
```
Every response carries `protocol` and `session_id`:
```json
{"type":"session.started","protocol":"voicestudio.speech.v1","session_id":"..."}
{"type":"partial","text":"hello wor...","session_id":"..."}
{"type":"final","final_kind":"summary","text":"Hello world.","session_id":"..."}
```
Streaming Sherpa models can also emit `final_kind: "utterance"` before the
authoritative whole-session `summary`. Existing `/ws/transcribe` clients keep
their unchanged legacy frames and `EOF` control.
To reuse native insertion with a custom capture client:
1. `POST /v1/output/sessions` on port 3902 before opening the microphone.
2. Stream audio and receive the final text on port 3900.
3. `POST /v1/output/sessions/{id}/insert` with `{"text":"..."}`.
4. If capture is cancelled, `DELETE /v1/output/sessions/{id}`.
Only one output session can own a focused destination at a time. Stale IDs are
rejected instead of inserting into a newer target.
## Batch and agent protocols
| Transport | Endpoint | Use |
|---|---|---|
| OpenAI-compatible HTTP | `POST :3900/v1/audio/transcriptions` | Files, scripts, existing SDKs |
| WebSocket | `:3900/v1/audio/transcriptions/stream` | Partial and final live text |
| MCP Streamable HTTP | `POST :3900/mcp` | Modern agent clients |
| MCP stdio | `python -m backend.mcp_shim` | Claude Code, Codex, and stdio-only clients |
| JSON-RPC | `POST :3902/rpc` | Native dictation control |
| Native CLI | VoiceStudio `--dictate-*` flags | Hooks and plugin actions |
## Integration map
| Interface | Recommended connection |
|---|---|
| Any desktop text field | Existing global shortcut or Rust `dictation.toggle` |
| Herdr | Merge [the example command bindings](../examples/speech-platform/herdr-config.toml) into Herdr's config; detached commands call the Rust API while the pane stays focused |
| Pi, Claude Code, Codex, Antigravity CLI | Dictate into the focused prompt through Rust; add MCP when the agent also needs file transcription or speech tools |
| VS Code | Call Rust HTTP from the extension host for app-wide dictation, or stream editor-owned mic audio over the versioned WebSocket |
| TUI or shell script | `python -m backend.speech_client` or HTTP/JSON-RPC |
| Browser/WebView UI | Stream audio to the Python data plane; browser pages cannot silently call native control |
| Remote microphone + local/remote GPU | Capture at the client edge and use the authenticated WebSocket/OpenAI endpoint |
Loopback clients need no credential. Remote native WebSocket clients can send
the configured bearer key. Browser clients should exchange that key for a
short-lived session, mint a path-bound ticket at `/api/auth/ws-ticket`, and
connect with `?ws_ticket=...`; see [API authentication](api-auth.md).
Keep remote endpoints restricted to a trusted network; an API key authenticates
a client but does not provide network isolation. Beyond a fully trusted LAN,
use HTTPS/WSS and never send bearer credentials or ticket exchanges over
plaintext HTTP/WebSocket.
## Security and privacy
- The native control sidecar binds only to IPv4 loopback and rejects untrusted
browser `Origin` headers, blocking ordinary websites from turning on the mic.
- Native control never accepts audio and is never exposed through Network
Sharing. Remote ASR stays on the existing API-key boundary.
- Microphones stay at the interface edge. A remote GPU backend never assumes
it owns the user's input device.
- No protocol adds a required network call, account, analytics event, or cloud
provider.
## Research basis
The design survey covered five pages of GitHub's
[`speech-to-text` topic](https://github.com/topics/speech-to-text):
[1](https://github.com/topics/speech-to-text?page=1),
[2](https://github.com/topics/speech-to-text?page=2),
[3](https://github.com/topics/speech-to-text?page=3),
[4](https://github.com/topics/speech-to-text?page=4), and
[5](https://github.com/topics/speech-to-text?page=5).
The platform keeps the strongest reusable ideas without copying their UI
boundaries:
| Source | Adopted idea |
|---|---|
| [Handy](https://github.com/cjpais/Handy) | Cross-platform offline dictation, external toggle control, VAD-oriented capture |
| [WhisperLiveKit](https://github.com/QuentinFuxa/WhisperLiveKit) | Live local transcription and compatibility-oriented serving |
| [RealtimeSTT](https://github.com/KoljaB/RealtimeSTT) | Low-latency partials, endpointing, and warm recognizers |
| [sherpa-onnx](https://github.com/k2-fsa/sherpa-onnx) | Portable CPU streaming models and WebSocket-friendly audio framing |
| [FunASR](https://github.com/modelscope/FunASR) | OpenAI-compatible and MCP-facing serving |
| [Vexa](https://github.com/Vexa-ai/vexa) | WebSocket transcripts plus agent access |
| [Voquill](https://github.com/voquill/voquill) | Provider independence, refinement, and personal-vocabulary direction |
| [Muesli](https://github.com/Muesli-HQ/muesli) | Machine-readable CLI contracts and session-safe automation |
| [Herdr](https://github.com/motionharvest/herdr) | One local control surface behind CLI, socket, hooks, and plugin integrations |
The differentiator is the connection layer: one bundled app offers native
capture/output control and a protocol-neutral ASR service, so every interface
does not rebuild model loading, desktop permissions, and insertion safety.
+3 -2
View File
@@ -1,12 +1,14 @@
# VoiceStudio Examples # VoiceStudio Examples
This directory contains scripts and configs for training, fine-tuning, and evaluating VoiceStudio. This directory contains integration examples plus scripts and configs for
training, fine-tuning, and evaluating VoiceStudio.
| Use Case | Script | Description | | Use Case | Script | Description |
|---|---|---| |---|---|---|
| Training from scratch | [run_emilia.sh](run_emilia.sh) | Full pipeline on the Emilia dataset (data check, tokenization, training) | | Training from scratch | [run_emilia.sh](run_emilia.sh) | Full pipeline on the Emilia dataset (data check, tokenization, training) |
| Fine-tuning | [run_finetune.sh](run_finetune.sh) | Fine-tune from a pretrained checkpoint using your own JSONL data | | Fine-tuning | [run_finetune.sh](run_finetune.sh) | Fine-tune from a pretrained checkpoint using your own JSONL data |
| Evaluation | [run_eval.sh](run_eval.sh) | Evaluate WER, speaker similarity, and UTMOS on standard test sets | | Evaluation | [run_eval.sh](run_eval.sh) | Evaluate WER, speaker similarity, and UTMOS on standard test sets |
| Herdr dictation | [speech-platform/herdr-config.toml](speech-platform/herdr-config.toml) | Trigger the bundled Rust dictation sidecar from detached Herdr command bindings |
--- ---
@@ -115,4 +117,3 @@ bash examples/run_eval.sh
``` ```
> See [docs/evaluation.md](../docs/evaluation.md) for metrics details, test set preparation, and running individual metrics. > See [docs/evaluation.md](../docs/evaluation.md) for metrics details, test set preparation, and running individual metrics.
@@ -0,0 +1,21 @@
# Merge these detached command bindings into ~/.config/herdr/config.toml.
# Herdr leaves the current pane focused while VoiceStudio captures its native
# insertion target, so the transcript returns to the pane that started it.
[[keys.command]]
key = "prefix+alt+d"
type = "shell"
command = "curl -fsS -X POST http://127.0.0.1:3902/v1/dictation/toggle"
description = "toggle VoiceStudio dictation"
[[keys.command]]
key = "prefix+alt+s"
type = "shell"
command = "curl -fsS -X POST http://127.0.0.1:3902/v1/dictation/start"
description = "start VoiceStudio dictation"
[[keys.command]]
key = "prefix+alt+x"
type = "shell"
command = "curl -fsS -X POST http://127.0.0.1:3902/v1/dictation/stop"
description = "stop VoiceStudio dictation"
+20 -1
View File
@@ -18,6 +18,7 @@ pub mod dictation_shortcut;
pub mod persistence_exit; pub mod persistence_exit;
pub mod reset; pub mod reset;
pub mod setup; pub mod setup;
pub mod speech_sidecar;
pub mod tools; pub mod tools;
pub mod uninstall; pub mod uninstall;
pub mod updater_channel; pub mod updater_channel;
@@ -530,7 +531,12 @@ pub fn run() {
let app = tauri::Builder::default() let app = tauri::Builder::default()
// Single-instance MUST be registered first. // Single-instance MUST be registered first.
.plugin(tauri_plugin_single_instance::init(move |app, _argv, _cwd| { .plugin(tauri_plugin_single_instance::init(move |app, argv, _cwd| {
if let Some(action) = speech_sidecar::cli_dictation_action(&argv) {
log::info!("Second-instance dictation control: {action:?}");
speech_sidecar::dispatch_action(app, action);
return;
}
log::info!("Second instance attempted — focusing existing window"); log::info!("Second instance attempted — focusing existing window");
// Always the studio window, never the widget. In pill mode this // Always the studio window, never the widget. In pill mode this
// used to target "widget" and show() it — which is precisely the // used to target "widget" and show() it — which is precisely the
@@ -723,6 +729,19 @@ pub fn run() {
capture: Mutex::new(CaptureDispatchState::default()), capture: Mutex::new(CaptureDispatchState::default()),
output: dictation_output::DictationOutput::default(), output: dictation_output::DictationOutput::default(),
}); });
match speech_sidecar::start(app.handle().clone()) {
Ok(sidecar) => {
log::info!("Speech control API ready on port {}", sidecar.port);
app.manage(sidecar);
}
Err(error) => {
log::warn!("Speech control API unavailable: {error}");
}
}
let initial_args: Vec<String> = std::env::args().collect();
if let Some(action) = speech_sidecar::cli_dictation_action(&initial_args) {
speech_sidecar::dispatch_action(app.handle(), action);
}
app.manage(persistence_exit::PersistenceExitState::default()); app.manage(persistence_exit::PersistenceExitState::default());
app.manage(TrayHandle { app.manage(TrayHandle {
tray: Mutex::new(None), tray: Mutex::new(None),
+557
View File
@@ -0,0 +1,557 @@
//! Headless loopback control plane for VoiceStudio dictation.
//!
//! The Python backend owns ASR data-plane protocols. This small Rust server
//! owns desktop authority: start/stop capture and the session-bound native
//! insertion target. Native integrations can therefore control the bundled
//! dictation service without embedding a WebView or depending on Tauri IPC.
use std::io::{Read, Write};
use std::net::{TcpListener, TcpStream};
use std::sync::atomic::{AtomicBool, Ordering};
use std::sync::Arc;
use std::thread;
use std::time::Duration;
use serde_json::{json, Value};
use tauri::Manager;
use crate::dictation_output::CaptureOrigin;
use crate::{backend_port, dispatch_dictation_capture, AppFlags};
const DEFAULT_SIDECAR_PORT: u16 = 3902;
const MAX_BODY_BYTES: usize = 1024 * 1024;
const MAX_REQUEST_BYTES: usize = MAX_BODY_BYTES + 16 * 1024;
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub enum DictationAction {
Start,
Stop,
Toggle,
}
impl DictationAction {
fn wire_name(self) -> &'static str {
match self {
Self::Start => "start",
Self::Stop => "stop",
Self::Toggle => "toggle",
}
}
}
pub struct SpeechSidecarState {
pub port: u16,
stop: Arc<AtomicBool>,
}
impl Drop for SpeechSidecarState {
fn drop(&mut self) {
self.stop.store(true, Ordering::SeqCst);
}
}
struct HttpRequest {
method: String,
path: String,
origin: Option<String>,
body: Vec<u8>,
}
struct HttpResponse {
status: u16,
body: Value,
}
impl HttpResponse {
fn ok(body: Value) -> Self {
Self { status: 200, body }
}
fn error(status: u16, code: &str, message: &str) -> Self {
Self {
status,
body: json!({"error": {"code": code, "message": message}}),
}
}
}
pub fn sidecar_port() -> u16 {
std::env::var("VOICESTUDIO_SPEECH_PORT")
.ok()
.and_then(|value| value.parse::<u16>().ok())
.filter(|port| *port > 0)
.unwrap_or(DEFAULT_SIDECAR_PORT)
}
pub fn cli_dictation_action(args: &[String]) -> Option<DictationAction> {
if args.iter().any(|arg| arg == "--dictate-start") {
return Some(DictationAction::Start);
}
if args.iter().any(|arg| arg == "--dictate-stop") {
return Some(DictationAction::Stop);
}
if args
.iter()
.any(|arg| arg == "--dictate" || arg == "--dictate-toggle")
{
return Some(DictationAction::Toggle);
}
None
}
fn plan_action(
requested: DictationAction,
is_recording: bool,
has_session: bool,
) -> Option<DictationAction> {
match requested {
DictationAction::Start if !is_recording && !has_session => Some(DictationAction::Start),
DictationAction::Stop if is_recording || has_session => Some(DictationAction::Stop),
DictationAction::Toggle if is_recording || has_session => Some(DictationAction::Stop),
DictationAction::Toggle => Some(DictationAction::Start),
_ => None,
}
}
pub fn dispatch_action(app: &tauri::AppHandle, action: DictationAction) -> Value {
let flags = app.state::<AppFlags>();
let before = flags.dictating.load(Ordering::SeqCst);
// A start event creates its output session synchronously, before the
// WebView can report `dictating=true`. Use both signals so two rapid
// toggle calls mean start-then-stop instead of duplicate starts.
let has_session = flags.output.current_session_id().is_some();
let planned = plan_action(action, before, has_session);
if let Some(planned) = planned {
dispatch_dictation_capture(app, planned.wire_name());
}
let session_id = app.state::<AppFlags>().output.current_session_id();
json!({
"accepted": true,
"action": action.wire_name(),
"dispatched_action": planned.map(DictationAction::wire_name),
"already_in_requested_state": planned.is_none(),
"recording_before_request": before,
"session_id": session_id,
})
}
fn status(app: &tauri::AppHandle, port: u16) -> Value {
let flags = app.state::<AppFlags>();
let engine_port = backend_port();
json!({
"schema": "voicestudio.speech-control-status",
"protocol": "voicestudio.speech.v1",
"service": "VoiceStudio",
"service_version": env!("CARGO_PKG_VERSION"),
"recording": flags.dictating.load(Ordering::SeqCst),
"session_id": flags.output.current_session_id(),
"control_url": format!("http://127.0.0.1:{port}"),
"engine_url": format!("http://127.0.0.1:{engine_port}"),
})
}
fn capabilities(port: u16) -> Value {
let engine_port = backend_port();
json!({
"schema": "voicestudio.speech-capabilities",
"protocol": "voicestudio.speech.v1",
"protocol_version": "1.0",
"service": "VoiceStudio",
"service_version": env!("CARGO_PKG_VERSION"),
"local_first": true,
"endpoints": {
"status": format!("http://127.0.0.1:{port}/v1/status"),
"dictation_start": format!("http://127.0.0.1:{port}/v1/dictation/start"),
"dictation_stop": format!("http://127.0.0.1:{port}/v1/dictation/stop"),
"dictation_toggle": format!("http://127.0.0.1:{port}/v1/dictation/toggle"),
"output_sessions": format!("http://127.0.0.1:{port}/v1/output/sessions"),
"json_rpc": format!("http://127.0.0.1:{port}/rpc"),
"batch_transcription": format!("http://127.0.0.1:{engine_port}/v1/audio/transcriptions"),
"streaming_transcription": format!("ws://127.0.0.1:{engine_port}/v1/audio/transcriptions/stream"),
"mcp": format!("http://127.0.0.1:{engine_port}/mcp"),
},
"cli": {
"start": "--dictate-start",
"stop": "--dictate-stop",
"toggle": "--dictate-toggle",
},
})
}
fn parse_output_session_path(path: &str) -> Option<(u64, bool)> {
let suffix = path.strip_prefix("/v1/output/sessions/")?;
if let Some(raw_id) = suffix.strip_suffix("/insert") {
return raw_id.parse().ok().map(|id| (id, true));
}
suffix.parse().ok().map(|id| (id, false))
}
fn begin_output_session(app: &tauri::AppHandle) -> HttpResponse {
let flags = app.state::<AppFlags>();
if flags.output.current_session_id().is_some() {
return HttpResponse::error(
409,
"output_busy",
"another dictation output session is active",
);
}
let session_id = flags.output.begin_session(CaptureOrigin::Shortcut);
HttpResponse::ok(json!({"session_id": session_id}))
}
fn insert_output_session(app: &tauri::AppHandle, session_id: u64, body: &[u8]) -> HttpResponse {
let request: Value = match serde_json::from_slice(body) {
Ok(value) => value,
Err(_) => return HttpResponse::error(400, "invalid_json", "body must be JSON"),
};
let Some(text) = request.get("text").and_then(Value::as_str) else {
return HttpResponse::error(400, "invalid_text", "body requires a string text field");
};
let output = app.state::<AppFlags>().output.clone();
if output.current_session_id() != Some(session_id) {
return HttpResponse::error(409, "stale_session", "output session is not active");
}
let result = output
.activate_session(session_id)
.and_then(|_| output.deliver(session_id, text));
output.finish_session(session_id);
match result {
Ok(outcome) => HttpResponse::ok(json!({
"session_id": session_id,
"outcome": outcome,
})),
Err(error) => HttpResponse::error(500, "delivery_failed", &error),
}
}
fn cancel_output_session(app: &tauri::AppHandle, session_id: u64) -> HttpResponse {
let output = app.state::<AppFlags>().output.clone();
if output.current_session_id() != Some(session_id) {
return HttpResponse::error(409, "stale_session", "output session is not active");
}
output.finish_session(session_id);
HttpResponse::ok(json!({"session_id": session_id, "cancelled": true}))
}
fn action_for_path(path: &str) -> Option<DictationAction> {
match path {
"/v1/dictation/start" => Some(DictationAction::Start),
"/v1/dictation/stop" => Some(DictationAction::Stop),
"/v1/dictation/toggle" => Some(DictationAction::Toggle),
_ => None,
}
}
fn action_for_rpc_method(method: &str) -> Option<DictationAction> {
match method {
"dictation.start" => Some(DictationAction::Start),
"dictation.stop" => Some(DictationAction::Stop),
"dictation.toggle" => Some(DictationAction::Toggle),
_ => None,
}
}
fn handle_rpc(app: &tauri::AppHandle, body: &[u8]) -> HttpResponse {
let request: Value = match serde_json::from_slice(body) {
Ok(value) => value,
Err(_) => {
return HttpResponse::ok(json!({
"jsonrpc": "2.0",
"id": null,
"error": {"code": -32700, "message": "Parse error"},
}));
}
};
let id = request.get("id").cloned().unwrap_or(Value::Null);
let Some(method) = request.get("method").and_then(Value::as_str) else {
return HttpResponse::ok(json!({
"jsonrpc": "2.0",
"id": id,
"error": {"code": -32600, "message": "Invalid Request"},
}));
};
let Some(action) = action_for_rpc_method(method) else {
return HttpResponse::ok(json!({
"jsonrpc": "2.0",
"id": id,
"error": {"code": -32601, "message": "Method not found"},
}));
};
HttpResponse::ok(json!({
"jsonrpc": "2.0",
"id": id,
"result": dispatch_action(app, action),
}))
}
fn origin_allowed(origin: Option<&str>) -> bool {
let Some(origin) = origin else {
return true;
};
matches!(
origin,
"tauri://localhost"
| "http://tauri.localhost"
| "https://tauri.localhost"
| "http://localhost:3901"
| "http://127.0.0.1:3901"
)
}
fn route(app: &tauri::AppHandle, request: HttpRequest, port: u16) -> HttpResponse {
if !origin_allowed(request.origin.as_deref()) {
return HttpResponse::error(
403,
"origin_denied",
"browser origins cannot control dictation",
);
}
let path = request.path.split('?').next().unwrap_or(&request.path);
match (request.method.as_str(), path) {
("GET", "/health") | ("GET", "/v1/status") => HttpResponse::ok(status(app, port)),
("GET", "/.well-known/voicestudio-speech") | ("GET", "/v1/capabilities") => {
HttpResponse::ok(capabilities(port))
}
("POST", "/v1/output/sessions") => begin_output_session(app),
("POST", "/rpc") => handle_rpc(app, &request.body),
("POST", path) if parse_output_session_path(path).is_some() => {
let (session_id, is_insert) = parse_output_session_path(path).expect("guarded above");
if !is_insert {
return HttpResponse::error(405, "method_not_allowed", "use DELETE");
}
insert_output_session(app, session_id, &request.body)
}
("DELETE", path) if parse_output_session_path(path).is_some() => {
let (session_id, is_insert) = parse_output_session_path(path).expect("guarded above");
if is_insert {
return HttpResponse::error(405, "method_not_allowed", "use POST");
}
cancel_output_session(app, session_id)
}
("POST", path) => match action_for_path(path) {
Some(action) => HttpResponse::ok(dispatch_action(app, action)),
None => HttpResponse::error(404, "not_found", "unknown speech-control endpoint"),
},
(_, "/v1/dictation/start" | "/v1/dictation/stop" | "/v1/dictation/toggle") => {
HttpResponse::error(405, "method_not_allowed", "use POST")
}
_ => HttpResponse::error(404, "not_found", "unknown speech-control endpoint"),
}
}
fn header_end(bytes: &[u8]) -> Option<usize> {
bytes.windows(4).position(|window| window == b"\r\n\r\n")
}
fn read_request(stream: &mut TcpStream) -> Result<HttpRequest, &'static str> {
stream
.set_read_timeout(Some(Duration::from_secs(2)))
.map_err(|_| "read timeout setup failed")?;
let mut bytes = Vec::new();
let mut chunk = [0_u8; 2048];
let mut expected_len = None;
loop {
let read = stream.read(&mut chunk).map_err(|_| "request read failed")?;
if read == 0 {
break;
}
bytes.extend_from_slice(&chunk[..read]);
if bytes.len() > MAX_REQUEST_BYTES {
return Err("request too large");
}
if let Some(end) = header_end(&bytes) {
if expected_len.is_none() {
let headers = String::from_utf8_lossy(&bytes[..end]);
let content_length = headers
.lines()
.find_map(|line| {
let (name, value) = line.split_once(':')?;
name.eq_ignore_ascii_case("content-length")
.then(|| value.trim().parse::<usize>().ok())
.flatten()
})
.unwrap_or(0);
if content_length > MAX_BODY_BYTES {
return Err("request body too large");
}
expected_len = Some(end + 4 + content_length);
}
if bytes.len() >= expected_len.unwrap_or(end + 4) {
break;
}
}
}
let end = header_end(&bytes).ok_or("incomplete request headers")?;
let headers = String::from_utf8_lossy(&bytes[..end]);
let mut lines = headers.lines();
let mut request_line = lines
.next()
.ok_or("missing request line")?
.split_whitespace();
let method = request_line.next().ok_or("missing method")?.to_owned();
let path = request_line.next().ok_or("missing path")?.to_owned();
let origin = lines.find_map(|line| {
let (name, value) = line.split_once(':')?;
name.eq_ignore_ascii_case("origin")
.then(|| value.trim().to_owned())
});
Ok(HttpRequest {
method,
path,
origin,
body: bytes[end + 4..].to_vec(),
})
}
fn write_response(stream: &mut TcpStream, response: HttpResponse) {
let body = response.body.to_string();
let reason = match response.status {
200 => "OK",
403 => "Forbidden",
404 => "Not Found",
405 => "Method Not Allowed",
409 => "Conflict",
413 => "Payload Too Large",
500 => "Internal Server Error",
_ => "Bad Request",
};
let head = format!(
"HTTP/1.1 {} {}\r\nContent-Type: application/json\r\nContent-Length: {}\r\nCache-Control: no-store\r\nConnection: close\r\n\r\n",
response.status,
reason,
body.len(),
);
let _ = stream.write_all(head.as_bytes());
let _ = stream.write_all(body.as_bytes());
let _ = stream.flush();
}
fn handle_connection(app: &tauri::AppHandle, mut stream: TcpStream, port: u16) {
match read_request(&mut stream) {
Ok(request) => write_response(&mut stream, route(app, request, port)),
Err(message) => {
let status = if message.contains("too large") {
413
} else {
400
};
write_response(
&mut stream,
HttpResponse::error(status, "invalid_request", message),
);
}
}
}
pub fn start(app: tauri::AppHandle) -> Result<SpeechSidecarState, String> {
let port = sidecar_port();
let listener = TcpListener::bind(("127.0.0.1", port))
.map_err(|error| format!("could not bind 127.0.0.1:{port}: {error}"))?;
listener
.set_nonblocking(true)
.map_err(|error| format!("could not configure speech sidecar: {error}"))?;
// The child Python backend inherits this and advertises the native control
// endpoint only when the desktop shell actually owns it.
std::env::set_var("VOICESTUDIO_SPEECH_CONTROL_PORT", port.to_string());
let stop = Arc::new(AtomicBool::new(false));
let thread_stop = stop.clone();
thread::Builder::new()
.name("speech-control-sidecar".into())
.spawn(move || {
log::info!("Speech control sidecar listening on 127.0.0.1:{port}");
while !thread_stop.load(Ordering::SeqCst) {
match listener.accept() {
Ok((stream, _)) => handle_connection(&app, stream, port),
Err(error) if error.kind() == std::io::ErrorKind::WouldBlock => {
thread::sleep(Duration::from_millis(30));
}
Err(error) => {
log::warn!("Speech control sidecar accept failed: {error}");
thread::sleep(Duration::from_millis(100));
}
}
}
})
.map_err(|error| format!("could not start speech sidecar thread: {error}"))?;
Ok(SpeechSidecarState { port, stop })
}
#[cfg(test)]
mod tests {
use super::{
action_for_path, action_for_rpc_method, cli_dictation_action, origin_allowed,
parse_output_session_path, plan_action, DictationAction,
};
fn args(values: &[&str]) -> Vec<String> {
values.iter().map(|value| (*value).to_owned()).collect()
}
#[test]
fn cli_flags_map_to_idempotent_actions() {
assert_eq!(
cli_dictation_action(&args(&["VoiceStudio", "--dictate-start"])),
Some(DictationAction::Start)
);
assert_eq!(
cli_dictation_action(&args(&["VoiceStudio", "--dictate-stop"])),
Some(DictationAction::Stop)
);
assert_eq!(
cli_dictation_action(&args(&["VoiceStudio", "--dictate-toggle"])),
Some(DictationAction::Toggle)
);
assert_eq!(cli_dictation_action(&args(&["VoiceStudio"])), None);
}
#[test]
fn http_and_json_rpc_share_the_same_action_vocabulary() {
assert_eq!(
action_for_path("/v1/dictation/start"),
Some(DictationAction::Start)
);
assert_eq!(
action_for_rpc_method("dictation.stop"),
Some(DictationAction::Stop)
);
assert_eq!(action_for_rpc_method("dictation.delete"), None);
}
#[test]
fn pending_start_makes_a_second_toggle_stop() {
assert_eq!(
plan_action(DictationAction::Toggle, false, false),
Some(DictationAction::Start)
);
assert_eq!(
plan_action(DictationAction::Toggle, false, true),
Some(DictationAction::Stop)
);
assert_eq!(plan_action(DictationAction::Start, false, true), None);
}
#[test]
fn browser_origins_cannot_silently_trigger_the_microphone() {
assert!(origin_allowed(None));
assert!(origin_allowed(Some("http://tauri.localhost")));
assert!(origin_allowed(Some("http://localhost:3901")));
assert!(!origin_allowed(Some("https://example.com")));
assert!(!origin_allowed(Some("http://localhost.evil.test")));
}
#[test]
fn output_session_paths_are_strictly_typed() {
assert_eq!(
parse_output_session_path("/v1/output/sessions/42"),
Some((42, false))
);
assert_eq!(
parse_output_session_path("/v1/output/sessions/42/insert"),
Some((42, true))
);
assert_eq!(parse_output_session_path("/v1/output/sessions/nope"), None);
}
}
+3
View File
@@ -22,6 +22,7 @@ DELETE /pronunciation/{entry_id}
DELETE /workers/inbound/connections/{endpoint} DELETE /workers/inbound/connections/{endpoint}
DELETE /workers/inbound/keys/{key_id} DELETE /workers/inbound/keys/{key_id}
DELETE /workers/{worker_id} DELETE /workers/{worker_id}
GET /.well-known/voicestudio-speech
GET /api/mcp/bindings GET /api/mcp/bindings
GET /api/settings/analytics GET /api/settings/analytics
GET /api/settings/asr-openai-compat GET /api/settings/asr-openai-compat
@@ -140,6 +141,7 @@ GET /system/tailscale/status
GET /tasks/stream/{task_id} GET /tasks/stream/{task_id}
GET /tools/effects GET /tools/effects
GET /tools/plugins GET /tools/plugins
GET /v1/audio/capabilities
GET /v1/audio/voices GET /v1/audio/voices
GET /watermark/status GET /watermark/status
GET /workers GET /workers
@@ -286,6 +288,7 @@ PUT /history/{history_id}/starred
PUT /profiles/{profile_id} PUT /profiles/{profile_id}
PUT /projects/{project_id} PUT /projects/{project_id}
PUT /pronunciation/{entry_id} PUT /pronunciation/{entry_id}
WS /v1/audio/transcriptions/stream
WS /ws/events WS /ws/events
WS /ws/transcribe WS /ws/transcribe
WS /ws/tts WS /ws/tts
+2
View File
@@ -57,7 +57,9 @@ _CRITICAL = [
"POST /audiobook", "POST /stories/encode", "POST /batch/enqueue", "POST /audiobook", "POST /stories/encode", "POST /batch/enqueue",
"POST /transcribe", "GET /api/settings/hf-token/state", "POST /transcribe", "GET /api/settings/hf-token/state",
"POST /v1/audio/speech", "POST /v1/audio/transcriptions", "POST /v1/audio/speech", "POST /v1/audio/transcriptions",
"GET /.well-known/voicestudio-speech", "GET /v1/audio/capabilities",
"WS /ws/events", "WS /ws/tts", "WS /ws/transcribe", "WS /ws/events", "WS /ws/tts", "WS /ws/transcribe",
"WS /v1/audio/transcriptions/stream",
] ]
+6 -2
View File
@@ -457,13 +457,17 @@ def test_legacy_cookie_migration_fails_without_exact_origin(origin):
) )
def test_session_can_mint_path_bound_ws_ticket(): @pytest.mark.parametrize(
"path",
["/ws/transcribe", "/v1/audio/transcriptions/stream"],
)
def test_session_can_mint_path_bound_ws_ticket(path):
client = _client() client = _client()
token = _issue_bearer(client).json()["token"] token = _issue_bearer(client).json()["token"]
response = client.post( response = client.post(
"/api/auth/ws-ticket", "/api/auth/ws-ticket",
json={"path": "/ws/transcribe"}, json={"path": path},
headers={"Authorization": f"Bearer {token}"}, headers={"Authorization": f"Bearer {token}"},
) )
+15
View File
@@ -255,6 +255,21 @@ def test_ws_ticket_is_path_bound_single_use_and_origin_checked(key_env):
assert exc_info.value.code == 1008 assert exc_info.value.code == 1008
def test_platform_ws_accepts_path_bound_ticket(key_env):
from services.admin_sessions import admin_session_store
path = "/v1/audio/transcriptions/stream"
session = admin_session_store.issue(key_env)
ticket = admin_session_store.issue_ws_ticket(session.token, path, key_env)
with _client().websocket_connect(
f"{path}?ws_ticket={ticket.token}",
headers={"Origin": "http://testserver"},
) as ws:
assert ws.receive_json()["type"] == "session.started"
ws.close()
def test_ws_ticket_wrong_path_consumes_ticket(key_env): def test_ws_ticket_wrong_path_consumes_ticket(key_env):
from services.admin_sessions import admin_session_store from services.admin_sessions import admin_session_store
+129
View File
@@ -0,0 +1,129 @@
from __future__ import annotations
from types import SimpleNamespace
import pytest
def test_url_join_does_not_duplicate_slashes():
from speech_client.__main__ import _join_url
assert _join_url("http://127.0.0.1:3902/", "/v1/status") == (
"http://127.0.0.1:3902/v1/status"
)
def test_multipart_matches_openai_audio_contract():
from speech_client.__main__ import _encode_multipart
body, content_type = _encode_multipart(
filename='sample.wav',
audio=b"RIFF-audio",
fields={"model": "whisper-1", "response_format": "text"},
boundary="fixed-boundary",
)
assert content_type == "multipart/form-data; boundary=fixed-boundary"
assert b'name="model"\r\n\r\nwhisper-1' in body
assert b'name="response_format"\r\n\r\ntext' in body
assert b'name="file"; filename="sample.wav"' in body
assert b"Content-Type: audio/wav" in body
assert b"RIFF-audio" in body
assert body.endswith(b"--fixed-boundary--\r\n")
def test_json_transcription_response_extracts_insertable_text():
from speech_client.__main__ import _response_text
assert _response_text(b'{"text":"hello"}', "application/json") == "hello"
assert _response_text(b"hello", "text/plain") == "hello"
def test_client_rejects_non_http_url_handlers():
from urllib import request
from speech_client.__main__ import SpeechClientError, _open
with pytest.raises(SpeechClientError, match="must use http"):
_open(request.Request("file:///etc/passwd"))
@pytest.mark.parametrize(
"path",
[
"/Users/alice/Private/voice.wav",
r"C:\Users\Alice\Private\voice.wav",
],
)
def test_audio_read_errors_hide_parent_directories(monkeypatch, path):
from pathlib import Path
from speech_client.__main__ import SpeechClientError, _read_audio
def denied(_self):
raise PermissionError(13, "Permission denied", path)
monkeypatch.setattr(Path, "read_bytes", denied)
with pytest.raises(SpeechClientError) as exc_info:
_read_audio(path, "audio.wav")
message = str(exc_info.value)
assert "voice.wav" in message
assert "alice" not in message.lower()
assert "users" not in message.lower()
def test_remote_bearer_rejects_plain_http_before_network():
from urllib import request
from speech_client.__main__ import SpeechClientError, _open
req = request.Request(
"http://gpu.example/v1/audio/transcriptions",
headers={"Authorization": "Bearer secret"},
)
with pytest.raises(SpeechClientError, match="require https"):
_open(req)
def test_credentialed_redirects_are_rejected():
from speech_client.__main__ import SpeechClientError, _RejectCredentialRedirect
handler = _RejectCredentialRedirect()
with pytest.raises(SpeechClientError, match="credentialed redirect"):
handler.redirect_request(None, None, 307, "redirect", {}, "https://other.test")
def test_interrupt_releases_focused_output_session(monkeypatch):
from speech_client import __main__ as client
calls = []
def fake_json(method, url, payload=None):
calls.append((method, url, payload))
if method == "POST" and url.endswith("/v1/output/sessions"):
return {"session_id": 42}
return {"ok": True}
monkeypatch.setattr(client, "_read_audio", lambda *_args: (b"audio", "audio.wav"))
monkeypatch.setattr(client, "_json_request", fake_json)
def interrupt(*_args, **_kwargs):
raise KeyboardInterrupt
monkeypatch.setattr(client, "_open", interrupt)
args = SimpleNamespace(
audio="ignored.wav",
stdin_filename="audio.wav",
model="whisper-1",
response_format="text",
language=None,
insert=True,
control_url="http://127.0.0.1:3902",
engine_url="http://127.0.0.1:3900",
)
with pytest.raises(KeyboardInterrupt):
client._transcribe(args)
assert calls[-1][0] == "DELETE"
assert calls[-1][1].endswith("/v1/output/sessions/42")
+120
View File
@@ -0,0 +1,120 @@
"""Public contract for the headless speech platform."""
from __future__ import annotations
import os
import pytest
os.environ.setdefault("OMNIVOICE_MODEL", "test")
os.environ.setdefault("OMNIVOICE_DISABLE_FILE_LOG", "1")
def test_capabilities_are_stable_and_side_effect_free(monkeypatch):
from api.routers.speech_platform import speech_capabilities
monkeypatch.delenv("VOICESTUDIO_SPEECH_CONTROL_PORT", raising=False)
body = speech_capabilities().model_dump(by_alias=True)
assert body["schema"] == "voicestudio.speech-capabilities"
assert body["protocol"] == "voicestudio.speech.v1"
assert body["local_first"] is True
assert body["endpoints"]["batch_transcription"]["path"] == (
"/v1/audio/transcriptions"
)
assert body["endpoints"]["streaming_transcription"]["path"] == (
"/v1/audio/transcriptions/stream"
)
assert body["stream_input"]["end_control"] == {"type": "input_audio.end"}
assert body["features"]["native_dictation_control"] is False
assert "native_dictation_control" not in body["endpoints"]
assert body["authentication"]["websocket_ticket_endpoint"] == (
"/api/auth/ws-ticket"
)
assert body["authentication"]["websocket_ticket_query_parameter"] == "ws_ticket"
def test_desktop_backend_advertises_rust_control_sidecar(monkeypatch):
from api.routers.speech_platform import speech_capabilities
monkeypatch.setenv("VOICESTUDIO_SPEECH_CONTROL_PORT", "4902")
body = speech_capabilities().model_dump(by_alias=True)
assert body["features"]["native_dictation_control"] is True
assert body["endpoints"]["native_dictation_control"]["path"] == (
"http://127.0.0.1:4902/v1/capabilities"
)
@pytest.mark.parametrize(
("frame", "expected"),
[
("EOF", True),
('{"type":"input_audio.end"}', True),
('{"type":"session.update"}', False),
("not-json", False),
(None, False),
],
)
def test_stream_end_control_is_versioned_and_legacy_compatible(frame, expected):
from api.routers.capture_ws import _is_end_control
assert _is_end_control(frame) is expected
@pytest.mark.usefixtures("asr_model_installed")
def test_versioned_stream_has_session_envelope(monkeypatch):
from fastapi import FastAPI
from fastapi.testclient import TestClient
from api.routers import capture_ws as capture
async def fake_full(_chunks, **_kwargs):
return {
"text": "platform works",
"segments": [],
"language": "en",
"engine": "stub",
}
monkeypatch.setattr(capture, "_transcribe_buffer_full", fake_full)
app = FastAPI()
app.include_router(capture.router)
client = TestClient(app, client=("127.0.0.1", 50000))
with client.websocket_connect(
"/v1/audio/transcriptions/stream?pcm=1&sr=16000"
) as websocket:
started = websocket.receive_json()
assert started["type"] == "session.started"
assert started["protocol"] == "voicestudio.speech.v1"
assert started["sample_rate"] == 16000
session_id = started["session_id"]
websocket.send_bytes(b"\x00" * 5000)
websocket.send_json({"type": "input_audio.end"})
final = websocket.receive_json()
assert final["type"] == "final"
assert final["final_kind"] == "summary"
assert final["session_id"] == session_id
assert final["protocol"] == "voicestudio.speech.v1"
assert final["text"] == "Platform works."
def test_versioned_stream_rejects_untrusted_browser_origin_before_accept():
from fastapi import FastAPI
from fastapi.testclient import TestClient
from starlette.websockets import WebSocketDisconnect
from api.routers import capture_ws as capture
app = FastAPI()
app.include_router(capture.router)
client = TestClient(app, client=("127.0.0.1", 50000))
with pytest.raises(WebSocketDisconnect) as exc_info:
with client.websocket_connect(
"/v1/audio/transcriptions/stream",
headers={"Origin": "https://malicious.example"},
):
pass
assert exc_info.value.code == 1008