Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
34 changes: 34 additions & 0 deletions examples/other/oruk_transcribe.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
"""Transcribe a PCM16 WAV file using ORUK_API_KEY, without a LiveKit room."""

import argparse
import asyncio
import wave

from livekit import rtc
from livekit.plugins import oruk


async def main() -> None:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("audio", help="PCM16 WAV, 45 ms to 60 seconds")
parser.add_argument("--model", default="oruk-spectra-2")
args = parser.parse_args()
with wave.open(args.audio) as wav:
if wav.getsampwidth() != 2:
raise ValueError("The example requires PCM16 WAV")
frame = rtc.AudioFrame(
wav.readframes(wav.getnframes()),
wav.getframerate(),
wav.getnchannels(),
wav.getnframes(),
)
recognizer = oruk.STT(model=args.model)
try:
event = await recognizer.recognize(frame)
print(event.alternatives[0].text)
finally:
await recognizer.aclose()


if __name__ == "__main__":
asyncio.run(main())
1 change: 1 addition & 0 deletions livekit-agents/pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -106,6 +106,7 @@ neuphonic = ["livekit-plugins-neuphonic>=1.8.3"]
nltk = ["livekit-plugins-nltk>=1.8.3"]
nvidia = ["livekit-plugins-nvidia>=1.8.3"]
openai = ["livekit-plugins-openai>=1.8.3"]
oruk = ["livekit-plugins-oruk>=1.8.3"]
palabra = ["livekit-plugins-palabra>=1.8.3"]
perplexity = ["livekit-plugins-perplexity>=1.8.3"]
protoface = ["livekit-plugins-protoface>=1.8.3"]
Expand Down
31 changes: 31 additions & 0 deletions livekit-plugins/livekit-plugins-oruk/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,31 @@
# Oruk hosted STT plugin

This plugin sends final utterances to the authenticated [Oruk speech API](https://oruk.ai/docs). It defaults to Spectra-2 (`oruk-spectra-2`), which requires preview access on the account behind the API key. This is separate from local Orukeet inference: audio leaves the device and uses the account's plan minutes or credits.

Before a package release, install from this checkout:

```bash
uv sync --package livekit-agents --extra oruk --extra silero --dev
export ORUK_API_KEY='your-own-api-key'
uv run python examples/other/oruk_transcribe.py recording.wav
```

The example accepts PCM16 WAV and makes a real authenticated request. It does not require a LiveKit room, OpenAI credentials, or a local model download. Never put your key in source code or a browser client.

For a voice agent, use LiveKit's existing VAD adapter:

```python
from livekit.agents import AgentSession, stt
from livekit.plugins import oruk, silero

session = AgentSession(
stt=stt.StreamAdapter(stt=oruk.STT(), vad=silero.VAD.load()),
# Supply the LLM and TTS providers for your application.
)
```

This is final-utterance recognition, not native partial streaming. The adapter waits for VAD to end an utterance. Keep utterances between 45 ms and 60 seconds; the plugin downmixes PCM to mono and resamples to 16 kHz. Spectra-2 returns automatic transcription across 25 languages and clip-level scores, but this STT interface returns the transcript only. It does not invent a language identifier, word timestamps, confidence, or diarization. Language forcing is unsupported. Check the current API docs for other model limits before selecting a different `model`.

Transport retries keep the same request ID and exact WAV bytes. A `model_busy` rejection gets a new ID as required by the API; `upload_busy` retains its ID. The plugin honors `Retry-After`, rejects redirects, does not retry authentication or 409 completion/uncertainty errors, and never retries a malformed successful response. A lost completed response cannot be retrieved by replay: reconcile its request ID before manually creating another recognition call.

Call `await recognizer.aclose()` when using it outside an agent session. A supplied `httpx.AsyncClient` stays owned by the caller. This integration makes no latency or quality guarantee; use your own held-out recordings to measure the complete VAD, network, and inference path.
Original file line number Diff line number Diff line change
@@ -0,0 +1,18 @@
"""Hosted Oruk speech-to-text for LiveKit Agents."""

import logging

from livekit.agents import Plugin

from .stt import STT
from .version import __version__

__all__ = ["STT", "__version__"]


class OrukPlugin(Plugin):
def __init__(self) -> None:
super().__init__(__name__, __version__, __package__, logging.getLogger(__name__))


Plugin.register_plugin(OrukPlugin())
Empty file.
196 changes: 196 additions & 0 deletions livekit-plugins/livekit-plugins-oruk/livekit/plugins/oruk/stt.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,196 @@
from __future__ import annotations

import asyncio
import math
import os
import uuid
from contextvars import ContextVar
from dataclasses import dataclass, replace

import httpx
import numpy as np

from livekit import rtc
from livekit.agents import (
APIConnectionError,
APIConnectOptions,
APIError,
APIStatusError,
APITimeoutError,
LanguageCode,
stt,
)
from livekit.agents.types import DEFAULT_API_CONNECT_OPTIONS, NOT_GIVEN, NotGivenOr
from livekit.agents.utils import AudioBuffer, is_given


@dataclass
class _Request:
id: str
retry_after: float = 0


class STT(stt.STT):
"""Final-utterance transcription through the authenticated Oruk API.

Spectra-2 preview entitlement is required for the default model. Use a VAD
StreamAdapter for a live agent; this file API does not produce partial words.
"""

def __init__(
self,
*,
api_key: NotGivenOr[str] = NOT_GIVEN,
model: str = "oruk-spectra-2",
base_url: str = "https://speech-api.oruk.ai",
http_client: httpx.AsyncClient | None = None,
) -> None:
super().__init__(capabilities=stt.STTCapabilities(streaming=False, interim_results=False))
key = api_key if is_given(api_key) else os.environ.get("ORUK_API_KEY")
if not key:
raise ValueError("Set ORUK_API_KEY or pass api_key")
if not model.strip():
raise ValueError("model must not be empty")
self._key = key
self._model = model
self._url = f"{base_url.rstrip('/')}/v1/audio/transcriptions"
self._client = http_client
self._owns_client = http_client is None
# Each concurrent utterance has its own ID; transport retries retain it.
self._request: ContextVar[_Request] = ContextVar("oruk_stt_request")

@property
def model(self) -> str:
return self._model

@property
def provider(self) -> str:
return "Oruk"

async def aclose(self) -> None:
if self._owns_client and self._client is not None:
await self._client.aclose()

async def recognize(
self,
buffer: AudioBuffer,
*,
language: NotGivenOr[str] = NOT_GIVEN,
conn_options: APIConnectOptions = DEFAULT_API_CONNECT_OPTIONS,
) -> stt.SpeechEvent:
request = _Request(str(uuid.uuid4()))
token = self._request.set(request)
try:
# The base batch loop retries all APIErrors. Respect retryable here,
# so auth failures and uncertain/completed request IDs do not replay.
for attempt in range(conn_options.max_retry + 1):
try:
return await super().recognize(
buffer,
language=language,
conn_options=replace(conn_options, max_retry=0),
)
except APIError as exc:
if not exc.retryable or attempt == conn_options.max_retry:
raise
await asyncio.sleep(
max(request.retry_after, conn_options._interval_for_retry(attempt))
)
request.retry_after = 0
raise RuntimeError("unreachable")
finally:
self._request.reset(token)

async def _recognize_impl(
self,
buffer: AudioBuffer,
*,
language: NotGivenOr[str] = NOT_GIVEN,
conn_options: APIConnectOptions,
) -> stt.SpeechEvent:
if is_given(language) and language:
raise ValueError(
"Oruk selects the language automatically; language forcing is unsupported"
)
wav = _wav(buffer)
request = self._request.get()
if self._client is None:
self._client = httpx.AsyncClient()
try:
response = await self._client.post(
self._url,
headers={"Authorization": f"Bearer {self._key}", "X-Request-ID": request.id},
data={"model": self._model},
files={"file": ("audio.wav", wav, "audio/wav")},
timeout=httpx.Timeout(conn_options.timeout),
follow_redirects=False,
)
except httpx.TimeoutException as exc:
raise APITimeoutError("Oruk transcription timed out") from exc
except httpx.RequestError as exc:
raise APIConnectionError("Oruk transcription connection failed") from exc

if not response.is_success:
code = ""
try:
body = response.json()
error = body.get("error", {}) if isinstance(body, dict) else {}
code = error.get("code", "") if isinstance(error, dict) else ""
except ValueError:
pass
retryable = response.status_code >= 500 or response.status_code == 429
if response.status_code == 429:
try:
delay = float(response.headers.get("Retry-After", "0"))
if math.isfinite(delay) and delay >= 0:
request.retry_after = delay
except ValueError:
pass
# model_busy closes the attempt without inference/charge. A new
# ID is required; upload_busy and transport failures retain it.
if code == "model_busy":
request.id = str(uuid.uuid4())
raise APIStatusError(
f"Oruk transcription failed ({code or response.status_code})",
status_code=response.status_code,
request_id=response.headers.get("X-Request-ID"),
retryable=retryable,
)
try:
result = response.json()
if not isinstance(result, dict) or not isinstance(result.get("text"), str):
raise ValueError("missing transcript")
language_code = result.get("language") or ""
if not isinstance(language_code, str):
raise ValueError("invalid language")
return stt.SpeechEvent(
type=stt.SpeechEventType.FINAL_TRANSCRIPT,
request_id=response.headers.get("X-Request-ID", request.id),
alternatives=[
stt.SpeechData(text=result["text"], language=LanguageCode(language_code))
],
)
except (ValueError, TypeError) as exc:
raise APIError("Invalid Oruk transcription response", retryable=False) from exc


def _wav(buffer: AudioBuffer) -> bytes:
frame = rtc.combine_audio_frames(buffer)
duration = frame.samples_per_channel / frame.sample_rate
if not 0.045 <= duration <= 60:
raise ValueError("Oruk utterances must be between 45 ms and 60 seconds")
if frame.num_channels != 1:
samples = np.frombuffer(frame.data, dtype=np.int16).reshape(-1, frame.num_channels)
mono = np.rint(samples.astype(np.float32).mean(axis=1)).astype(np.int16)
frame = rtc.AudioFrame(
data=mono.tobytes(),
sample_rate=frame.sample_rate,
num_channels=1,
samples_per_channel=frame.samples_per_channel,
)
if frame.sample_rate != 16000:
resampler = rtc.AudioResampler(
input_rate=frame.sample_rate, output_rate=16000, num_channels=1
)
frame = rtc.combine_audio_frames([*resampler.push(frame), *resampler.flush()])
return frame.to_wav_bytes()
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
__version__ = "1.8.3"
26 changes: 26 additions & 0 deletions livekit-plugins/livekit-plugins-oruk/pyproject.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,26 @@
[build-system]
requires = ["hatchling"]
build-backend = "hatchling.build"

[project]
name = "livekit-plugins-oruk"
dynamic = ["version"]
description = "LiveKit Agents plugin for the hosted Oruk speech API"
readme = "README.md"
license = "Apache-2.0"
requires-python = ">=3.10.0"
authors = [{ name = "LiveKit", email = "hello@livekit.io" }]
dependencies = ["livekit-agents>=1.8.3", "httpx>=0.28,<1"]

[project.urls]
Documentation = "https://oruk.ai/docs"
Source = "https://github.com/livekit/agents"

[tool.hatch.version]
path = "livekit/plugins/oruk/version.py"

[tool.hatch.build.targets.wheel]
packages = ["livekit"]

[tool.hatch.build.targets.sdist]
include = ["/livekit", "/README.md"]
1 change: 1 addition & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,7 @@ livekit-plugins-neuphonic = { workspace = true }
livekit-plugins-nltk = { workspace = true }
livekit-plugins-nvidia = { workspace = true }
livekit-plugins-openai = { workspace = true }
livekit-plugins-oruk = { workspace = true }
livekit-plugins-palabra = { workspace = true }
livekit-plugins-perplexity = { workspace = true }
livekit-plugins-protoface = { workspace = true }
Expand Down
Loading
Loading