Appearance
Example scripts
Five runnable scripts covering every endpoint, built on the stock openai Python client. Copy one and change the key.
They share a single client.py that resolves the key and base URL, so swapping in a secrets manager or a per-tenant key means editing one file rather than five.
bash
pip install "openai[realtime]>=2.43"
export XVOICE_API_KEY=sk_test_...
export XVOICE_BASE_URL=https://your-xvoice-host/v1Every script takes --help, --api-key and --base-url.
A round trip — speak some text, then transcribe it back three ways:
bash
python tts.py "Hello, this is a round trip through xVoice." # writes output/speech.wav
python stt.py output/speech.wav
python stt_stream.py output/speech.wav --language en
python realtime.py output/speech.wavclient.py
Shared setup: configuration and error formatting, in one place.
python
"""One place to build the client, so every example reads the key and server the same way.
xVoice speaks the OpenAI audio API, so the stock `openai` client works as-is: only the
base URL and the API key change.
Configuration, highest priority first:
1. --api-key / --base-url flags
2. XVOICE_API_KEY / XVOICE_BASE_URL environment variables
3. base URL defaults to a local server on port 8000
"""
import argparse
import asyncio
import json
import os
import sys
from collections.abc import Awaitable, Callable
from openai import APIConnectionError, APIError, AsyncOpenAI, OpenAI
DEFAULT_BASE_URL = "http://localhost:8000/v1"
def add_client_arguments(parser: argparse.ArgumentParser) -> None:
parser.add_argument(
"--api-key",
default=os.environ.get("XVOICE_API_KEY"),
help="xVoice API key (default: $XVOICE_API_KEY)",
)
parser.add_argument(
"--base-url",
default=os.environ.get("XVOICE_BASE_URL", DEFAULT_BASE_URL),
help=f"xVoice API base URL including /v1 (default: $XVOICE_BASE_URL or {DEFAULT_BASE_URL})",
)
def _require_key(args: argparse.Namespace) -> None:
if not args.api_key:
sys.exit(
"No API key. Create one in the dashboard (Projects → API keys), then either\n"
" export XVOICE_API_KEY=sk_test_...\n"
"or pass --api-key sk_test_..."
)
def make_client(args: argparse.Namespace) -> OpenAI:
_require_key(args)
return OpenAI(api_key=args.api_key, base_url=args.base_url)
def make_async_client(args: argparse.Namespace) -> AsyncOpenAI:
"""The async client. `realtime.connect()` exists only on this one."""
_require_key(args)
return AsyncOpenAI(api_key=args.api_key, base_url=args.base_url)
def describe(status: object, error: dict) -> str:
return (
f"Error {status} ({error.get('code', 'unknown')}): {error.get('message', '')}\n"
f"request_id: {error.get('request_id')}"
)
def _rejected_handshake(exc: BaseException) -> str | None:
"""A WebSocket refused *before* the upgrade is an HTTP response, not an SDK error.
The realtime endpoint answers a bad key or an exhausted balance that way, so the
body is the usual xVoice envelope but nothing in `openai` will have parsed it.
Duck-typed on purpose: this is `websockets.exceptions.InvalidStatus`, an
implementation detail of the SDK's transport that we would rather not import.
"""
response = getattr(exc, "response", None)
status = getattr(response, "status_code", None)
body = getattr(response, "body", None)
if status is None or body is None:
return None
try:
error = json.loads(body or b"{}").get("error", {})
except ValueError:
error = {}
return describe(status, error)
def run(main: Callable[[], None]) -> None:
"""Run an example, turning API errors into a readable message instead of a traceback."""
try:
main()
except (APIConnectionError, APIError) as exc:
sys.exit(_message(exc))
def run_async(main: Callable[[], Awaitable[None]]) -> None:
"""`run`, for the examples that need an event loop."""
try:
asyncio.run(main())
except KeyboardInterrupt:
sys.exit(130)
except (APIConnectionError, APIError) as exc:
sys.exit(_message(exc))
except Exception as exc: # noqa: BLE001 - re-raised below unless we recognise it
refusal = _rejected_handshake(exc)
if refusal is None:
raise
sys.exit(refusal)
def _message(exc: APIError) -> str:
if isinstance(exc, APIConnectionError):
refusal = _rejected_handshake(exc.__cause__ or exc)
if refusal is not None:
return refusal
return f"Could not reach the xVoice server ({exc.message}) Is it running at the base URL?"
# The client unwraps xVoice's {"error": {type, code, message, request_id}} into
# `body`. Status errors carry the HTTP status; a failure mid-stream has none.
body = exc.body if isinstance(exc.body, dict) else {}
error = body.get("error", body)
error.setdefault("message", exc.message)
return describe(getattr(exc, "status_code", "stream"), error)list_models.py
What your key can call — models, their capabilities, and the voices.
python
"""List the models and voices available to your API key.
uv run list_models.py
"""
import argparse
from client import add_client_arguments, make_client, run
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawTextHelpFormatter)
add_client_arguments(parser)
client = make_client(parser.parse_args())
models = client.models.list()
# /voices is an xVoice extension, so there is no typed SDK method for it; the
# client's generic get() still handles auth, retries and errors.
voices = client.get("/voices", cast_to=object)
print("Models")
for model in models:
# xVoice adds type and supported_languages to OpenAI's model object.
extra = model.model_extra or {}
languages = ", ".join(extra.get("supported_languages", []))
print(f" {model.id:<16} {extra.get('type', ''):<4} {languages}")
print("\nVoices")
for voice in voices["data"]:
print(f" {voice['id']:<16} {voice['language']:<3} {voice['gender']}")
if __name__ == "__main__":
run(main)tts.py
Text to a complete audio file.
python
"""Text to speech: generate a complete audio file.
uv run tts.py "Hello from xVoice."
uv run tts.py "Hallo!" --voice de-female --format mp3 --out output/hallo.mp3
"""
import argparse
from pathlib import Path
from client import add_client_arguments, make_client, run
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawTextHelpFormatter)
parser.add_argument("text", help="text to speak")
parser.add_argument("--model", default="xvoice-tts1")
parser.add_argument("--voice", default="neutral-female", help="see list_models.py")
parser.add_argument("--format", default="wav", choices=["wav", "mp3", "flac", "pcm", "opus"])
parser.add_argument("--out", type=Path, help="default: output/speech.<format>")
add_client_arguments(parser)
args = parser.parse_args()
client = make_client(args)
out = args.out or Path("output") / f"speech.{args.format}"
out.parent.mkdir(parents=True, exist_ok=True)
response = client.audio.speech.create(
model=args.model,
voice=args.voice,
input=args.text,
response_format=args.format,
)
response.write_to_file(out)
request_id = response.response.headers.get("x-request-id")
print(f"Wrote {out} ({out.stat().st_size:,} bytes), request {request_id}")
if __name__ == "__main__":
run(main)tts_stream.py
Text to speech, streamed, with time to first audio.
python
"""Text to speech, streamed: audio is written as it is generated.
uv run tts_stream.py "A longer passage that benefits from hearing it start sooner."
Streaming needs `stream_format` set; `with_streaming_response` alone only stops the
client from buffering, while the server would still generate the whole clip first.
Audio arrives as raw PCM: 16-bit signed little-endian, mono, 24 kHz, with no header,
which is what a live player wants. To listen to the saved file:
ffplay -f s16le -ar 24000 -ch_layout mono output/speech-stream.pcm
"""
import argparse
import time
from pathlib import Path
from client import add_client_arguments, make_client, run
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawTextHelpFormatter)
parser.add_argument("text", help="text to speak")
parser.add_argument("--model", default="xvoice-tts1")
parser.add_argument("--voice", default="neutral-female", help="see list_models.py")
parser.add_argument("--out", type=Path, default=Path("output/speech-stream.pcm"))
add_client_arguments(parser)
args = parser.parse_args()
client = make_client(args)
args.out.parent.mkdir(parents=True, exist_ok=True)
started = time.perf_counter()
first_audio = None
total = 0
with client.audio.speech.with_streaming_response.create(
model=args.model,
voice=args.voice,
input=args.text,
response_format="pcm",
stream_format="audio",
) as response, args.out.open("wb") as file:
for chunk in response.iter_bytes():
if first_audio is None:
first_audio = time.perf_counter() - started
file.write(chunk)
total += len(chunk)
# Hand `chunk` to an audio player here for live playback.
elapsed = time.perf_counter() - started
if first_audio is None:
raise SystemExit(f"No audio received after {elapsed:.2f}s")
print(f"Wrote {args.out} ({total:,} bytes)")
print(f"First audio after {first_audio:.2f}s, finished after {elapsed:.2f}s")
if __name__ == "__main__":
run(main)stt.py
Transcribe an audio file.
python
"""Speech to text: transcribe an audio file.
uv run stt.py output/speech.wav
uv run stt.py meeting.m4a --language en --prompt "Names: Acme, xVoice"
Accepts flac, m4a, mp3, mp4, ogg, wav and webm, up to 25 MB and 15 minutes.
Billed per second of audio.
"""
import argparse
from pathlib import Path
from client import add_client_arguments, make_client, run
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawTextHelpFormatter)
parser.add_argument("file", type=Path, help="audio file to transcribe")
parser.add_argument("--model", default="xvoice-stt1")
parser.add_argument(
"--language",
help="ISO-639-1 code, e.g. en. Omit to auto-detect; note the model writes the "
"transcript in this language.",
)
parser.add_argument("--prompt", help="spellings or context to guide the transcript")
add_client_arguments(parser)
args = parser.parse_args()
client = make_client(args)
optional = {}
if args.language:
optional["language"] = args.language
if args.prompt:
optional["prompt"] = args.prompt
with args.file.open("rb") as audio:
transcription = client.audio.transcriptions.create(
model=args.model, file=audio, **optional
)
print(transcription.text)
# xVoice reports the billed duration the same way OpenAI does.
if transcription.usage is not None:
print(f"\n[{transcription.usage.seconds} s billed]")
if __name__ == "__main__":
run(main)stt_stream.py
Transcribe a file, printing the text as it arrives.
python
"""Speech to text, streamed: print the transcript as it is produced.
uv run stt_stream.py long-recording.mp3
The file is still uploaded in full first; streaming makes the text of long
recordings start arriving before the whole transcript is ready.
"""
import argparse
import time
from pathlib import Path
from client import add_client_arguments, make_client, run
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawTextHelpFormatter)
parser.add_argument("file", type=Path, help="audio file to transcribe")
parser.add_argument("--model", default="xvoice-stt1")
parser.add_argument("--language", help="ISO-639-1 code, e.g. en; omit to auto-detect")
add_client_arguments(parser)
args = parser.parse_args()
client = make_client(args)
optional = {"language": args.language} if args.language else {}
started = time.perf_counter()
with args.file.open("rb") as audio:
stream = client.audio.transcriptions.create(
model=args.model, file=audio, stream=True, **optional
)
# A failure after streaming began arrives as an error event, which the client
# raises as openai.APIError from this loop.
for event in stream:
if event.type == "transcript.text.delta":
print(event.delta, end="", flush=True)
elif event.type == "transcript.text.done":
elapsed = time.perf_counter() - started
print(f"\n\n[{event.usage.seconds} s billed, done after {elapsed:.2f}s]")
if __name__ == "__main__":
run(main)realtime.py
A live WebSocket session: audio in as it is captured, transcripts back while the speaker is still talking. The longest of the five, because it is the only one that has to send and receive at the same time.
python
"""Realtime speech to text: a live session over a WebSocket.
uv run realtime.py output/speech.wav
uv run realtime.py meeting.m4a --language en --segment 10
`stt_stream.py` uploads a finished recording and streams the text back. This opens a
session and feeds audio while it is still being produced, which is what a phone call
or a browser microphone does: the first words come back before the speaker has
finished.
Here ffmpeg stands in for the capture device -- it decodes the file to the wire
format and the script paces it at real speed. A live client sends the same bytes
straight from its microphone and needs no ffmpeg at all.
Two things differ from the file endpoints and catch people out:
* **There is no voice activity detection.** The server never decides that an
utterance has ended, so nothing comes back until you send
`input_audio_buffer.commit`. `--segment` controls how often this script does that.
* **The audio format is fixed:** 24 kHz mono PCM16, little endian, base64 inside
JSON. There is no container and no header, so there is nothing to negotiate.
"""
import argparse
import asyncio
import base64
import shutil
import sys
import time
from collections.abc import AsyncIterator
from contextlib import aclosing
from dataclasses import dataclass, field
from pathlib import Path
from client import add_client_arguments, make_async_client, run_async
# The wire format. These are protocol constants, not settings: the endpoint accepts
# nothing else, and a rate mismatch is rejected rather than resampled.
SAMPLE_RATE = 24_000
BYTES_PER_SAMPLE = 2
FRAME_MS = 20
FRAME_BYTES = SAMPLE_RATE * BYTES_PER_SAMPLE * FRAME_MS // 1000 # 960
TRANSCRIPTION = "conversation.item.input_audio_transcription"
class SessionError(Exception):
"""Something ended the session: a server `error` event, or audio we could not read."""
def closed(exc: BaseException) -> str | None:
"""Render a WebSocket close, if that is what this exception is.
The server always sends an `error` event before closing, and websockets delivers
queued frames before it raises, so reaching this means the connection dropped
without one: the deadline, a proxy, or the network.
Duck-typed on `websockets.exceptions.ConnectionClosed`, which is the SDK's
transport rather than part of its API.
"""
frame = getattr(exc, "rcvd", None) or getattr(exc, "sent", None)
if frame is None:
return None
return f"the server closed the session (code {frame.code}): {frame.reason or 'no reason given'}"
def explain(error: dict) -> str:
return (
f"{error.get('code', 'unknown')}: {error.get('message', '')} "
f"(request_id: {error.get('request_id')})"
)
# ----------------------------------------------------------------------- the audio
async def frames(path: Path) -> AsyncIterator[bytes]:
"""Decode any audio file to 20 ms frames of 24 kHz mono PCM16.
Frame size is a choice, not a requirement -- the server takes appends of any even
length up to 15 MB. 20 ms is what capture devices hand you and what keeps the
round trip short; larger frames trade latency for fewer messages.
"""
process = await asyncio.create_subprocess_exec(
"ffmpeg", "-v", "error", "-i", str(path),
"-f", "s16le", "-acodec", "pcm_s16le", "-ac", "1", "-ar", str(SAMPLE_RATE), "-",
stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.PIPE,
) # fmt: skip
assert process.stdout is not None
buffered = b""
try:
while chunk := await process.stdout.read(FRAME_BYTES * 16):
buffered += chunk
while len(buffered) >= FRAME_BYTES:
yield buffered[:FRAME_BYTES]
buffered = buffered[FRAME_BYTES:]
if buffered:
yield buffered # ffmpeg's output is always sample-aligned
_, errors = await process.communicate()
if process.returncode:
raise SessionError(f"ffmpeg could not read {path}:\n{errors.decode().strip()}")
finally:
if process.returncode is None:
process.kill()
# ------------------------------------------------------------------ the transcript
@dataclass
class Turn:
number: int
streamed: str = ""
transcript: str = ""
seconds: float = 0.0
@dataclass
class Session:
"""What has been said so far, and how much of it is still being transcribed.
Utterances are concurrent: committing frees the input buffer at once and the
engine keeps decoding in the background, so a long turn can finish after a short
one that followed it. Everything is therefore keyed by `item_id`, never held as a
single "current" turn.
"""
turns: dict[str, Turn] = field(default_factory=dict)
open_line: str | None = None
committed: int = 0
finished: int = 0
input_closed: bool = False
@property
def settled(self) -> bool:
return self.input_closed and self.finished >= self.committed
def began(self, item_id: str) -> Turn:
"""Start numbering a turn, whichever event mentioned it first.
Text streams while the buffer is still open, so the first delta usually
arrives before the `input_audio_buffer.committed` that announces the item.
Waiting for the announcement would mean printing nothing until the turn was
already over, which is the opposite of what a streaming session is for.
"""
turn = self.turns.get(item_id)
if turn is None:
turn = self.turns[item_id] = Turn(number=len(self.turns) + 1)
return turn
def delta(self, item_id: str, text: str) -> None:
"""Print deltas as they land, starting a new line when the turn changes.
Deltas are immutable and never revised, so they can go straight to the
terminal -- but two turns can be producing them at once, hence the tag.
"""
turn = self.began(item_id)
turn.streamed += text
if self.open_line != item_id:
self.end_line()
print(f"[{turn.number}] ", end="")
self.open_line = item_id
print(text, end="", flush=True)
def completed(self, item_id: str, transcript: str, seconds: float) -> None:
turn = self.began(item_id)
turn.transcript = transcript
turn.seconds = seconds
self.finished += 1
self.end_line()
# The deltas were only ever a preview; this is what the engine settled on.
# Usually the same text, so reprint the turn only when it is not -- the
# deltas carry their own spacing, hence the strip rather than ==.
if transcript.strip() != turn.streamed.strip():
print(f"[{turn.number}] {transcript}")
def failed(self, item_id: str, error: dict) -> None:
turn = self.began(item_id)
self.finished += 1
self.end_line()
print(f"[{turn.number}] failed -- {explain(error)}", file=sys.stderr)
def end_line(self) -> None:
if self.open_line is not None:
print()
self.open_line = None
async def receive(connection, session: Session) -> None:
"""Relay server events into `session` until every committed turn has settled."""
async for event in connection:
if event.type == "error":
raise SessionError(explain(event.error.to_dict()))
elif event.type == "input_audio_buffer.committed":
session.began(event.item_id)
elif event.type == f"{TRANSCRIPTION}.delta":
session.delta(event.item_id, event.delta)
elif event.type == f"{TRANSCRIPTION}.completed":
session.completed(
event.item_id, event.transcript, getattr(event.usage, "seconds", 0.0)
)
elif event.type == f"{TRANSCRIPTION}.failed":
session.failed(event.item_id, event.error.to_dict())
if session.settled:
return
async def send(
connection, session: Session, path: Path, *, segment: float, pace: bool
) -> float:
"""Feed the file in, committing a turn every `segment` seconds of audio."""
sent = 0
since_commit = 0
per_segment = int(segment * SAMPLE_RATE * BYTES_PER_SAMPLE)
started = time.perf_counter()
async with aclosing(frames(path)) as audio:
async for frame in audio:
await connection.send(
{
"type": "input_audio_buffer.append",
"audio": base64.b64encode(frame).decode(),
}
)
sent += len(frame)
since_commit += len(frame)
if per_segment and since_commit >= per_segment:
await connection.send({"type": "input_audio_buffer.commit"})
session.committed += 1
since_commit = 0
if pace:
# Hold the send rate to real time, the way a microphone would.
# Sleeping a flat 20 ms per frame would drift, so aim at where the
# clock ought to be by now.
spent = time.perf_counter() - started
behind = sent / (SAMPLE_RATE * BYTES_PER_SAMPLE) - spent
if behind > 0:
await asyncio.sleep(behind)
if since_commit:
await connection.send({"type": "input_audio_buffer.commit"})
session.committed += 1
session.input_closed = True
return sent / (SAMPLE_RATE * BYTES_PER_SAMPLE)
# ------------------------------------------------------------------------- the run
async def stream(connection, session: Session, args: argparse.Namespace) -> float:
"""Send audio and receive events at once.
They have to overlap: the first deltas of one turn arrive while the next is
still being spoken, and an `error` event has to cut the sender short rather than
be noticed once the whole file has gone out.
"""
sender = asyncio.create_task(
send(connection, session, args.file, segment=args.segment, pace=not args.fast)
)
listener = asyncio.create_task(receive(connection, session))
try:
await asyncio.wait({sender, listener}, return_when=asyncio.FIRST_COMPLETED)
if listener.done():
await listener # an error, or a session that settled before we noticed
audio_seconds = await sender
if not session.settled:
await asyncio.wait_for(asyncio.shield(listener), timeout=args.timeout)
return audio_seconds
except TimeoutError:
raise SessionError(
f"no transcript for the last turn within {args.timeout:g}s"
) from None
except Exception as exc:
reason = closed(exc)
if reason is None:
raise
raise SessionError(reason) from exc
finally:
session.end_line()
for task in (sender, listener):
task.cancel()
await asyncio.gather(sender, listener, return_exceptions=True)
async def main() -> None:
parser = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawTextHelpFormatter
)
parser.add_argument("file", type=Path, help="audio to stream, in any format ffmpeg reads")
parser.add_argument("--model", default="xvoice-stt1")
parser.add_argument("--language", help="ISO-639-1 code, e.g. en; omit to auto-detect")
parser.add_argument("--prompt", help="spellings or context to guide the transcript")
parser.add_argument(
"--segment",
type=float,
default=15.0,
help="commit a turn every N seconds of audio (default: 15). "
"0 sends the whole file as a single turn.",
)
parser.add_argument(
"--fast",
action="store_true",
help="send as fast as the socket allows instead of at real speed",
)
parser.add_argument(
"--timeout",
type=float,
default=60.0,
help="seconds to wait for the last turn after the audio ends (default: 60)",
)
add_client_arguments(parser)
args = parser.parse_args()
if not args.file.is_file():
sys.exit(f"No such file: {args.file}")
if shutil.which("ffmpeg") is None:
sys.exit(
"ffmpeg is not installed. It is only needed to turn a file into "
"microphone-shaped audio;\na live client already has 24 kHz mono PCM16 to hand."
)
transcription: dict[str, str] = {"model": args.model}
if args.language:
transcription["language"] = args.language
if args.prompt:
transcription["prompt"] = args.prompt
client = make_async_client(args)
session = Session()
started = time.perf_counter()
async with client.realtime.connect(
# No `model` query parameter: the model is named in session.update below, so
# one connection could serve several. `intent=transcription` is what selects
# this endpoint's behaviour, exactly as it does against OpenAI.
extra_query={"intent": "transcription"},
max_retries=0,
) as connection:
created = await connection.recv()
if created.type != "session.created":
raise SessionError(f"expected session.created, got {created.type}")
await connection.send(
{
"type": "session.update",
"session": {
"type": "transcription",
"audio": {
"input": {
"format": {"type": "audio/pcm", "rate": SAMPLE_RATE},
"transcription": transcription,
# Explicitly off. The server refuses any other value
# rather than quietly ignoring it.
"turn_detection": None,
}
},
},
}
)
updated = await connection.recv()
if updated.type == "error":
raise SessionError(explain(updated.error.to_dict()))
if updated.type != "session.updated":
raise SessionError(f"expected session.updated, got {updated.type}")
# The engine is attached and the session is billable from here.
audio_seconds = await stream(connection, session, args)
billed = sum(turn.seconds for turn in session.turns.values())
print(
f"\n[{len(session.turns)} turn(s), {audio_seconds:.1f} s of audio, "
f"{billed:.3f} s billed, {time.perf_counter() - started:.1f} s wall clock]"
)
if __name__ == "__main__":
try:
run_async(main)
except SessionError as exc:
sys.exit(str(exc))