From 1fefd4d4b1943403bd29a555287290f90f0e4ee5 Mon Sep 17 00:00:00 2001 From: will wade Date: Sun, 16 Aug 2026 15:55:16 +0000 Subject: [PATCH] docs+chore: make streaming claims precise everywhere The PR #15 README edit collided with the PR #14 wording, duplicating the estimated-boundaries sentence; the engine table still said 'Chunked' uniformly, which is both uninformative and wrong for Google/ElevenLabs-with-timestamps (JSON base64, post-response) and understated Azure/Edge (real-time WS) and sherpa (sentence batches). - README table: Streaming column now distinguishes 'Real-time (WS)', 'Streamed' (as bytes arrive), 'After response (JSON)', and 'Sentence batches' (sherpa) - README streaming bullet: de-duplicated; one accurate paragraph - sherpaonnx_engine: stale 'synthesises the whole clip up-front' doc comments on STREAMING_CHUNK_SIZE/deliver_pcm updated; deliver_pcm removed (dead since the streaming path) and the generate callback now re-chunks batches to the documented 8 KB multi-callback shape instead of emitting one large buffer per sentence --- README.md | 38 +++++++++++++++++++------------------- src/sherpaonnx_engine.rs | 32 ++++++++++---------------------- 2 files changed, 29 insertions(+), 41 deletions(-) diff --git a/README.md b/README.md index 8586b6c..e26004e 100644 --- a/README.md +++ b/README.md @@ -6,30 +6,30 @@ Cross-platform TTS (Text-to-Speech) wrapper with C ABI. Mirrors [js-tts-wrapper] | Engine | Type | Credentials | Streaming | Voice List | Word Boundaries | Speech Markdown | |--------|------|-------------|-----------|------------|-----------------|-----------------| -| System (speech-dispatcher) | Local | None | — | — | Estimated | — | -| Sherpa-ONNX | Local (1300+ models) | None | Chunked | Speakers | Estimated | — | -| Azure | Cloud | Key + Region | Chunked | API | **Real** (WS) | Platform-aware | -| Microsoft Edge (Read Aloud) | Cloud | **None** (free) | Chunked | API | **Real** (WS) | Platform-aware | -| Google Cloud | Cloud | API Key | Chunked | API | **Real** (v1beta1 timepoints) | Platform-aware | -| OpenAI | Cloud | API Key | Chunked | — | Estimated | Platform-aware | -| ElevenLabs | Cloud | API Key | Chunked | API | Estimated | Platform-aware | -| Cartesia | Cloud | API Key | Chunked | API | Estimated | Platform-aware | -| Deepgram | Cloud | API Key | Chunked | — | Estimated | Platform-aware | -| PlayHT | Cloud | API Key + User ID | Chunked | — | Estimated | Platform-aware | -| Fish Audio | Cloud | API Key | Chunked | — | Estimated | Platform-aware | -| Hume AI | Cloud | API Key | Chunked | — | Estimated | Platform-aware | -| Mistral | Cloud | API Key | Chunked | — | Estimated | Platform-aware | -| Murf | Cloud | API Key | Chunked | — | Estimated | Platform-aware | -| Resemble AI | Cloud | API Key | Chunked | — | Estimated | Platform-aware | -| Unreal Speech | Cloud | API Key | Chunked | — | Estimated | Platform-aware | -| UpliftAI | Cloud | API Key | Chunked | — | Estimated | Platform-aware | -| Amazon Polly | Cloud | Key + Secret + Region | Chunked | — | Estimated | Platform-aware | +| System (speech-dispatcher) | Local | None | — (daemon plays) | — | Estimated | — | +| Sherpa-ONNX | Local (1300+ models) | None | Sentence batches | Speakers | Estimated | — | +| Azure | Cloud | Key + Region | Real-time (WS) / Streamed (REST) | API | **Real** (WS) | Platform-aware | +| Microsoft Edge (Read Aloud) | Cloud | **None** (free) | Real-time (WS) | API | **Real** (WS) | Platform-aware | +| Google Cloud | Cloud | API Key | After response (JSON) | API | **Real** (v1beta1 timepoints) | Platform-aware | +| OpenAI | Cloud | API Key | Streamed | — | Estimated | Platform-aware | +| ElevenLabs | Cloud | API Key | Streamed (JSON w/ timestamps) | API | Estimated | Platform-aware | +| Cartesia | Cloud | API Key | Streamed | API | Estimated | Platform-aware | +| Deepgram | Cloud | API Key | Streamed | — | Estimated | Platform-aware | +| PlayHT | Cloud | API Key + User ID | Streamed | — | Estimated | Platform-aware | +| Fish Audio | Cloud | API Key | Streamed | — | Estimated | Platform-aware | +| Hume AI | Cloud | API Key | Streamed | — | Estimated | Platform-aware | +| Mistral | Cloud | API Key | Streamed | — | Estimated | Platform-aware | +| Murf | Cloud | API Key | Streamed | — | Estimated | Platform-aware | +| Resemble AI | Cloud | API Key | Streamed | — | Estimated | Platform-aware | +| Unreal Speech | Cloud | API Key | Streamed | — | Estimated | Platform-aware | +| UpliftAI | Cloud | API Key | Streamed | — | Estimated | Platform-aware | +| Amazon Polly | Cloud | Key + Secret + Region | Streamed | — | Estimated | Platform-aware | | IBM Watson | Cloud | Key + Region + Instance | Chunked | — | Estimated | Platform-aware | | Wit.ai | Cloud | Token | Chunked | — | Estimated | Platform-aware | | xAI | Cloud | API Key | Chunked | — | Estimated | Platform-aware | | ModelsLab | Cloud | API Key | Chunked | — | Estimated | Platform-aware | -- **Streaming**: Audio is delivered through the `on_audio` callback in chunks. REST engines and Edge stream as bytes arrive over the network (MP3 is decoded to PCM16 mono incrementally, on a background reader thread); Azure's WebSocket delivers PCM frames per message. Engines whose APIs return a single JSON document with base64 audio (Google, ElevenLabs `with-timestamps`) deliver only once the response completes — an API limitation. Sherpa-ONNX delivers each sentence batch as it is synthesised (sentence-level streaming via the generate progress callback; single-sentence utterances still complete before delivery). Estimated word boundaries (engines without API timing data) fire progressively during streaming, anchored to delivered audio, rather than all at once when the response completes. Estimated word boundaries (engines without API timing data) fire progressively during streaming, anchored to delivered audio, rather than all at once when the response completes. +- **Streaming**: Audio is delivered through the `on_audio` callback in chunks, as it becomes available. REST engines stream the response body as bytes arrive over the network (MP3 decoded to PCM16 mono incrementally on a background reader thread; raw-PCM providers pass straight through); Azure and Edge deliver real-time over WebSockets; Sherpa-ONNX delivers each sentence batch as it is synthesised (via the generate progress callback — a single-sentence utterance still completes before delivery). Exceptions: Google and ElevenLabs `with-timestamps` return one JSON document with base64 audio, so they can only deliver after the response completes (an API limitation, not buffering). Estimated word boundaries (engines without API timing data) fire progressively during streaming, anchored to delivered audio, rather than all at once when the response completes. - **Native engine varies by platform**: the table shows `system` (Linux speech-dispatcher); macOS uses `avsynth` (AVSpeechSynthesizer) and Windows uses `sapi`. "22 total" counts one native engine + Sherpa-ONNX + the 20 cloud engines, per platform. ## Formatting & Testing diff --git a/src/sherpaonnx_engine.rs b/src/sherpaonnx_engine.rs index 9e9b024..9ea71fd 100644 --- a/src/sherpaonnx_engine.rs +++ b/src/sherpaonnx_engine.rs @@ -33,10 +33,11 @@ thread_local! { const { std::cell::RefCell::new(None) }; } -/// PCM delivery chunk size. Sherpa-ONNX synthesises the whole clip up-front, -/// so we slice the rendered PCM into 8 KB chunks before pushing them through -/// `on_audio` — matching the cloud engines' streamed-chunk shape so callers -/// see the same multi-callback delivery instead of one giant buffer. +/// PCM delivery chunk size. Sentence batches arrive whole from the +/// generate progress callback and larger ones are sliced into 8 KB chunks +/// before pushing them through `on_audio` — matching the cloud engines' +/// streamed-chunk shape so callers see the same multi-callback delivery +/// instead of one giant buffer. const STREAMING_CHUNK_SIZE: usize = 8 * 1024; /// Maps a 2-letter ISO 639-1 code to its 3-letter ISO 639-3 equivalent for the @@ -551,7 +552,11 @@ impl TtsEngine for SherpaOnnxEngine { STREAM_AUDIO_CB.with(|c| { if let Some(ptr) = *c.borrow() { // SAFETY (stash): see the thread-local docs. - unsafe { (*ptr)(&pcm) }; + // Chunk to keep the documented 8 KB multi- + // callback delivery shape. + for chunk in pcm.chunks(STREAMING_CHUNK_SIZE) { + unsafe { (*ptr)(chunk) }; + } } }); if let Some(f) = firer_for_cb.as_ref() { @@ -742,23 +747,6 @@ fn samples_to_le_bytes(samples: &[f32]) -> Vec { pcm } -/// Scale `samples` by `volume_factor`, convert to little-endian PCM16 bytes, -/// and push them through `cb` in `STREAMING_CHUNK_SIZE`-byte chunks. Volume -/// and pitch are applied to the full buffer by `apply_volume_and_pitch` -/// before this runs, so callers pass `1.0` here. -#[allow(clippy::cast_possible_truncation, clippy::cast_precision_loss)] -fn deliver_pcm(cb: &mut dyn FnMut(&[u8]), samples: &[f32], volume_factor: f32) { - let mut pcm = Vec::with_capacity(samples.len() * 2); - for &s in samples { - let scaled = (s * volume_factor).clamp(-1.0, 1.0); - let s16 = (scaled * 32767.0) as i16; - pcm.extend_from_slice(&s16.to_le_bytes()); - } - for chunk in pcm.chunks(STREAMING_CHUNK_SIZE) { - cb(chunk); - } -} - /// Write a 16-bit PCM mono WAV file. Returns `false` on I/O error. fn write_wav(path: &std::path::Path, samples: &[f32], sample_rate: i32) -> bool { use std::io::Write;