mirror of
https://github.com/block/buzz.git
synced 2026-08-18 06:50:31 +02:00
Preserve TTS lead-in through speed processing
Signed-off-by: John Tennant <jtennant@squareup.com>
This commit is contained in:
committed by
John Tennant
parent
613dd82171
commit
80899bab8d
@@ -1,29 +1,41 @@
|
||||
use std::time::Instant;
|
||||
|
||||
#[path = "../src/huddle/playback_speed_dsp.rs"]
|
||||
mod playback_speed_dsp;
|
||||
|
||||
const SAMPLE_RATE: u32 = 24_000;
|
||||
const SENTENCE_LEAD_IN_SAMPLES: usize = 480;
|
||||
const TONE_SECONDS: f64 = 10.0;
|
||||
|
||||
fn main() {
|
||||
let input: Vec<f32> = (0..SAMPLE_RATE * 10)
|
||||
let tone: Vec<f32> = (0..SAMPLE_RATE * TONE_SECONDS as u32)
|
||||
.map(|index| {
|
||||
(2.0 * std::f32::consts::PI * 220.0 * index as f32 / SAMPLE_RATE as f32).sin() * 0.25
|
||||
})
|
||||
.collect();
|
||||
let mut input = Vec::with_capacity(SENTENCE_LEAD_IN_SAMPLES + tone.len());
|
||||
input.resize(SENTENCE_LEAD_IN_SAMPLES, 0.0);
|
||||
input.extend_from_slice(&tone);
|
||||
|
||||
let compensated_lookahead = playback_speed_dsp::compensated_output_latency_samples(SAMPLE_RATE);
|
||||
let compensated_lookahead_ms = compensated_lookahead as f64 * 1_000.0 / SAMPLE_RATE as f64;
|
||||
|
||||
for speed in [0.75_f32, 1.25, 1.5] {
|
||||
let mut processor = ssstretch::Stretch::new();
|
||||
processor.preset_default(1, SAMPLE_RATE as f32);
|
||||
let output_len = (input.len() as f32 / speed).round() as usize;
|
||||
let inputs = [input.clone()];
|
||||
let mut outputs = [Vec::with_capacity(output_len)];
|
||||
let latency_ms = processor.output_latency() as f64 * 1_000.0 / SAMPLE_RATE as f64;
|
||||
let started = Instant::now();
|
||||
processor.process_vec(&inputs, input.len() as i32, &mut outputs, output_len as i32);
|
||||
let output = playback_speed_dsp::process_complete_chunk_preserving_lead_in(
|
||||
&input,
|
||||
SENTENCE_LEAD_IN_SAMPLES,
|
||||
speed,
|
||||
SAMPLE_RATE,
|
||||
)
|
||||
.expect("process production-shaped buffer");
|
||||
let elapsed = started.elapsed();
|
||||
println!(
|
||||
"{speed:.2}x: latency={latency_ms:.1}ms, CPU={:.2}ms for 10s ({:.3}% realtime), output={}",
|
||||
"{speed:.2}x: processing={:.2}ms, {:.3}% realtime, output={} samples, compensated \
|
||||
lookahead={compensated_lookahead_ms:.1}ms",
|
||||
elapsed.as_secs_f64() * 1_000.0,
|
||||
elapsed.as_secs_f64() / 10.0 * 100.0,
|
||||
outputs[0].len(),
|
||||
elapsed.as_secs_f64() / TONE_SECONDS * 100.0,
|
||||
output.len(),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -19,15 +19,18 @@ use tauri::{AppHandle, Manager, State};
|
||||
|
||||
use crate::app_state::AppState;
|
||||
|
||||
/// Slowest supported generated-speech playback speed.
|
||||
pub const MIN_PLAYBACK_SPEED: f32 = 0.75;
|
||||
/// Fastest supported generated-speech playback speed.
|
||||
pub const MAX_PLAYBACK_SPEED: f32 = 1.5;
|
||||
/// Default generated-speech playback speed.
|
||||
pub const DEFAULT_PLAYBACK_SPEED: f32 = 1.0;
|
||||
#[path = "playback_speed_dsp.rs"]
|
||||
mod playback_speed_dsp;
|
||||
pub(crate) use playback_speed_dsp::process_complete_chunk_preserving_lead_in;
|
||||
#[allow(unused_imports)]
|
||||
pub use playback_speed_dsp::{
|
||||
process_complete_chunk, validate_speed, DEFAULT_PLAYBACK_SPEED, MAX_PLAYBACK_SPEED,
|
||||
MIN_PLAYBACK_SPEED,
|
||||
};
|
||||
|
||||
const SETTINGS_FILE: &str = "tts-playback-settings.json";
|
||||
const UNITY_EPSILON: f32 = 0.000_1;
|
||||
#[cfg(test)]
|
||||
use playback_speed_dsp::UNITY_EPSILON;
|
||||
|
||||
/// Lock-free shared control read by the TTS worker before each synthesis chunk.
|
||||
#[derive(Clone, Debug)]
|
||||
@@ -152,79 +155,6 @@ fn processor_kind(speed: f32) -> ProcessorKind {
|
||||
}
|
||||
}
|
||||
|
||||
/// Pitch-preserve one already-buffered Pocket chunk.
|
||||
///
|
||||
/// The returned buffer has exactly `input.len() / speed` samples. Signalsmith
|
||||
/// starts a reset processor `input_latency` samples before the supplied audio
|
||||
/// and emits another `output_latency` samples of pre-roll. Both components are
|
||||
/// removed after draining, so the returned chunk retains its beginning and
|
||||
/// tail without buffering any later Pocket chunk.
|
||||
pub fn process_complete_chunk(
|
||||
input: &[f32],
|
||||
speed: f32,
|
||||
sample_rate: u32,
|
||||
) -> Result<Vec<f32>, String> {
|
||||
validate_speed(speed)?;
|
||||
if input.is_empty() || (speed - DEFAULT_PLAYBACK_SPEED).abs() <= UNITY_EPSILON {
|
||||
return Ok(input.to_vec());
|
||||
}
|
||||
|
||||
let expected = (input.len() as f64 / speed as f64).round() as usize;
|
||||
let mut stretch = ssstretch::Stretch::new();
|
||||
stretch.preset_default(1, sample_rate as f32);
|
||||
let input_latency = stretch.input_latency().max(0) as usize;
|
||||
let output_latency = stretch.output_latency().max(0) as usize;
|
||||
let reset_pre_roll = (input_latency as f64 / speed as f64).ceil() as usize;
|
||||
|
||||
let inputs = [input.to_vec()];
|
||||
let mut output = [Vec::with_capacity(expected)];
|
||||
stretch.process_vec(
|
||||
&inputs,
|
||||
i32_len(input.len())?,
|
||||
&mut output,
|
||||
i32_len(expected)?,
|
||||
);
|
||||
|
||||
let latency_input = [vec![0.0; input_latency]];
|
||||
let mut latency_output = [Vec::with_capacity(reset_pre_roll)];
|
||||
stretch.process_vec(
|
||||
&latency_input,
|
||||
i32_len(input_latency)?,
|
||||
&mut latency_output,
|
||||
i32_len(reset_pre_roll)?,
|
||||
);
|
||||
output[0].extend_from_slice(&latency_output[0]);
|
||||
|
||||
let mut flushed = [Vec::with_capacity(output_latency)];
|
||||
stretch.flush_vec(&mut flushed, i32_len(output_latency)?);
|
||||
output[0].extend_from_slice(&flushed[0]);
|
||||
|
||||
let start = reset_pre_roll.saturating_add(output_latency);
|
||||
let end = start.saturating_add(expected);
|
||||
if output[0].len() < end {
|
||||
return Err(format!(
|
||||
"time stretcher produced {} samples, need {end}",
|
||||
output[0].len()
|
||||
));
|
||||
}
|
||||
Ok(output[0][start..end].to_vec())
|
||||
}
|
||||
|
||||
/// Validate a generated-speech playback speed.
|
||||
pub fn validate_speed(speed: f32) -> Result<(), String> {
|
||||
if speed.is_finite() && (MIN_PLAYBACK_SPEED..=MAX_PLAYBACK_SPEED).contains(&speed) {
|
||||
Ok(())
|
||||
} else {
|
||||
Err(format!(
|
||||
"Speech playback speed must be between {MIN_PLAYBACK_SPEED} and {MAX_PLAYBACK_SPEED}"
|
||||
))
|
||||
}
|
||||
}
|
||||
|
||||
fn i32_len(length: usize) -> Result<i32, String> {
|
||||
i32::try_from(length).map_err(|_| "audio chunk is too large to process".to_string())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
@@ -0,0 +1,116 @@
|
||||
//! Pure pitch-preserving playback-speed processing.
|
||||
//!
|
||||
//! This module intentionally has no application, persistence, or filesystem
|
||||
//! dependencies so production playback and diagnostic examples can compile
|
||||
//! the exact same DSP boundary.
|
||||
|
||||
/// Slowest supported generated-speech playback speed.
|
||||
pub const MIN_PLAYBACK_SPEED: f32 = 0.75;
|
||||
/// Fastest supported generated-speech playback speed.
|
||||
pub const MAX_PLAYBACK_SPEED: f32 = 1.5;
|
||||
/// Default generated-speech playback speed.
|
||||
pub const DEFAULT_PLAYBACK_SPEED: f32 = 1.0;
|
||||
|
||||
pub(super) const UNITY_EPSILON: f32 = 0.000_1;
|
||||
|
||||
/// Pitch-preserve one already-buffered Pocket chunk.
|
||||
///
|
||||
/// The returned buffer has exactly `input.len() / speed` samples. Signalsmith
|
||||
/// starts a reset processor `input_latency` samples before the supplied audio
|
||||
/// and emits another `output_latency` samples of pre-roll. Both components are
|
||||
/// removed after draining, so the returned chunk retains its beginning and
|
||||
/// tail without buffering any later Pocket chunk.
|
||||
pub fn process_complete_chunk(
|
||||
input: &[f32],
|
||||
speed: f32,
|
||||
sample_rate: u32,
|
||||
) -> Result<Vec<f32>, String> {
|
||||
validate_speed(speed)?;
|
||||
if input.is_empty() || (speed - DEFAULT_PLAYBACK_SPEED).abs() <= UNITY_EPSILON {
|
||||
return Ok(input.to_vec());
|
||||
}
|
||||
|
||||
let expected = (input.len() as f64 / speed as f64).round() as usize;
|
||||
let mut stretch = ssstretch::Stretch::new();
|
||||
stretch.preset_default(1, sample_rate as f32);
|
||||
let input_latency = stretch.input_latency().max(0) as usize;
|
||||
let output_latency = stretch.output_latency().max(0) as usize;
|
||||
let reset_pre_roll = (input_latency as f64 / speed as f64).ceil() as usize;
|
||||
|
||||
let inputs = [input.to_vec()];
|
||||
let mut output = [Vec::with_capacity(expected)];
|
||||
stretch.process_vec(
|
||||
&inputs,
|
||||
i32_len(input.len())?,
|
||||
&mut output,
|
||||
i32_len(expected)?,
|
||||
);
|
||||
|
||||
let latency_input = [vec![0.0; input_latency]];
|
||||
let mut latency_output = [Vec::with_capacity(reset_pre_roll)];
|
||||
stretch.process_vec(
|
||||
&latency_input,
|
||||
i32_len(input_latency)?,
|
||||
&mut latency_output,
|
||||
i32_len(reset_pre_roll)?,
|
||||
);
|
||||
output[0].extend_from_slice(&latency_output[0]);
|
||||
|
||||
let mut flushed = [Vec::with_capacity(output_latency)];
|
||||
stretch.flush_vec(&mut flushed, i32_len(output_latency)?);
|
||||
output[0].extend_from_slice(&flushed[0]);
|
||||
|
||||
let start = reset_pre_roll.saturating_add(output_latency);
|
||||
let end = start.saturating_add(expected);
|
||||
if output[0].len() < end {
|
||||
return Err(format!(
|
||||
"time stretcher produced {} samples, need {end}",
|
||||
output[0].len()
|
||||
));
|
||||
}
|
||||
Ok(output[0][start..end].to_vec())
|
||||
}
|
||||
|
||||
/// Preserve a fixed device lead-in while pitch-preserving the remaining audio.
|
||||
pub(crate) fn process_complete_chunk_preserving_lead_in(
|
||||
input: &[f32],
|
||||
fixed_lead_in_samples: usize,
|
||||
speed: f32,
|
||||
sample_rate: u32,
|
||||
) -> Result<Vec<f32>, String> {
|
||||
if fixed_lead_in_samples > input.len() {
|
||||
return Err(format!(
|
||||
"fixed lead-in of {fixed_lead_in_samples} samples exceeds input length {}",
|
||||
input.len()
|
||||
));
|
||||
}
|
||||
|
||||
let processed = process_complete_chunk(&input[fixed_lead_in_samples..], speed, sample_rate)?;
|
||||
let mut output = Vec::with_capacity(fixed_lead_in_samples + processed.len());
|
||||
output.extend_from_slice(&input[..fixed_lead_in_samples]);
|
||||
output.extend_from_slice(&processed);
|
||||
Ok(output)
|
||||
}
|
||||
|
||||
/// Return Signalsmith's compensated output lookahead for descriptive reporting.
|
||||
#[allow(dead_code)]
|
||||
pub(crate) fn compensated_output_latency_samples(sample_rate: u32) -> usize {
|
||||
let mut stretch = ssstretch::Stretch::new();
|
||||
stretch.preset_default(1, sample_rate as f32);
|
||||
stretch.output_latency().max(0) as usize
|
||||
}
|
||||
|
||||
/// Validate a generated-speech playback speed.
|
||||
pub fn validate_speed(speed: f32) -> Result<(), String> {
|
||||
if speed.is_finite() && (MIN_PLAYBACK_SPEED..=MAX_PLAYBACK_SPEED).contains(&speed) {
|
||||
Ok(())
|
||||
} else {
|
||||
Err(format!(
|
||||
"Speech playback speed must be between {MIN_PLAYBACK_SPEED} and {MAX_PLAYBACK_SPEED}"
|
||||
))
|
||||
}
|
||||
}
|
||||
|
||||
fn i32_len(length: usize) -> Result<i32, String> {
|
||||
i32::try_from(length).map_err(|_| "audio chunk is too large to process".to_string())
|
||||
}
|
||||
@@ -885,6 +885,31 @@ fn sentence_append_buffer_is_one_contiguous_source() {
|
||||
);
|
||||
}
|
||||
|
||||
/// Regression guard: playback decoration happens after speed processing, so
|
||||
/// the fixed device cushion is never time-stretched.
|
||||
#[test]
|
||||
fn playback_speed_preserves_production_sentence_lead_in() {
|
||||
let processed = process_complete_chunk(&vec![0.5; 2_400], 1.5, SAMPLE_RATE)
|
||||
.expect("process synthesized model audio");
|
||||
let mut first = true;
|
||||
let processed =
|
||||
build_sentence_append_buffer(&mut first, processed, 2_400, true, true);
|
||||
|
||||
assert_eq!(SENTENCE_LEAD_IN_SAMPLES, 480);
|
||||
assert!(
|
||||
processed[..SENTENCE_LEAD_IN_SAMPLES]
|
||||
.iter()
|
||||
.all(|&sample| sample == 0.0),
|
||||
"the fixed device cushion must remain pure zero"
|
||||
);
|
||||
assert!(
|
||||
processed[SENTENCE_LEAD_IN_SAMPLES..]
|
||||
.iter()
|
||||
.any(|sample| sample.abs() > 0.1),
|
||||
"speech energy must remain after the fixed device cushion"
|
||||
);
|
||||
}
|
||||
|
||||
// ── clamp_to_full_scale tests ─────────────────────────────────────────────
|
||||
|
||||
/// In-range speech audio passes through bit-exact — no gain is applied.
|
||||
|
||||
Reference in New Issue
Block a user