Adapt playback speed to chunk audio boundary

Signed-off-by: John Tennant <johnmatthewtennant@gmail.com>
This commit is contained in:
John Tennant
2026-07-29 12:42:03 -04:00
committed by John Tennant
parent 752101dd2c
commit 8b87b09381
3 changed files with 6 additions and 34 deletions
+4 -11
View File
@@ -13,26 +13,19 @@ fn main() {
(2.0 * std::f32::consts::PI * 220.0 * index as f32 / SAMPLE_RATE as f32).sin() * 0.25
})
.collect();
let mut input = Vec::with_capacity(SENTENCE_LEAD_IN_SAMPLES + tone.len());
input.resize(SENTENCE_LEAD_IN_SAMPLES, 0.0);
input.extend_from_slice(&tone);
let compensated_lookahead = playback_speed_dsp::compensated_output_latency_samples(SAMPLE_RATE);
let compensated_lookahead_ms = compensated_lookahead as f64 * 1_000.0 / SAMPLE_RATE as f64;
let fixed_lead_in_ms = SENTENCE_LEAD_IN_SAMPLES as f64 * 1_000.0 / SAMPLE_RATE as f64;
for speed in [0.75_f32, 1.25, 1.5] {
let started = Instant::now();
let output = playback_speed_dsp::process_complete_chunk_preserving_lead_in(
&input,
SENTENCE_LEAD_IN_SAMPLES,
speed,
SAMPLE_RATE,
)
.expect("process production-shaped buffer");
let output = playback_speed_dsp::process_complete_chunk(&tone, speed, SAMPLE_RATE)
.expect("process synthesized model audio");
let elapsed = started.elapsed();
println!(
"{speed:.2}x: processing={:.2}ms, {:.3}% realtime, output={} samples, compensated \
lookahead={compensated_lookahead_ms:.1}ms",
lookahead={compensated_lookahead_ms:.1}ms, fixed playback lead-in={fixed_lead_in_ms:.1}ms",
elapsed.as_secs_f64() * 1_000.0,
elapsed.as_secs_f64() / TONE_SECONDS * 100.0,
output.len(),
@@ -21,12 +21,12 @@ use crate::app_state::AppState;
#[path = "playback_speed_dsp.rs"]
mod playback_speed_dsp;
pub(crate) use playback_speed_dsp::process_complete_chunk_preserving_lead_in;
pub(crate) use playback_speed_dsp::process_complete_chunk;
use playback_speed_dsp::{validate_speed, DEFAULT_PLAYBACK_SPEED};
const SETTINGS_FILE: &str = "tts-playback-settings.json";
#[cfg(test)]
use playback_speed_dsp::{process_complete_chunk, UNITY_EPSILON};
use playback_speed_dsp::UNITY_EPSILON;
/// Lock-free shared control read by the TTS worker before each synthesis chunk.
#[derive(Clone, Debug)]
@@ -71,27 +71,6 @@ pub fn process_complete_chunk(
Ok(output[0][start..end].to_vec())
}
/// Preserve a fixed device lead-in while pitch-preserving the remaining audio.
pub(crate) fn process_complete_chunk_preserving_lead_in(
input: &[f32],
fixed_lead_in_samples: usize,
speed: f32,
sample_rate: u32,
) -> Result<Vec<f32>, String> {
if fixed_lead_in_samples > input.len() {
return Err(format!(
"fixed lead-in of {fixed_lead_in_samples} samples exceeds input length {}",
input.len()
));
}
let processed = process_complete_chunk(&input[fixed_lead_in_samples..], speed, sample_rate)?;
let mut output = Vec::with_capacity(fixed_lead_in_samples + processed.len());
output.extend_from_slice(&input[..fixed_lead_in_samples]);
output.extend_from_slice(&processed);
Ok(output)
}
/// Return Signalsmith's compensated output lookahead for descriptive reporting.
#[allow(dead_code)]
pub(crate) fn compensated_output_latency_samples(sample_rate: u32) -> usize {