vois/examples/latency_probe.rs
loki5512344 b0d685a110 Initial release: real-time voice changer in pure Rust
Phase-vocoder pitch/formant shifting, 12 presets, effect chain,
noise gate, WAV recording, full-screen TUI (arrow/mouse control)
and a built-in PipeWire/PulseAudio virtual mic (vois.rs).
GPL-3.0-or-later.
2026-08-02 17:54:50 +02:00

217 lines
7.4 KiB
Rust

//! Latency + throughput probe for the vocoder DSP.
//!
//! This is a dev-only example, not part of the app. `vois` is a binary-only
//! crate (DSP lives behind private `mod dsp` in `main.rs`), so we pull in the
//! real DSP sources with `#[path]` and compile them into this example. No new
//! dependencies are needed: `rustfft` and `anyhow` are already in Cargo.toml.
// The example only exercises `Processor` + `PitchShifter`; the rest of the
// worker/Control surface compiled in from `dsp::chain` is intentionally unused
// here, so silence dead-code warnings for this example target only.
#![allow(dead_code)]
use std::sync::{Arc, Mutex};
use std::time::Instant;
use anyhow::Result;
/// Minimal stand-in for `crate::audio::record` so `dsp::chain` compiles here.
mod audio {
pub mod record {
use std::sync::mpsc;
use std::thread;
pub fn spawn_writer(
_path: String,
_fs: u32,
) -> (mpsc::Sender<Vec<f32>>, thread::JoinHandle<()>) {
let (tx, rx) = mpsc::channel();
let handle = thread::spawn(move || while rx.recv().is_ok() {});
(tx, handle)
}
}
}
#[path = "../src/dsp/mod.rs"]
mod dsp;
use dsp::chain::{Control, Processor};
use dsp::pitch::PitchShifter;
use dsp::stats::Stats;
const FS: u32 = 48000;
/// Harmonic-rich saw (like the unit tests use): narrow pitch lines + broad
/// envelope, closer to a voice than a pure sine.
fn saw(freq: f32, fs: u32, samples: usize) -> Vec<f32> {
(0..samples)
.map(|i| {
let t = freq * i as f32 / fs as f32;
let mut s = 0.0;
for h in 1..=12 {
s += ((2.0 * std::f32::consts::PI * h as f32 * t).sin()) / h as f32;
}
s
})
.collect()
}
/// Feed `input` through the processor in the same fixed-size blocks the DSP
/// worker uses, collecting the output.
fn feed_blocks(p: &mut Processor, input: &[f32], block: usize) -> Vec<f32> {
let mut out_all = Vec::with_capacity(input.len());
for chunk in input.chunks(block) {
let mut b = chunk.to_vec();
b.resize(block, 0.0);
p.process_block(&mut b);
out_all.extend_from_slice(&b[..chunk.len()]);
}
out_all
}
/// CPU time needed to process one second of audio (in ms), measured over 10 s
/// of a 220 Hz saw fed in `block`-sized chunks. Preset "Girl / Anime" (pitch
/// +5 st, formant x1.25) exercises the heaviest vocoder path.
fn throughput(block: usize) -> f64 {
let control = Arc::new(Mutex::new(Control {
preset_idx: 1,
..Control::default()
}));
let mut p = Processor::new(FS, control, Arc::new(Mutex::new(Stats::default())));
let audio = saw(220.0, FS, 10 * FS as usize);
let t0 = Instant::now();
let out = feed_blocks(&mut p, &audio, block);
let dt = t0.elapsed();
assert_eq!(out.len(), audio.len());
dt.as_secs_f64() / 10.0 * 1000.0
}
/// Pitch-shifter group delay: feed a sharp transient, report the first output
/// sample above `thresh_frac` of the output peak. A unit impulse at sample 0
/// is annihilated by the Hann window (w[0] == 0 exactly), so the transient is
/// placed mid-frame where the window is at its peak.
fn group_delay(n: usize, hop: usize, impulse_at: usize, thresh_frac: f32) -> Option<(usize, f32)> {
let mut sh = PitchShifter::new(n, hop, FS);
sh.pitch_ratio = 1.0;
let mut input = vec![0.0f32; 2 * n];
input[impulse_at] = 1.0;
let mut out = Vec::new();
sh.process(&input, &mut out);
let peak = out.iter().fold(0.0f32, |a, &v| a.max(v.abs()));
let thr = peak * thresh_frac;
out.iter().position(|&v| v.abs() > thr).map(|i| (i, peak))
}
/// Practical onset delay through the whole chain (what a user would hear):
/// feed a sine burst that starts after a quiet lead-in and measure the gap
/// between the input onset and the first output sample above a threshold.
fn chain_onset_delay(block: usize, n: usize) -> Option<usize> {
let control = Arc::new(Mutex::new(Control {
preset_idx: 0,
pitch_delta: 5.0,
..Control::default()
}));
let mut p = Processor::new(FS, control, Arc::new(Mutex::new(Stats::default())));
let lead = 2 * n;
let burst_len = 2048;
let mut input = vec![0.0f32; lead + burst_len];
for (i, s) in input.iter_mut().enumerate().skip(lead) {
let t = 220.0 * (i - lead) as f32 / FS as f32;
*s = (2.0 * std::f32::consts::PI * t).sin();
}
let out = feed_blocks(&mut p, &input, block);
let peak = out[lead..].iter().fold(0.0f32, |a, &v| a.max(v.abs()));
let thr = peak * 1e-2;
out.iter().position(|&v| v.abs() > thr).map(|i| i - lead)
}
/// Sanity check: feeding the same audio at block=1024 vs block=256 must
/// produce (nearly) identical output — the block-size change only reduces
/// buffering latency, not audio content.
fn block_size_identity() {
let saw2 = saw(220.0, FS, 2 * FS as usize);
let mk = || {
Arc::new(Mutex::new(Control {
preset_idx: 1,
..Control::default()
}))
};
let mut a = Processor::new(FS, mk(), Arc::new(Mutex::new(Stats::default())));
let mut b = Processor::new(FS, mk(), Arc::new(Mutex::new(Stats::default())));
let oa = feed_blocks(&mut a, &saw2, 1024);
let ob = feed_blocks(&mut b, &saw2, 256);
let max_diff = oa
.iter()
.zip(ob.iter())
.map(|(x, y)| (x - y).abs())
.fold(0.0f32, f32::max);
println!("block 1024 vs 256 output: max |diff| = {max_diff:.3e}");
}
fn ms(samples: usize) -> f64 {
samples as f64 / FS as f64 * 1000.0
}
fn main() -> Result<()> {
let audio_seconds = 10.0;
let (n, hop) = (1024usize, 256usize);
println!("=== vois DSP latency probe (fs = {FS} Hz) ===");
// a) Throughput.
for block in [1024usize, 256usize] {
let cpu_ms = throughput(block);
println!(
"throughput block={block:>4}: {cpu_ms:7.2} ms CPU per 1s audio \
({:5.1}% of one core, {:.1}x realtime)",
cpu_ms / 10.0,
audio_seconds / (cpu_ms / 1000.0)
);
}
// b) Pitch-shifter group delay (impulse mid-frame, ratio = 1.0).
let (gd, peak) = group_delay(n, hop, n / 2, 0.1).expect("no impulse response found");
println!(
"shifter group delay (n={n}, hop={hop}, impulse@{}, >10% of peak {:.2e}): \
{gd} samples = {:.2} ms",
n / 2,
peak,
ms(gd)
);
// Frame-fill latency: the shifter needs `n` samples before it can emit its
// first hop, i.e. output lags input by n - hop in the stream.
println!(
" streaming frame-fill latency n - hop = {} samples = {:.2} ms",
n - hop,
ms(n - hop)
);
// c) End-to-end estimate = measured group delay + fixed prime + block
// buffering, for the current settings.
let prime = n - hop;
for block in [1024usize, 256usize] {
let total = gd + prime + block;
println!(
"end-to-end (block={block:>4}): group delay {gd} + prime {prime} + buffering {block} \
= {total} samples = {ms:.2} ms",
ms = ms(total)
);
}
// Chain onset delay is the number users would actually feel.
for block in [1024usize, 256usize] {
if let Some(d) = chain_onset_delay(block, n) {
println!(
"chain onset delay (block={block:>4}): {d} samples = {:.2} ms after burst start",
ms(d)
);
}
}
block_size_identity();
Ok(())
}