refactor: extract audio visualizer

This commit is contained in:
Stevan Freeborn
2026-07-28 09:32:16 -05:00
parent 6acc79a5d1
commit 19a3d0927e
+298
View File
@@ -0,0 +1,298 @@
use std::{
sync::{
Arc, Mutex,
atomic::{AtomicBool, Ordering},
},
thread,
};
use cpal::traits::{DeviceTrait, HostTrait, StreamTrait};
use realfft::{RealFftPlanner, num_complex};
pub struct AudioVisualizer {
pub bar_data: Arc<Mutex<Vec<u64>>>,
pub stop_flag: Arc<AtomicBool>,
}
// This is a implementation largely lifted from these
// open source implementation. I just wrote it in Rust here:
// 1. CAVA (C): https://github.com/karlstav/cava
// 2. cli-visualizer (C++): https://github.com/dpayne/cli-visualizer
impl AudioVisualizer {
pub fn new(num_bars: usize) -> Self {
let bar_data = Arc::new(Mutex::new(vec![0; num_bars]));
let bar_data_clone = Arc::clone(&bar_data);
let stop_flag = Arc::new(AtomicBool::new(false));
let stop_clone = Arc::clone(&stop_flag);
thread::spawn(move || {
if let Err(e) = Self::run_audio_loop(bar_data_clone, num_bars, stop_clone) {
eprintln!("Audio capture error: {:?}", e);
}
});
Self {
bar_data,
stop_flag,
}
}
// Doing best effort to support cross-platform functionality
fn get_audio_device(
host: &cpal::Host,
) -> Result<(cpal::Device, cpal::StreamConfig), Box<dyn std::error::Error>> {
// Try windows WASAPI loopback on default output device
#[cfg(target_os = "windows")]
{
if let Some(device) = host.default_output_device()
&& let Ok(config) = device.default_output_config()
{
return Ok((device, config.into()));
}
}
// Try Linux PipeWire/PulseAudio output monitor device
#[cfg(target_os = "linux")]
{
if let Ok(devices) = host.devices() {
for dev in devices {
if let Ok(desc) = dev.description() {
let name = desc.to_string();
// Monitor devices mirror system output under PulseAudio/PipeWire
if name.contains("monitor") {
if let Ok(config) = dev.default_input_config() {
return Ok((dev, config.into()));
}
}
}
}
}
}
// Fallback to default input device
let device = host
.default_input_device()
.or_else(|| host.default_output_device())
.ok_or("No audio input or output device found")?;
let config = device
.default_input_config()
.or_else(|_| device.default_output_config())?
.into();
Ok((device, config))
}
// we here in doubling octaves but fft outputs linearly spaced bins.
// so we bundle bins on a log scale from 20 hz to 12 kHz
// this allows bass, mids, and trembls to have equal visual
// proportions during display
//
// i.e.
// bins like this [0-500Hz] [500-1k] [1k-1.5k] [1.5k-2k] [2k-2.5k] [2.5k-3k] [3k-12kHz]
// vs
// bins like this [20-60Hz] [60-250Hz] [250-500Hz] [500-2kHz] [2k-4kHz] [4k-8kHz] [8k-12kHz]
fn build_log_bins(num_bars: usize, sample_rate: f32, chunk_size: usize) -> Vec<(usize, usize)> {
let nyquist = sample_rate / 2.0;
let max_hz = 12000.0f32;
(0..num_bars)
.map(|i| {
let low_hz = 20.0 * (max_hz / 20.0).powf(i as f32 / num_bars as f32);
let high_hz = 20.0 * (max_hz / 20.0).powf((i + 1) as f32 / num_bars as f32);
let low = ((low_hz / nyquist) * (chunk_size as f32 / 2.0)) as usize;
let high = ((high_hz / nyquist) * (chunk_size as f32 / 2.0)) as usize;
(low.max(1), high.max(low + 1))
})
.collect()
}
// without we get sharp edges which gives noise in FFT processing
fn apply_hann_window(samples: &[f32], input_buffer: &mut [f32]) {
let chunk_size = samples.len();
for (i, sample) in samples.iter().enumerate() {
let window = 0.5 * (1.0 - (2.0 * std::f32::consts::PI * i as f32 / chunk_size as f32).cos());
input_buffer[i] = sample * window;
}
}
fn process_fft_magnitudes(
spectrum: &[num_complex::Complex32],
log_bins: &[(usize, usize)],
freq_boost: &[f32],
prev_heights: &[f32],
autosens: f32,
smoothing: f32,
falloff: f32,
) -> Vec<f32> {
let num_bars = log_bins.len();
let mut current_bars = vec![0.0f32; num_bars];
// determine magnitude
for i in 0..num_bars {
let (start, stop) = log_bins[i];
let bin_slice = &spectrum[start..stop.min(spectrum.len())];
let magnitude_sum: f32 = bin_slice.iter().map(|c| c.norm()).sum();
let avg_mag = if !bin_slice.is_empty() {
magnitude_sum / bin_slice.len() as f32
} else {
0.0
};
// ignore quiet static noise below specific amplitude
let raw_val = if avg_mag < 0.02 {
0.0
} else {
(avg_mag * freq_boost[i] * autosens + 1.0).log10() * 3.5
};
// rise smoothly from previous height
let target = (raw_val * (1.0 - smoothing)) + (prev_heights[i] * smoothing);
// prevent sudden drops
if target < prev_heights[i] {
current_bars[i] = (prev_heights[i] - falloff).max(0.0);
} else {
current_bars[i] = target;
}
}
current_bars
}
// blend the heights between neighboring freq bins
// so instead of sharp spikes we get more of waves
fn apply_monstercat_smoothing(bars: &[f32]) -> Vec<f32> {
let num_bars = bars.len();
let mut smoothed = bars.to_vec();
for i in 1..(num_bars - 1) {
smoothed[i] = (bars[i - 1] * 0.25) + (bars[i] * 0.50) + (bars[i + 1] * 0.25);
}
smoothed
}
// songs can be quiet and loud so we try to adjust
// sensitivity to avoid bars becoming flattened or
// clipped
fn adjust_autosens(autosens: &mut f32, bars: &[f32]) {
let max_val = bars.iter().copied().fold(0.0f32, f32::max);
if max_val > 8.0 {
*autosens *= 0.98;
} else if max_val < 3.0 && *autosens < 3.0 {
*autosens *= 1.01;
}
}
// captures output and runs through pipeline
// system audio -> audio buffer -> windowing -> fft process -> binning -> floor/autosens -> smoothing
fn run_audio_loop(
bar_data: Arc<Mutex<Vec<u64>>>,
num_bars: usize,
stop: Arc<AtomicBool>,
) -> Result<(), Box<dyn std::error::Error>> {
let host = cpal::default_host();
let (device, config) = Self::get_audio_device(&host)?;
let sample_rate = config.sample_rate as f32;
let channels = config.channels as usize;
let chunk_size = 2048;
let mut planner = RealFftPlanner::<f32>::new();
let fft = planner.plan_fft_forward(chunk_size);
let mut windowed_buffer = fft.make_input_vec();
let mut spectrum = fft.make_output_vec();
let mut prev_heights = vec![0.0f32; num_bars];
let smoothing = 0.70f32;
let falloff = 0.08f32;
let mut autosens = 1.0f32;
let log_bins = Self::build_log_bins(num_bars, sample_rate, chunk_size);
let freq_boost: Vec<f32> = (0..num_bars)
.map(|i| 1.0 + (3.5 * (i as f32 / num_bars as f32).powf(1.2)))
.collect();
let audio_buffer = Arc::new(Mutex::new(Vec::<f32>::with_capacity(chunk_size * 2)));
let buffer_clone = Arc::clone(&audio_buffer);
let stream = device.build_input_stream(
config,
move |data: &[f32], _| {
if let Ok(mut buf) = buffer_clone.lock() {
for chunk in data.chunks(channels) {
let mono: f32 = chunk.iter().sum::<f32>() / channels as f32;
buf.push(mono);
}
if buf.len() > chunk_size * 2 {
let drain_amt = buf.len() - chunk_size;
buf.drain(0..drain_amt);
}
}
},
|err| eprintln!("Stream error: {}", err),
None,
)?;
stream.play()?;
while !stop.load(Ordering::Relaxed) {
thread::sleep(std::time::Duration::from_millis(16));
let samples = {
let buf = match audio_buffer.lock() {
Ok(b) => b,
Err(_) => continue,
};
if buf.len() < chunk_size {
continue;
}
buf[buf.len() - chunk_size..].to_vec()
};
Self::apply_hann_window(&samples, &mut windowed_buffer);
if let Err(e) = fft.process(&mut windowed_buffer, &mut spectrum) {
eprintln!("FFT error: {:?}", e);
continue;
}
let raw_bars = Self::process_fft_magnitudes(
&spectrum,
&log_bins,
&freq_boost,
&prev_heights,
autosens,
smoothing,
falloff,
);
let smoothed_bars = Self::apply_monstercat_smoothing(&raw_bars);
prev_heights = smoothed_bars.clone();
Self::adjust_autosens(&mut autosens, &smoothed_bars);
if let Ok(mut bars) = bar_data.lock() {
for (i, val) in smoothed_bars.iter().enumerate() {
bars[i] = ((*val * 10.0).clamp(0.0, 100.0)) as u64;
}
}
}
Ok(())
}
}
impl Drop for AudioVisualizer {
fn drop(&mut self) {
self.stop_flag.store(true, Ordering::Relaxed);
}
}