use std::{ sync::{ Arc, Mutex, atomic::{AtomicBool, Ordering}, }, thread, }; use cpal::traits::{DeviceTrait, HostTrait, StreamTrait}; use realfft::{RealFftPlanner, num_complex}; pub struct AudioVisualizer { pub bar_data: Arc>>, pub stop_flag: Arc, } // This is a implementation largely lifted from these // open source implementation. I just wrote it in Rust here: // 1. CAVA (C): https://github.com/karlstav/cava // 2. cli-visualizer (C++): https://github.com/dpayne/cli-visualizer impl AudioVisualizer { pub fn new(num_bars: usize) -> Self { let bar_data = Arc::new(Mutex::new(vec![0; num_bars])); let bar_data_clone = Arc::clone(&bar_data); let stop_flag = Arc::new(AtomicBool::new(false)); let stop_clone = Arc::clone(&stop_flag); thread::spawn(move || { if let Err(e) = Self::run_audio_loop(bar_data_clone, num_bars, stop_clone) { eprintln!("Audio capture error: {:?}", e); } }); Self { bar_data, stop_flag, } } // Doing best effort to support cross-platform functionality fn get_audio_device( host: &cpal::Host, ) -> Result<(cpal::Device, cpal::StreamConfig), Box> { // Try windows WASAPI loopback on default output device #[cfg(target_os = "windows")] { if let Some(device) = host.default_output_device() && let Ok(config) = device.default_output_config() { return Ok((device, config.into())); } } // Try Linux PipeWire/PulseAudio output monitor device #[cfg(target_os = "linux")] { if let Ok(devices) = host.devices() { for dev in devices { if let Ok(desc) = dev.description() { let name = desc.to_string(); // Monitor devices mirror system output under PulseAudio/PipeWire if name.contains("monitor") { if let Ok(config) = dev.default_input_config() { return Ok((dev, config.into())); } } } } } } // Fallback to default input device let device = host .default_input_device() .or_else(|| host.default_output_device()) .ok_or("No audio input or output device found")?; let config = device .default_input_config() .or_else(|_| device.default_output_config())? .into(); Ok((device, config)) } // we here in doubling octaves but fft outputs linearly spaced bins. // so we bundle bins on a log scale from 20 hz to 12 kHz // this allows bass, mids, and trembls to have equal visual // proportions during display // // i.e. // bins like this [0-500Hz] [500-1k] [1k-1.5k] [1.5k-2k] [2k-2.5k] [2.5k-3k] [3k-12kHz] // vs // bins like this [20-60Hz] [60-250Hz] [250-500Hz] [500-2kHz] [2k-4kHz] [4k-8kHz] [8k-12kHz] fn build_log_bins(num_bars: usize, sample_rate: f32, chunk_size: usize) -> Vec<(usize, usize)> { let nyquist = sample_rate / 2.0; let max_hz = 12000.0f32; (0..num_bars) .map(|i| { let low_hz = 20.0 * (max_hz / 20.0).powf(i as f32 / num_bars as f32); let high_hz = 20.0 * (max_hz / 20.0).powf((i + 1) as f32 / num_bars as f32); let low = ((low_hz / nyquist) * (chunk_size as f32 / 2.0)) as usize; let high = ((high_hz / nyquist) * (chunk_size as f32 / 2.0)) as usize; (low.max(1), high.max(low + 1)) }) .collect() } // without we get sharp edges which gives noise in FFT processing fn apply_hann_window(samples: &[f32], input_buffer: &mut [f32]) { let chunk_size = samples.len(); for (i, sample) in samples.iter().enumerate() { let window = 0.5 * (1.0 - (2.0 * std::f32::consts::PI * i as f32 / chunk_size as f32).cos()); input_buffer[i] = sample * window; } } fn process_fft_magnitudes( spectrum: &[num_complex::Complex32], log_bins: &[(usize, usize)], freq_boost: &[f32], prev_heights: &[f32], autosens: f32, smoothing: f32, falloff: f32, ) -> Vec { let num_bars = log_bins.len(); let mut current_bars = vec![0.0f32; num_bars]; // determine magnitude for i in 0..num_bars { let (start, stop) = log_bins[i]; let bin_slice = &spectrum[start..stop.min(spectrum.len())]; let magnitude_sum: f32 = bin_slice.iter().map(|c| c.norm()).sum(); let avg_mag = if !bin_slice.is_empty() { magnitude_sum / bin_slice.len() as f32 } else { 0.0 }; // ignore quiet static noise below specific amplitude let raw_val = if avg_mag < 0.02 { 0.0 } else { (avg_mag * freq_boost[i] * autosens + 1.0).log10() * 3.5 }; // rise smoothly from previous height let target = (raw_val * (1.0 - smoothing)) + (prev_heights[i] * smoothing); // prevent sudden drops if target < prev_heights[i] { current_bars[i] = (prev_heights[i] - falloff).max(0.0); } else { current_bars[i] = target; } } current_bars } // blend the heights between neighboring freq bins // so instead of sharp spikes we get more of waves fn apply_monstercat_smoothing(bars: &[f32]) -> Vec { let num_bars = bars.len(); let mut smoothed = bars.to_vec(); for i in 1..(num_bars - 1) { smoothed[i] = (bars[i - 1] * 0.25) + (bars[i] * 0.50) + (bars[i + 1] * 0.25); } smoothed } // songs can be quiet and loud so we try to adjust // sensitivity to avoid bars becoming flattened or // clipped fn adjust_autosens(autosens: &mut f32, bars: &[f32]) { let max_val = bars.iter().copied().fold(0.0f32, f32::max); if max_val > 8.0 { *autosens *= 0.98; } else if max_val < 3.0 && *autosens < 3.0 { *autosens *= 1.01; } } // captures output and runs through pipeline // system audio -> audio buffer -> windowing -> fft process -> binning -> floor/autosens -> smoothing fn run_audio_loop( bar_data: Arc>>, num_bars: usize, stop: Arc, ) -> Result<(), Box> { let host = cpal::default_host(); let (device, config) = Self::get_audio_device(&host)?; let sample_rate = config.sample_rate as f32; let channels = config.channels as usize; let chunk_size = 2048; let mut planner = RealFftPlanner::::new(); let fft = planner.plan_fft_forward(chunk_size); let mut windowed_buffer = fft.make_input_vec(); let mut spectrum = fft.make_output_vec(); let mut prev_heights = vec![0.0f32; num_bars]; let smoothing = 0.70f32; let falloff = 0.08f32; let mut autosens = 1.0f32; let log_bins = Self::build_log_bins(num_bars, sample_rate, chunk_size); let freq_boost: Vec = (0..num_bars) .map(|i| 1.0 + (3.5 * (i as f32 / num_bars as f32).powf(1.2))) .collect(); let audio_buffer = Arc::new(Mutex::new(Vec::::with_capacity(chunk_size * 2))); let buffer_clone = Arc::clone(&audio_buffer); let stream = device.build_input_stream( config, move |data: &[f32], _| { if let Ok(mut buf) = buffer_clone.lock() { for chunk in data.chunks(channels) { let mono: f32 = chunk.iter().sum::() / channels as f32; buf.push(mono); } if buf.len() > chunk_size * 2 { let drain_amt = buf.len() - chunk_size; buf.drain(0..drain_amt); } } }, |err| eprintln!("Stream error: {}", err), None, )?; stream.play()?; while !stop.load(Ordering::Relaxed) { thread::sleep(std::time::Duration::from_millis(16)); let samples = { let mut buf = match audio_buffer.lock() { Ok(b) => b, Err(_) => continue, }; // decay heights smoothly if buf.len() < chunk_size { for val in prev_heights.iter_mut() { *val = (*val - falloff).max(0.0); } if let Ok(mut bars) = bar_data.lock() { for (i, val) in prev_heights.iter().enumerate() { bars[i] = ((*val * 10.0).clamp(0.0, 100.0)) as u64; } } continue; } // drain the consumed samples buf.drain(0..chunk_size).collect::>() }; Self::apply_hann_window(&samples, &mut windowed_buffer); if let Err(e) = fft.process(&mut windowed_buffer, &mut spectrum) { eprintln!("FFT error: {:?}", e); continue; } let raw_bars = Self::process_fft_magnitudes( &spectrum, &log_bins, &freq_boost, &prev_heights, autosens, smoothing, falloff, ); let smoothed_bars = Self::apply_monstercat_smoothing(&raw_bars); prev_heights = smoothed_bars.clone(); Self::adjust_autosens(&mut autosens, &smoothed_bars); if let Ok(mut bars) = bar_data.lock() { for (i, val) in smoothed_bars.iter().enumerate() { bars[i] = ((*val * 10.0).clamp(0.0, 100.0)) as u64; } } } Ok(()) } } impl Drop for AudioVisualizer { fn drop(&mut self) { self.stop_flag.store(true, Ordering::Relaxed); } }