diff --git a/src/audio.rs b/src/audio.rs new file mode 100644 index 0000000..74761d3 --- /dev/null +++ b/src/audio.rs @@ -0,0 +1,298 @@ +use std::{ + sync::{ + Arc, Mutex, + atomic::{AtomicBool, Ordering}, + }, + thread, +}; + +use cpal::traits::{DeviceTrait, HostTrait, StreamTrait}; +use realfft::{RealFftPlanner, num_complex}; + +pub struct AudioVisualizer { + pub bar_data: Arc>>, + pub stop_flag: Arc, +} + +// This is a implementation largely lifted from these +// open source implementation. I just wrote it in Rust here: +// 1. CAVA (C): https://github.com/karlstav/cava +// 2. cli-visualizer (C++): https://github.com/dpayne/cli-visualizer +impl AudioVisualizer { + pub fn new(num_bars: usize) -> Self { + let bar_data = Arc::new(Mutex::new(vec![0; num_bars])); + let bar_data_clone = Arc::clone(&bar_data); + let stop_flag = Arc::new(AtomicBool::new(false)); + let stop_clone = Arc::clone(&stop_flag); + + thread::spawn(move || { + if let Err(e) = Self::run_audio_loop(bar_data_clone, num_bars, stop_clone) { + eprintln!("Audio capture error: {:?}", e); + } + }); + + Self { + bar_data, + stop_flag, + } + } + + // Doing best effort to support cross-platform functionality + fn get_audio_device( + host: &cpal::Host, + ) -> Result<(cpal::Device, cpal::StreamConfig), Box> { + // Try windows WASAPI loopback on default output device + #[cfg(target_os = "windows")] + { + if let Some(device) = host.default_output_device() + && let Ok(config) = device.default_output_config() + { + return Ok((device, config.into())); + } + } + + // Try Linux PipeWire/PulseAudio output monitor device + #[cfg(target_os = "linux")] + { + if let Ok(devices) = host.devices() { + for dev in devices { + if let Ok(desc) = dev.description() { + let name = desc.to_string(); + + // Monitor devices mirror system output under PulseAudio/PipeWire + if name.contains("monitor") { + if let Ok(config) = dev.default_input_config() { + return Ok((dev, config.into())); + } + } + } + } + } + } + + // Fallback to default input device + let device = host + .default_input_device() + .or_else(|| host.default_output_device()) + .ok_or("No audio input or output device found")?; + + let config = device + .default_input_config() + .or_else(|_| device.default_output_config())? + .into(); + + Ok((device, config)) + } + + // we here in doubling octaves but fft outputs linearly spaced bins. + // so we bundle bins on a log scale from 20 hz to 12 kHz + // this allows bass, mids, and trembls to have equal visual + // proportions during display + // + // i.e. + // bins like this [0-500Hz] [500-1k] [1k-1.5k] [1.5k-2k] [2k-2.5k] [2.5k-3k] [3k-12kHz] + // vs + // bins like this [20-60Hz] [60-250Hz] [250-500Hz] [500-2kHz] [2k-4kHz] [4k-8kHz] [8k-12kHz] + fn build_log_bins(num_bars: usize, sample_rate: f32, chunk_size: usize) -> Vec<(usize, usize)> { + let nyquist = sample_rate / 2.0; + let max_hz = 12000.0f32; + + (0..num_bars) + .map(|i| { + let low_hz = 20.0 * (max_hz / 20.0).powf(i as f32 / num_bars as f32); + let high_hz = 20.0 * (max_hz / 20.0).powf((i + 1) as f32 / num_bars as f32); + + let low = ((low_hz / nyquist) * (chunk_size as f32 / 2.0)) as usize; + let high = ((high_hz / nyquist) * (chunk_size as f32 / 2.0)) as usize; + + (low.max(1), high.max(low + 1)) + }) + .collect() + } + + // without we get sharp edges which gives noise in FFT processing + fn apply_hann_window(samples: &[f32], input_buffer: &mut [f32]) { + let chunk_size = samples.len(); + + for (i, sample) in samples.iter().enumerate() { + let window = 0.5 * (1.0 - (2.0 * std::f32::consts::PI * i as f32 / chunk_size as f32).cos()); + input_buffer[i] = sample * window; + } + } + + fn process_fft_magnitudes( + spectrum: &[num_complex::Complex32], + log_bins: &[(usize, usize)], + freq_boost: &[f32], + prev_heights: &[f32], + autosens: f32, + smoothing: f32, + falloff: f32, + ) -> Vec { + let num_bars = log_bins.len(); + let mut current_bars = vec![0.0f32; num_bars]; + + // determine magnitude + for i in 0..num_bars { + let (start, stop) = log_bins[i]; + let bin_slice = &spectrum[start..stop.min(spectrum.len())]; + let magnitude_sum: f32 = bin_slice.iter().map(|c| c.norm()).sum(); + let avg_mag = if !bin_slice.is_empty() { + magnitude_sum / bin_slice.len() as f32 + } else { + 0.0 + }; + + // ignore quiet static noise below specific amplitude + let raw_val = if avg_mag < 0.02 { + 0.0 + } else { + (avg_mag * freq_boost[i] * autosens + 1.0).log10() * 3.5 + }; + + // rise smoothly from previous height + let target = (raw_val * (1.0 - smoothing)) + (prev_heights[i] * smoothing); + + // prevent sudden drops + if target < prev_heights[i] { + current_bars[i] = (prev_heights[i] - falloff).max(0.0); + } else { + current_bars[i] = target; + } + } + + current_bars + } + + // blend the heights between neighboring freq bins + // so instead of sharp spikes we get more of waves + fn apply_monstercat_smoothing(bars: &[f32]) -> Vec { + let num_bars = bars.len(); + let mut smoothed = bars.to_vec(); + + for i in 1..(num_bars - 1) { + smoothed[i] = (bars[i - 1] * 0.25) + (bars[i] * 0.50) + (bars[i + 1] * 0.25); + } + + smoothed + } + + // songs can be quiet and loud so we try to adjust + // sensitivity to avoid bars becoming flattened or + // clipped + fn adjust_autosens(autosens: &mut f32, bars: &[f32]) { + let max_val = bars.iter().copied().fold(0.0f32, f32::max); + + if max_val > 8.0 { + *autosens *= 0.98; + } else if max_val < 3.0 && *autosens < 3.0 { + *autosens *= 1.01; + } + } + + // captures output and runs through pipeline + // system audio -> audio buffer -> windowing -> fft process -> binning -> floor/autosens -> smoothing + fn run_audio_loop( + bar_data: Arc>>, + num_bars: usize, + stop: Arc, + ) -> Result<(), Box> { + let host = cpal::default_host(); + let (device, config) = Self::get_audio_device(&host)?; + + let sample_rate = config.sample_rate as f32; + let channels = config.channels as usize; + + let chunk_size = 2048; + let mut planner = RealFftPlanner::::new(); + let fft = planner.plan_fft_forward(chunk_size); + let mut windowed_buffer = fft.make_input_vec(); + let mut spectrum = fft.make_output_vec(); + + let mut prev_heights = vec![0.0f32; num_bars]; + let smoothing = 0.70f32; + let falloff = 0.08f32; + let mut autosens = 1.0f32; + + let log_bins = Self::build_log_bins(num_bars, sample_rate, chunk_size); + + let freq_boost: Vec = (0..num_bars) + .map(|i| 1.0 + (3.5 * (i as f32 / num_bars as f32).powf(1.2))) + .collect(); + + let audio_buffer = Arc::new(Mutex::new(Vec::::with_capacity(chunk_size * 2))); + let buffer_clone = Arc::clone(&audio_buffer); + + let stream = device.build_input_stream( + config, + move |data: &[f32], _| { + if let Ok(mut buf) = buffer_clone.lock() { + for chunk in data.chunks(channels) { + let mono: f32 = chunk.iter().sum::() / channels as f32; + buf.push(mono); + } + + if buf.len() > chunk_size * 2 { + let drain_amt = buf.len() - chunk_size; + buf.drain(0..drain_amt); + } + } + }, + |err| eprintln!("Stream error: {}", err), + None, + )?; + + stream.play()?; + + while !stop.load(Ordering::Relaxed) { + thread::sleep(std::time::Duration::from_millis(16)); + + let samples = { + let buf = match audio_buffer.lock() { + Ok(b) => b, + Err(_) => continue, + }; + if buf.len() < chunk_size { + continue; + } + buf[buf.len() - chunk_size..].to_vec() + }; + + Self::apply_hann_window(&samples, &mut windowed_buffer); + + if let Err(e) = fft.process(&mut windowed_buffer, &mut spectrum) { + eprintln!("FFT error: {:?}", e); + continue; + } + + let raw_bars = Self::process_fft_magnitudes( + &spectrum, + &log_bins, + &freq_boost, + &prev_heights, + autosens, + smoothing, + falloff, + ); + + let smoothed_bars = Self::apply_monstercat_smoothing(&raw_bars); + prev_heights = smoothed_bars.clone(); + + Self::adjust_autosens(&mut autosens, &smoothed_bars); + + if let Ok(mut bars) = bar_data.lock() { + for (i, val) in smoothed_bars.iter().enumerate() { + bars[i] = ((*val * 10.0).clamp(0.0, 100.0)) as u64; + } + } + } + + Ok(()) + } +} + +impl Drop for AudioVisualizer { + fn drop(&mut self) { + self.stop_flag.store(true, Ordering::Relaxed); + } +}