refactor: extract audio visualizer
This commit is contained in:
+298
@@ -0,0 +1,298 @@
|
||||
use std::{
|
||||
sync::{
|
||||
Arc, Mutex,
|
||||
atomic::{AtomicBool, Ordering},
|
||||
},
|
||||
thread,
|
||||
};
|
||||
|
||||
use cpal::traits::{DeviceTrait, HostTrait, StreamTrait};
|
||||
use realfft::{RealFftPlanner, num_complex};
|
||||
|
||||
pub struct AudioVisualizer {
|
||||
pub bar_data: Arc<Mutex<Vec<u64>>>,
|
||||
pub stop_flag: Arc<AtomicBool>,
|
||||
}
|
||||
|
||||
// This is a implementation largely lifted from these
|
||||
// open source implementation. I just wrote it in Rust here:
|
||||
// 1. CAVA (C): https://github.com/karlstav/cava
|
||||
// 2. cli-visualizer (C++): https://github.com/dpayne/cli-visualizer
|
||||
impl AudioVisualizer {
|
||||
pub fn new(num_bars: usize) -> Self {
|
||||
let bar_data = Arc::new(Mutex::new(vec![0; num_bars]));
|
||||
let bar_data_clone = Arc::clone(&bar_data);
|
||||
let stop_flag = Arc::new(AtomicBool::new(false));
|
||||
let stop_clone = Arc::clone(&stop_flag);
|
||||
|
||||
thread::spawn(move || {
|
||||
if let Err(e) = Self::run_audio_loop(bar_data_clone, num_bars, stop_clone) {
|
||||
eprintln!("Audio capture error: {:?}", e);
|
||||
}
|
||||
});
|
||||
|
||||
Self {
|
||||
bar_data,
|
||||
stop_flag,
|
||||
}
|
||||
}
|
||||
|
||||
// Doing best effort to support cross-platform functionality
|
||||
fn get_audio_device(
|
||||
host: &cpal::Host,
|
||||
) -> Result<(cpal::Device, cpal::StreamConfig), Box<dyn std::error::Error>> {
|
||||
// Try windows WASAPI loopback on default output device
|
||||
#[cfg(target_os = "windows")]
|
||||
{
|
||||
if let Some(device) = host.default_output_device()
|
||||
&& let Ok(config) = device.default_output_config()
|
||||
{
|
||||
return Ok((device, config.into()));
|
||||
}
|
||||
}
|
||||
|
||||
// Try Linux PipeWire/PulseAudio output monitor device
|
||||
#[cfg(target_os = "linux")]
|
||||
{
|
||||
if let Ok(devices) = host.devices() {
|
||||
for dev in devices {
|
||||
if let Ok(desc) = dev.description() {
|
||||
let name = desc.to_string();
|
||||
|
||||
// Monitor devices mirror system output under PulseAudio/PipeWire
|
||||
if name.contains("monitor") {
|
||||
if let Ok(config) = dev.default_input_config() {
|
||||
return Ok((dev, config.into()));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback to default input device
|
||||
let device = host
|
||||
.default_input_device()
|
||||
.or_else(|| host.default_output_device())
|
||||
.ok_or("No audio input or output device found")?;
|
||||
|
||||
let config = device
|
||||
.default_input_config()
|
||||
.or_else(|_| device.default_output_config())?
|
||||
.into();
|
||||
|
||||
Ok((device, config))
|
||||
}
|
||||
|
||||
// we here in doubling octaves but fft outputs linearly spaced bins.
|
||||
// so we bundle bins on a log scale from 20 hz to 12 kHz
|
||||
// this allows bass, mids, and trembls to have equal visual
|
||||
// proportions during display
|
||||
//
|
||||
// i.e.
|
||||
// bins like this [0-500Hz] [500-1k] [1k-1.5k] [1.5k-2k] [2k-2.5k] [2.5k-3k] [3k-12kHz]
|
||||
// vs
|
||||
// bins like this [20-60Hz] [60-250Hz] [250-500Hz] [500-2kHz] [2k-4kHz] [4k-8kHz] [8k-12kHz]
|
||||
fn build_log_bins(num_bars: usize, sample_rate: f32, chunk_size: usize) -> Vec<(usize, usize)> {
|
||||
let nyquist = sample_rate / 2.0;
|
||||
let max_hz = 12000.0f32;
|
||||
|
||||
(0..num_bars)
|
||||
.map(|i| {
|
||||
let low_hz = 20.0 * (max_hz / 20.0).powf(i as f32 / num_bars as f32);
|
||||
let high_hz = 20.0 * (max_hz / 20.0).powf((i + 1) as f32 / num_bars as f32);
|
||||
|
||||
let low = ((low_hz / nyquist) * (chunk_size as f32 / 2.0)) as usize;
|
||||
let high = ((high_hz / nyquist) * (chunk_size as f32 / 2.0)) as usize;
|
||||
|
||||
(low.max(1), high.max(low + 1))
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
// without we get sharp edges which gives noise in FFT processing
|
||||
fn apply_hann_window(samples: &[f32], input_buffer: &mut [f32]) {
|
||||
let chunk_size = samples.len();
|
||||
|
||||
for (i, sample) in samples.iter().enumerate() {
|
||||
let window = 0.5 * (1.0 - (2.0 * std::f32::consts::PI * i as f32 / chunk_size as f32).cos());
|
||||
input_buffer[i] = sample * window;
|
||||
}
|
||||
}
|
||||
|
||||
fn process_fft_magnitudes(
|
||||
spectrum: &[num_complex::Complex32],
|
||||
log_bins: &[(usize, usize)],
|
||||
freq_boost: &[f32],
|
||||
prev_heights: &[f32],
|
||||
autosens: f32,
|
||||
smoothing: f32,
|
||||
falloff: f32,
|
||||
) -> Vec<f32> {
|
||||
let num_bars = log_bins.len();
|
||||
let mut current_bars = vec![0.0f32; num_bars];
|
||||
|
||||
// determine magnitude
|
||||
for i in 0..num_bars {
|
||||
let (start, stop) = log_bins[i];
|
||||
let bin_slice = &spectrum[start..stop.min(spectrum.len())];
|
||||
let magnitude_sum: f32 = bin_slice.iter().map(|c| c.norm()).sum();
|
||||
let avg_mag = if !bin_slice.is_empty() {
|
||||
magnitude_sum / bin_slice.len() as f32
|
||||
} else {
|
||||
0.0
|
||||
};
|
||||
|
||||
// ignore quiet static noise below specific amplitude
|
||||
let raw_val = if avg_mag < 0.02 {
|
||||
0.0
|
||||
} else {
|
||||
(avg_mag * freq_boost[i] * autosens + 1.0).log10() * 3.5
|
||||
};
|
||||
|
||||
// rise smoothly from previous height
|
||||
let target = (raw_val * (1.0 - smoothing)) + (prev_heights[i] * smoothing);
|
||||
|
||||
// prevent sudden drops
|
||||
if target < prev_heights[i] {
|
||||
current_bars[i] = (prev_heights[i] - falloff).max(0.0);
|
||||
} else {
|
||||
current_bars[i] = target;
|
||||
}
|
||||
}
|
||||
|
||||
current_bars
|
||||
}
|
||||
|
||||
// blend the heights between neighboring freq bins
|
||||
// so instead of sharp spikes we get more of waves
|
||||
fn apply_monstercat_smoothing(bars: &[f32]) -> Vec<f32> {
|
||||
let num_bars = bars.len();
|
||||
let mut smoothed = bars.to_vec();
|
||||
|
||||
for i in 1..(num_bars - 1) {
|
||||
smoothed[i] = (bars[i - 1] * 0.25) + (bars[i] * 0.50) + (bars[i + 1] * 0.25);
|
||||
}
|
||||
|
||||
smoothed
|
||||
}
|
||||
|
||||
// songs can be quiet and loud so we try to adjust
|
||||
// sensitivity to avoid bars becoming flattened or
|
||||
// clipped
|
||||
fn adjust_autosens(autosens: &mut f32, bars: &[f32]) {
|
||||
let max_val = bars.iter().copied().fold(0.0f32, f32::max);
|
||||
|
||||
if max_val > 8.0 {
|
||||
*autosens *= 0.98;
|
||||
} else if max_val < 3.0 && *autosens < 3.0 {
|
||||
*autosens *= 1.01;
|
||||
}
|
||||
}
|
||||
|
||||
// captures output and runs through pipeline
|
||||
// system audio -> audio buffer -> windowing -> fft process -> binning -> floor/autosens -> smoothing
|
||||
fn run_audio_loop(
|
||||
bar_data: Arc<Mutex<Vec<u64>>>,
|
||||
num_bars: usize,
|
||||
stop: Arc<AtomicBool>,
|
||||
) -> Result<(), Box<dyn std::error::Error>> {
|
||||
let host = cpal::default_host();
|
||||
let (device, config) = Self::get_audio_device(&host)?;
|
||||
|
||||
let sample_rate = config.sample_rate as f32;
|
||||
let channels = config.channels as usize;
|
||||
|
||||
let chunk_size = 2048;
|
||||
let mut planner = RealFftPlanner::<f32>::new();
|
||||
let fft = planner.plan_fft_forward(chunk_size);
|
||||
let mut windowed_buffer = fft.make_input_vec();
|
||||
let mut spectrum = fft.make_output_vec();
|
||||
|
||||
let mut prev_heights = vec![0.0f32; num_bars];
|
||||
let smoothing = 0.70f32;
|
||||
let falloff = 0.08f32;
|
||||
let mut autosens = 1.0f32;
|
||||
|
||||
let log_bins = Self::build_log_bins(num_bars, sample_rate, chunk_size);
|
||||
|
||||
let freq_boost: Vec<f32> = (0..num_bars)
|
||||
.map(|i| 1.0 + (3.5 * (i as f32 / num_bars as f32).powf(1.2)))
|
||||
.collect();
|
||||
|
||||
let audio_buffer = Arc::new(Mutex::new(Vec::<f32>::with_capacity(chunk_size * 2)));
|
||||
let buffer_clone = Arc::clone(&audio_buffer);
|
||||
|
||||
let stream = device.build_input_stream(
|
||||
config,
|
||||
move |data: &[f32], _| {
|
||||
if let Ok(mut buf) = buffer_clone.lock() {
|
||||
for chunk in data.chunks(channels) {
|
||||
let mono: f32 = chunk.iter().sum::<f32>() / channels as f32;
|
||||
buf.push(mono);
|
||||
}
|
||||
|
||||
if buf.len() > chunk_size * 2 {
|
||||
let drain_amt = buf.len() - chunk_size;
|
||||
buf.drain(0..drain_amt);
|
||||
}
|
||||
}
|
||||
},
|
||||
|err| eprintln!("Stream error: {}", err),
|
||||
None,
|
||||
)?;
|
||||
|
||||
stream.play()?;
|
||||
|
||||
while !stop.load(Ordering::Relaxed) {
|
||||
thread::sleep(std::time::Duration::from_millis(16));
|
||||
|
||||
let samples = {
|
||||
let buf = match audio_buffer.lock() {
|
||||
Ok(b) => b,
|
||||
Err(_) => continue,
|
||||
};
|
||||
if buf.len() < chunk_size {
|
||||
continue;
|
||||
}
|
||||
buf[buf.len() - chunk_size..].to_vec()
|
||||
};
|
||||
|
||||
Self::apply_hann_window(&samples, &mut windowed_buffer);
|
||||
|
||||
if let Err(e) = fft.process(&mut windowed_buffer, &mut spectrum) {
|
||||
eprintln!("FFT error: {:?}", e);
|
||||
continue;
|
||||
}
|
||||
|
||||
let raw_bars = Self::process_fft_magnitudes(
|
||||
&spectrum,
|
||||
&log_bins,
|
||||
&freq_boost,
|
||||
&prev_heights,
|
||||
autosens,
|
||||
smoothing,
|
||||
falloff,
|
||||
);
|
||||
|
||||
let smoothed_bars = Self::apply_monstercat_smoothing(&raw_bars);
|
||||
prev_heights = smoothed_bars.clone();
|
||||
|
||||
Self::adjust_autosens(&mut autosens, &smoothed_bars);
|
||||
|
||||
if let Ok(mut bars) = bar_data.lock() {
|
||||
for (i, val) in smoothed_bars.iter().enumerate() {
|
||||
bars[i] = ((*val * 10.0).clamp(0.0, 100.0)) as u64;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for AudioVisualizer {
|
||||
fn drop(&mut self) {
|
||||
self.stop_flag.store(true, Ordering::Relaxed);
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user