Rodio audio (#37786)
Adds input to the experimental rodio_audio pipeline.
Enable with:
```json
"audio": {
"experimental.rodio_audio": true
}
```
Additionally enables automatic volume
control for incoming audio:
```json
"audio": {
"experimental.control_output_volume": true
}
```
Release Notes:
- N/A
This commit is contained in:
@@ -97,9 +97,13 @@ impl Room {
|
||||
|
||||
pub async fn publish_local_microphone_track(
|
||||
&self,
|
||||
user_name: String,
|
||||
is_staff: bool,
|
||||
cx: &mut AsyncApp,
|
||||
) -> Result<(LocalTrackPublication, playback::AudioStream)> {
|
||||
let (track, stream) = self.playback.capture_local_microphone_track()?;
|
||||
let (track, stream) = self
|
||||
.playback
|
||||
.capture_local_microphone_track(user_name, is_staff, &cx)?;
|
||||
let publication = self
|
||||
.local_participant()
|
||||
.publish_track(
|
||||
@@ -129,7 +133,7 @@ impl Room {
|
||||
cx: &mut App,
|
||||
) -> Result<playback::AudioStream> {
|
||||
if AudioSettings::get_global(cx).rodio_audio {
|
||||
info!("Using experimental.rodio_audio audio pipeline");
|
||||
info!("Using experimental.rodio_audio audio pipeline for output");
|
||||
playback::play_remote_audio_track(&track.0, cx)
|
||||
} else {
|
||||
Ok(self.playback.play_remote_audio_track(&track.0))
|
||||
|
||||
@@ -1,10 +1,12 @@
|
||||
use anyhow::{Context as _, Result};
|
||||
|
||||
use audio::{AudioSettings, CHANNEL_COUNT, SAMPLE_RATE};
|
||||
use cpal::traits::{DeviceTrait, StreamTrait as _};
|
||||
use futures::channel::mpsc::UnboundedSender;
|
||||
use futures::{Stream, StreamExt as _};
|
||||
use gpui::{
|
||||
BackgroundExecutor, ScreenCaptureFrame, ScreenCaptureSource, ScreenCaptureStream, Task,
|
||||
AsyncApp, BackgroundExecutor, ScreenCaptureFrame, ScreenCaptureSource, ScreenCaptureStream,
|
||||
Task,
|
||||
};
|
||||
use libwebrtc::native::{apm, audio_mixer, audio_resampler};
|
||||
use livekit::track;
|
||||
@@ -17,8 +19,11 @@ use livekit::webrtc::{
|
||||
video_source::{RtcVideoSource, VideoResolution, native::NativeVideoSource},
|
||||
video_stream::native::NativeVideoStream,
|
||||
};
|
||||
use log::info;
|
||||
use parking_lot::Mutex;
|
||||
use rodio::Source;
|
||||
use serde::{Deserialize, Serialize};
|
||||
use settings::Settings;
|
||||
use std::cell::RefCell;
|
||||
use std::sync::Weak;
|
||||
use std::sync::atomic::{AtomicBool, AtomicI32, Ordering};
|
||||
@@ -36,27 +41,28 @@ pub(crate) struct AudioStack {
|
||||
next_ssrc: AtomicI32,
|
||||
}
|
||||
|
||||
// NOTE: We use WebRTC's mixer which only supports
|
||||
// 16kHz, 32kHz and 48kHz. As 48 is the most common "next step up"
|
||||
// for audio output devices like speakers/bluetooth, we just hard-code
|
||||
// this; and downsample when we need to.
|
||||
const SAMPLE_RATE: u32 = 48000;
|
||||
const NUM_CHANNELS: u32 = 2;
|
||||
|
||||
pub(crate) fn play_remote_audio_track(
|
||||
track: &livekit::track::RemoteAudioTrack,
|
||||
cx: &mut gpui::App,
|
||||
) -> Result<AudioStream> {
|
||||
let stop_handle = Arc::new(AtomicBool::new(false));
|
||||
let stop_handle_clone = stop_handle.clone();
|
||||
let stream = source::LiveKitStream::new(cx.background_executor(), track)
|
||||
let stream = source::LiveKitStream::new(cx.background_executor(), track);
|
||||
|
||||
let stream = stream
|
||||
.stoppable()
|
||||
.periodic_access(Duration::from_millis(50), move |s| {
|
||||
if stop_handle.load(Ordering::Relaxed) {
|
||||
s.stop();
|
||||
}
|
||||
});
|
||||
audio::Audio::play_source(stream, cx).context("Could not play audio")?;
|
||||
|
||||
let speaker: Speaker = serde_urlencoded::from_str(&track.name()).unwrap_or_else(|_| Speaker {
|
||||
name: track.name(),
|
||||
is_staff: false,
|
||||
});
|
||||
audio::Audio::play_voip_stream(stream, speaker.name, speaker.is_staff, cx)
|
||||
.context("Could not play audio")?;
|
||||
|
||||
let on_drop = util::defer(move || {
|
||||
stop_handle_clone.store(true, Ordering::Relaxed);
|
||||
@@ -90,8 +96,8 @@ impl AudioStack {
|
||||
let next_ssrc = self.next_ssrc.fetch_add(1, Ordering::Relaxed);
|
||||
let source = AudioMixerSource {
|
||||
ssrc: next_ssrc,
|
||||
sample_rate: SAMPLE_RATE,
|
||||
num_channels: NUM_CHANNELS,
|
||||
sample_rate: SAMPLE_RATE.get(),
|
||||
num_channels: CHANNEL_COUNT.get() as u32,
|
||||
buffer: Arc::default(),
|
||||
};
|
||||
self.mixer.lock().add_source(source.clone());
|
||||
@@ -131,7 +137,7 @@ impl AudioStack {
|
||||
let apm = self.apm.clone();
|
||||
let mixer = self.mixer.clone();
|
||||
async move {
|
||||
Self::play_output(apm, mixer, SAMPLE_RATE, NUM_CHANNELS)
|
||||
Self::play_output(apm, mixer, SAMPLE_RATE.get(), CHANNEL_COUNT.get().into())
|
||||
.await
|
||||
.log_err();
|
||||
}
|
||||
@@ -142,17 +148,26 @@ impl AudioStack {
|
||||
|
||||
pub(crate) fn capture_local_microphone_track(
|
||||
&self,
|
||||
user_name: String,
|
||||
is_staff: bool,
|
||||
cx: &AsyncApp,
|
||||
) -> Result<(crate::LocalAudioTrack, AudioStream)> {
|
||||
let source = NativeAudioSource::new(
|
||||
// n.b. this struct's options are always ignored, noise cancellation is provided by apm.
|
||||
AudioSourceOptions::default(),
|
||||
SAMPLE_RATE,
|
||||
NUM_CHANNELS,
|
||||
SAMPLE_RATE.get(),
|
||||
CHANNEL_COUNT.get().into(),
|
||||
10,
|
||||
);
|
||||
|
||||
let track_name = serde_urlencoded::to_string(Speaker {
|
||||
name: user_name,
|
||||
is_staff,
|
||||
})
|
||||
.context("Could not encode user information in track name")?;
|
||||
|
||||
let track = track::LocalAudioTrack::create_audio_track(
|
||||
"microphone",
|
||||
&track_name,
|
||||
RtcAudioSource::Native(source.clone()),
|
||||
);
|
||||
|
||||
@@ -166,9 +181,24 @@ impl AudioStack {
|
||||
}
|
||||
}
|
||||
});
|
||||
let capture_task = self.executor.spawn(async move {
|
||||
Self::capture_input(apm, frame_tx, SAMPLE_RATE, NUM_CHANNELS).await
|
||||
});
|
||||
let rodio_pipeline =
|
||||
AudioSettings::try_read_global(cx, |setting| setting.rodio_audio).unwrap_or_default();
|
||||
let capture_task = if rodio_pipeline {
|
||||
info!("Using experimental.rodio_audio audio pipeline");
|
||||
let voip_parts = audio::VoipParts::new(cx)?;
|
||||
thread::spawn(move || {
|
||||
// microphone is non send on mac
|
||||
let microphone = audio::Audio::open_microphone(voip_parts)?;
|
||||
send_to_livekit(frame_tx, microphone);
|
||||
Ok::<(), anyhow::Error>(())
|
||||
});
|
||||
Task::ready(Ok(()))
|
||||
} else {
|
||||
self.executor.spawn(async move {
|
||||
Self::capture_input(apm, frame_tx, SAMPLE_RATE.get(), CHANNEL_COUNT.get().into())
|
||||
.await
|
||||
})
|
||||
};
|
||||
|
||||
let on_drop = util::defer(|| {
|
||||
drop(transmit_task);
|
||||
@@ -346,6 +376,36 @@ impl AudioStack {
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize)]
|
||||
struct Speaker {
|
||||
name: String,
|
||||
is_staff: bool,
|
||||
}
|
||||
|
||||
fn send_to_livekit(frame_tx: UnboundedSender<AudioFrame<'static>>, mut microphone: impl Source) {
|
||||
use cpal::Sample;
|
||||
loop {
|
||||
let sampled: Vec<_> = microphone
|
||||
.by_ref()
|
||||
.take(audio::BUFFER_SIZE)
|
||||
.map(|s| s.to_sample())
|
||||
.collect();
|
||||
|
||||
if frame_tx
|
||||
.unbounded_send(AudioFrame {
|
||||
sample_rate: SAMPLE_RATE.get(),
|
||||
num_channels: CHANNEL_COUNT.get() as u32,
|
||||
samples_per_channel: sampled.len() as u32 / CHANNEL_COUNT.get() as u32,
|
||||
data: Cow::Owned(sampled),
|
||||
})
|
||||
.is_err()
|
||||
{
|
||||
// must rx has dropped or is not consuming
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
use super::LocalVideoTrack;
|
||||
|
||||
pub enum AudioStream {
|
||||
|
||||
@@ -1,15 +1,23 @@
|
||||
use std::num::NonZero;
|
||||
|
||||
use futures::StreamExt;
|
||||
use libwebrtc::{audio_stream::native::NativeAudioStream, prelude::AudioFrame};
|
||||
use livekit::track::RemoteAudioTrack;
|
||||
use rodio::{Source, buffer::SamplesBuffer, conversions::SampleTypeConverter};
|
||||
use rodio::{Source, buffer::SamplesBuffer, conversions::SampleTypeConverter, nz};
|
||||
|
||||
use crate::livekit_client::playback::{NUM_CHANNELS, SAMPLE_RATE};
|
||||
use audio::{CHANNEL_COUNT, SAMPLE_RATE};
|
||||
|
||||
fn frame_to_samplesbuffer(frame: AudioFrame) -> SamplesBuffer {
|
||||
let samples = frame.data.iter().copied();
|
||||
let samples = SampleTypeConverter::<_, _>::new(samples);
|
||||
let samples: Vec<f32> = samples.collect();
|
||||
SamplesBuffer::new(frame.num_channels as u16, frame.sample_rate, samples)
|
||||
SamplesBuffer::new(
|
||||
// here be dragons
|
||||
// NonZero::new(frame.num_channels as u16).expect("audio frame channels is nonzero"),
|
||||
nz!(2),
|
||||
NonZero::new(frame.sample_rate).expect("audio frame sample rate is nonzero"),
|
||||
samples,
|
||||
)
|
||||
}
|
||||
|
||||
pub struct LiveKitStream {
|
||||
@@ -20,8 +28,11 @@ pub struct LiveKitStream {
|
||||
|
||||
impl LiveKitStream {
|
||||
pub fn new(executor: &gpui::BackgroundExecutor, track: &RemoteAudioTrack) -> Self {
|
||||
let mut stream =
|
||||
NativeAudioStream::new(track.rtc_track(), SAMPLE_RATE as i32, NUM_CHANNELS as i32);
|
||||
let mut stream = NativeAudioStream::new(
|
||||
track.rtc_track(),
|
||||
SAMPLE_RATE.get() as i32,
|
||||
CHANNEL_COUNT.get().into(),
|
||||
);
|
||||
let (queue_input, queue_output) = rodio::queue::queue(true);
|
||||
// spawn rtc stream
|
||||
let receiver_task = executor.spawn({
|
||||
@@ -54,11 +65,17 @@ impl Source for LiveKitStream {
|
||||
}
|
||||
|
||||
fn channels(&self) -> rodio::ChannelCount {
|
||||
self.inner.channels()
|
||||
// This must be hardcoded because the playback source assumes constant
|
||||
// sample rate and channel count. The queue upon which this is build
|
||||
// will however report different counts and rates. Even though we put in
|
||||
// only items with our (constant) CHANNEL_COUNT & SAMPLE_RATE this will
|
||||
// play silence on one channel and at 44100 which is not what our
|
||||
// constants are.
|
||||
CHANNEL_COUNT
|
||||
}
|
||||
|
||||
fn sample_rate(&self) -> rodio::SampleRate {
|
||||
self.inner.sample_rate()
|
||||
SAMPLE_RATE // see comment on channels
|
||||
}
|
||||
|
||||
fn total_duration(&self) -> Option<std::time::Duration> {
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
use std::{
|
||||
env,
|
||||
num::NonZero,
|
||||
path::{Path, PathBuf},
|
||||
sync::{Arc, Mutex},
|
||||
time::Duration,
|
||||
@@ -83,8 +84,12 @@ fn write_out(
|
||||
.expect("Stream has ended, callback cant hold the lock"),
|
||||
);
|
||||
let samples: Vec<f32> = SampleTypeConverter::<_, f32>::new(samples.into_iter()).collect();
|
||||
let mut samples = SamplesBuffer::new(config.channels(), config.sample_rate().0, samples);
|
||||
match rodio::output_to_wav(&mut samples, path) {
|
||||
let mut samples = SamplesBuffer::new(
|
||||
NonZero::new(config.channels()).expect("config channel is never zero"),
|
||||
NonZero::new(config.sample_rate().0).expect("config sample_rate is never zero"),
|
||||
samples,
|
||||
);
|
||||
match rodio::wav_to_file(&mut samples, path) {
|
||||
Ok(_) => Ok(()),
|
||||
Err(e) => Err(anyhow::anyhow!("Failed to write wav file: {}", e)),
|
||||
}
|
||||
|
||||
@@ -728,6 +728,8 @@ impl Room {
|
||||
|
||||
pub async fn publish_local_microphone_track(
|
||||
&self,
|
||||
_track_name: String,
|
||||
_is_staff: bool,
|
||||
cx: &mut AsyncApp,
|
||||
) -> Result<(LocalTrackPublication, AudioStream)> {
|
||||
self.local_participant().publish_microphone_track(cx).await
|
||||
|
||||
Reference in New Issue
Block a user