diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml
index e362ea8d..02bd9f46 100644
--- a/.github/workflows/quality-gates.yml
+++ b/.github/workflows/quality-gates.yml
@@ -130,7 +130,7 @@ jobs:
Frontend/src-tauri/target/debug/build
Frontend/src-tauri/target/debug/deps
Frontend/src-tauri/target/debug/incremental
- key: scriber-rust-quality-v1-${{ runner.os }}-${{ runner.arch }}-${{ steps.rust-quality-environment.outputs.identity }}-${{ hashFiles('.node-version', 'Frontend/src-tauri/Cargo.toml', 'Frontend/src-tauri/Cargo.lock', 'Frontend/src-tauri/build.rs', 'Frontend/src-tauri/tauri.conf.json', 'Frontend/src-tauri/capabilities/**', '.cargo/config', '.cargo/config.toml', 'Frontend/src-tauri/.cargo/config', 'Frontend/src-tauri/.cargo/config.toml', 'native/scriber-audio-sidecar/Cargo.toml', 'native/scriber-audio-sidecar/Cargo.lock', 'native/scriber-audio-sidecar/build.rs', 'native/scriber-audio-sidecar/windows-app-manifest.xml', 'scripts/stage_audio_worker.mjs', '.github/workflows/quality-gates.yml') }}
+ key: scriber-rust-quality-v1-${{ runner.os }}-${{ runner.arch }}-${{ steps.rust-quality-environment.outputs.identity }}-${{ hashFiles('.node-version', 'Frontend/src-tauri/Cargo.toml', 'Frontend/src-tauri/Cargo.lock', 'Frontend/src-tauri/build.rs', 'Frontend/src-tauri/tauri.conf.json', 'Frontend/src-tauri/capabilities/**', '.cargo/config', '.cargo/config.toml', 'Frontend/src-tauri/.cargo/config', 'Frontend/src-tauri/.cargo/config.toml', 'native/scriber-localvqe/**', 'native/scriber-audio-sidecar/Cargo.toml', 'native/scriber-audio-sidecar/Cargo.lock', 'native/scriber-audio-sidecar/build.rs', 'native/scriber-audio-sidecar/windows-app-manifest.xml', 'scripts/stage_audio_worker.mjs', '.github/workflows/quality-gates.yml') }}
- name: Check Rust formatting
working-directory: Frontend/src-tauri
run: cargo fmt --all -- --check
@@ -153,6 +153,8 @@ jobs:
run: cargo clippy --locked --manifest-path native/scriber-audio-sidecar/Cargo.toml --target-dir Frontend/src-tauri/target --all-targets -- -D warnings
- name: Run standalone audio tests
run: cargo test --locked --manifest-path native/scriber-audio-sidecar/Cargo.toml --target-dir Frontend/src-tauri/target
+ - name: Verify LocalVQE against pinned upstream reference
+ run: cargo test --locked --manifest-path native/scriber-audio-sidecar/Cargo.toml --target-dir Frontend/src-tauri/target -p scriber-localvqe
- name: Save trusted Rust quality compilation cache
if: >-
success() &&
diff --git a/AGENTS.md b/AGENTS.md
index 1cf8bd9a..8c174095 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -381,7 +381,13 @@ Frontend and shell:
bounded 16 kHz mono signed-16 PCM fixture; it has the same play-once and
test-only constraints.
Meeting capture uses one sidecar process for 48 kHz microphone plus loopback,
- pinned AEC3 processing, and shared-timeline raw mic/system/clean mic pipes.
+ pinned LocalVQE v1.3 processing, and shared-timeline raw mic/system/clean mic pipes.
+ `native/scriber-localvqe` owns the single embedded 4.8M F32 model and static
+ GGML bridge; source/model revisions and hashes live in its `inputs.json`.
+ x64 enhancement requires AVX2/FMA/F16C, checked before native inference.
+ Preserve the 16 kHz 256-sample hop adapter, one-hop delay compensation,
+ bounded shared-timeline queue, and Stop/Pause/EOF tail flush. Model construction
+ must emit no stdout because that channel carries the sidecar JSON protocol.
The token-protected Meeting device test must reuse this path, remain explicit
and local-only, return only bounded level/activity statistics, and always stop
its ephemeral sidecar capture without persisting or uploading PCM. Product
@@ -1694,7 +1700,7 @@ Packaging and scripts:
and `paused` become resumable `interrupted` capture. `stopping` and
`finalizing` become `finalization_failed`, and `analyzing` becomes
`analysis_failed`; never offer capture resume for a post-capture crash.
-- Keep audio format layers separate. AEC3, Silero, Smart Turn, live STT, and
+- Keep audio format layers separate. LocalVQE v1.3, Silero, Smart Turn, live STT, and
checkpoint capture use PCM. The verified long-lived archive is lossless
Matroska/FLAC; timeline-aligned mix, clean-microphone, and system Opus files
are playback derivatives, not canonical inference input. Persist and validate
diff --git a/Frontend/client/src/i18n/i18n.test.ts b/Frontend/client/src/i18n/i18n.test.ts
index 0d83c540..883a7bae 100644
--- a/Frontend/client/src/i18n/i18n.test.ts
+++ b/Frontend/client/src/i18n/i18n.test.ts
@@ -116,8 +116,8 @@ test("localizes legacy relative date labels", () => {
test("dynamic interface labels have complete German translations", () => {
assert.equal(translate("de", "Your voice"), "Deine Stimme");
assert.equal(translate("de", "Other participants"), "Andere Teilnehmende");
- assert.equal(translate("de", "Echo control"), "Echounterdrückung");
- assert.equal(translate("de", "Reduces speaker echo"), "Reduziert Lautsprecherechos");
+ assert.equal(translate("de", "Audio cleanup"), "Audiobereinigung");
+ assert.equal(translate("de", "Reduces echo, noise, and reverberation"), "Reduziert Echo, Rauschen und Hall");
assert.equal(translate("de", "Preparing audio"), "Audio wird vorbereitet");
assert.equal(translate("de", "Preparing audio download..."), "Audiodownload wird vorbereitet …");
assert.equal(
diff --git a/Frontend/client/src/i18n/translations/de/meetings.ts b/Frontend/client/src/i18n/translations/de/meetings.ts
index 514a9729..7e92ea10 100644
--- a/Frontend/client/src/i18n/translations/de/meetings.ts
+++ b/Frontend/client/src/i18n/translations/de/meetings.ts
@@ -491,8 +491,8 @@ export const meetingsTranslations = {
"System audio": "Systemaudio",
"Your voice": "Deine Stimme",
"Other participants": "Andere Teilnehmende",
- "Echo control": "Echounterdrückung",
- "Reduces speaker echo": "Reduziert Lautsprecherechos",
+ "Audio cleanup": "Audiobereinigung",
+ "Reduces echo, noise, and reverberation": "Reduziert Echo, Rauschen und Hall",
"Audio playback could not start.": "Die Audiowiedergabe konnte nicht gestartet werden.",
"Saved audio is not available for this meeting.": "Für dieses Meeting ist kein gespeichertes Audio verfügbar.",
"No saved voice sample is available for this speaker.":
diff --git a/Frontend/client/src/i18n/translations/de/settings.ts b/Frontend/client/src/i18n/translations/de/settings.ts
index cd5866de..485adf37 100644
--- a/Frontend/client/src/i18n/translations/de/settings.ts
+++ b/Frontend/client/src/i18n/translations/de/settings.ts
@@ -568,10 +568,9 @@ export const settingsTranslations = {
"Why Scriber does not upload one-minute pieces": "Warum Scriber keine einminütigen Abschnitte hochlädt",
"Small cloud requests do not reduce the audio duration you pay for and can reset speaker labels or cut words at the boundary. Scriber instead protects audio locally every 30 seconds, then gives the final service the longest supported context.":
"Kleine Cloud-Anfragen verringern nicht die berechnete Audiodauer und können Sprecherbezeichnungen zurücksetzen oder Wörter an Abschnittsgrenzen abschneiden. Scriber sichert das Audio stattdessen alle 30 Sekunden lokal und übergibt dem finalen Dienst anschließend den längsten unterstützten Kontext.",
- "Reduce speaker echo": "Lautsprecherecho reduzieren",
- "Helps prevent voices from your speakers being recorded again through your microphone.":
- "Verhindert, dass Stimmen aus deinen Lautsprechern erneut über dein Mikrofon aufgenommen werden.",
- "Reduce speaker echo in meetings": "Lautsprecherecho in Meetings reduzieren",
+ "Clean up meeting audio": "Meeting-Audio bereinigen",
+ "Reduces speaker echo, background noise, and reverberation locally on your computer.":
+ "Reduziert Lautsprecherechos, Hintergrundgeräusche und Hall lokal auf deinem Computer.",
"Summaries and storage": "Zusammenfassungen und Speicherung",
"Choose how Scriber creates the meeting brief and how long it keeps local audio.":
"Lege fest, wie Scriber die Meeting-Zusammenfassung erstellt und wie lange lokales Audio aufbewahrt wird.",
diff --git a/Frontend/client/src/pages/Meetings.tsx b/Frontend/client/src/pages/Meetings.tsx
index cd1e6162..56d945dc 100644
--- a/Frontend/client/src/pages/Meetings.tsx
+++ b/Frontend/client/src/pages/Meetings.tsx
@@ -3069,7 +3069,7 @@ export default function Meetings({ params }: { params?: { id?: string } }) {
{[
{ icon: Mic2, label: "Microphone", detail: "Your voice" },
{ icon: Headphones, label: "System audio", detail: "Other participants" },
- { icon: Waves, label: "Echo control", detail: "Reduces speaker echo" },
+ { icon: Waves, label: "Audio cleanup", detail: "Reduces echo, noise, and reverberation" },
].map(({ icon: Icon, label, detail }) => (
)}
diff --git a/Frontend/client/src/pages/Settings.tsx b/Frontend/client/src/pages/Settings.tsx
index 7b1ec7dc..fd474a8b 100644
--- a/Frontend/client/src/pages/Settings.tsx
+++ b/Frontend/client/src/pages/Settings.tsx
@@ -6019,15 +6019,15 @@ export default function Settings() {
void updateMeetingPreferences({ meetingAecEnabled: enabled })}
- aria-label={t("Reduce speaker echo in meetings")}
+ aria-label={t("Clean up meeting audio")}
/>
diff --git a/Frontend/package-lock.json b/Frontend/package-lock.json
index 427166b1..65afbcab 100644
--- a/Frontend/package-lock.json
+++ b/Frontend/package-lock.json
@@ -1,12 +1,12 @@
{
"name": "scriber",
- "version": "0.5.125",
+ "version": "0.5.126",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "scriber",
- "version": "0.5.125",
+ "version": "0.5.126",
"license": "MIT",
"dependencies": {
"@carrot-kpi/switzer-font": "^0.1.0",
diff --git a/Frontend/package.json b/Frontend/package.json
index efc6ea16..83acd026 100644
--- a/Frontend/package.json
+++ b/Frontend/package.json
@@ -1,6 +1,6 @@
{
"name": "scriber",
- "version": "0.5.125",
+ "version": "0.5.126",
"type": "module",
"license": "MIT",
"engines": {
diff --git a/Frontend/src-tauri/src/audio_sidecar.rs b/Frontend/src-tauri/src/audio_sidecar.rs
index d1369e18..551d6214 100644
--- a/Frontend/src-tauri/src/audio_sidecar.rs
+++ b/Frontend/src-tauri/src/audio_sidecar.rs
@@ -12,7 +12,7 @@ use audio_frame_pipe::{
AUDIO_FRAME_HEADER_LEN, AUDIO_FRAME_VERSION,
};
use audio_prepare::{AudioPreparationSubmit, AudioPreparationWorker, CaptureArtifactTarget};
-use meeting_aec::{MeetingAec3, MEETING_AEC_FRAME_SAMPLES};
+use meeting_aec::{MeetingEnhancer, MEETING_AEC_FRAME_SAMPLES, MEETING_OUTPUT_SAMPLES};
use redaction::hash_sensitive_identifier;
use serde_json::{json, Value};
use std::collections::{HashMap, VecDeque};
@@ -521,6 +521,14 @@ impl AudioSidecarState {
if !wasapi_capture_enabled() && !synthetic_capture_enabled() {
return Err("Rust meeting capture is unavailable".to_string());
}
+ if payload
+ .get("aecEnabled")
+ .and_then(Value::as_bool)
+ .unwrap_or(true)
+ && !scriber_localvqe::cpu_supported()
+ {
+ return Err("LocalVQE requires an AVX2/FMA/F16C-capable CPU".into());
+ }
let meeting_clock_origin = Instant::now();
let mut microphone_request = CaptureRequest::from_payload(payload);
microphone_request.clock_origin = Some(meeting_clock_origin);
@@ -1339,7 +1347,7 @@ fn start_meeting_aec_relay(
let (stop_tx, stop_rx) = mpsc::channel();
let thread_handles: Vec = handles.iter().map(|handle| *handle as isize).collect();
let join_handle = thread::Builder::new()
- .name("scriber-meeting-aec3-relay".to_string())
+ .name("scriber-meeting-localvqe-relay".to_string())
.spawn(move || {
run_meeting_aec_relay(
microphone_pipe,
@@ -1356,7 +1364,7 @@ fn start_meeting_aec_relay(
CloseHandle(handle);
}
}
- format!("meeting AEC3 relay thread spawn failed: {error}")
+ format!("meeting LocalVQE relay thread spawn failed: {error}")
})?;
let session = MeetingCaptureSession {
meeting_capture_id: meeting_capture_id.to_string(),
@@ -1387,7 +1395,7 @@ fn start_meeting_aec_relay(
_delay_ms: i32,
_aec_enabled: bool,
) -> Result<(MeetingCaptureSession, Value), String> {
- Err("meeting AEC3 relay is only implemented on Windows".to_string())
+ Err("meeting LocalVQE relay is only implemented on Windows".to_string())
}
#[derive(Debug, Default)]
@@ -3612,7 +3620,7 @@ fn meeting_frame_energy(samples: &[i16]) -> f64 {
}
fn meeting_render_active_energy_threshold() -> f64 {
- 64.0_f64.powi(2) * MEETING_AEC_FRAME_SAMPLES as f64
+ 64.0_f64.powi(2) * MEETING_OUTPUT_SAMPLES as f64
}
#[cfg(windows)]
@@ -3678,13 +3686,66 @@ fn meeting_alignment_action(
}
}
+#[cfg(windows)]
+struct PendingMeetingFrame {
+ header: AudioFrameHeader,
+ microphone: [i16; MEETING_OUTPUT_SAMPLES],
+ system: [i16; MEETING_OUTPUT_SAMPLES],
+}
+
+#[cfg(windows)]
+fn emit_meeting_frame(
+ outputs: &[HANDLE],
+ frame: PendingMeetingFrame,
+ clean: &[i16],
+ enhanced: bool,
+ stats: &mut MeetingRelayStats,
+ payload: &mut Vec,
+) -> Result<(), String> {
+ let energy = meeting_frame_energy(&frame.system);
+ if enhanced && energy >= meeting_render_active_energy_threshold() {
+ stats.aec_render_active_frames = stats.aec_render_active_frames.saturating_add(1);
+ stats.aec_render_energy += energy;
+ stats.aec_raw_mic_energy += meeting_frame_energy(&frame.microphone);
+ stats.aec_clean_mic_energy += meeting_frame_energy(clean);
+ }
+ for (handle, samples) in
+ outputs
+ .iter()
+ .zip([frame.microphone.as_slice(), frame.system.as_slice(), clean])
+ {
+ pcm_bytes_into(samples, payload);
+ stats.bytes_forwarded += write_meeting_frame(*handle, frame.header, payload)?;
+ }
+ stats.frames_processed += 1;
+ Ok(())
+}
+
+#[cfg(windows)]
+fn drain_meeting_enhancement(
+ processor: &mut MeetingEnhancer,
+ pending: &mut VecDeque,
+ outputs: &[HANDLE],
+ stats: &mut MeetingRelayStats,
+ clean: &mut Vec,
+ payload: &mut Vec,
+) -> Result<(), String> {
+ while processor.pop(clean) {
+ let frame = pending
+ .pop_front()
+ .ok_or("LocalVQE output exceeded capture timeline")?;
+ emit_meeting_frame(outputs, frame, clean, true, stats, payload)?;
+ }
+ Ok(())
+}
+
#[cfg(windows)]
fn run_meeting_aec_relay(
microphone_pipe: String,
system_pipe: String,
output_handles: Vec,
stop_rx: mpsc::Receiver<()>,
- delay_ms: i32,
+ _delay_ms: i32,
aec_enabled: bool,
) -> MeetingRelayStats {
let mut stats = MeetingRelayStats::default();
@@ -3708,7 +3769,7 @@ fn run_meeting_aec_relay(
wait_for_meeting_output_pipe_client(*handle, &stop_rx)?;
}
let mut aec = if aec_enabled {
- Some(MeetingAec3::new(delay_ms)?)
+ Some(MeetingEnhancer::new()?)
} else {
None
};
@@ -3719,190 +3780,224 @@ fn run_meeting_aec_relay(
let mut system_done = false;
let mut mic_samples = Vec::with_capacity(MEETING_AEC_FRAME_SAMPLES);
let mut system_samples = Vec::with_capacity(MEETING_AEC_FRAME_SAMPLES);
- let mut clean_samples = Vec::with_capacity(MEETING_AEC_FRAME_SAMPLES);
let output_samples = MEETING_AEC_FRAME_SAMPLES / 3;
let mut microphone_16k = Vec::with_capacity(output_samples);
let mut system_16k = Vec::with_capacity(output_samples);
let mut clean_16k = Vec::with_capacity(output_samples);
- let mut microphone_payload = Vec::with_capacity(output_samples * 2);
- let mut system_payload = Vec::with_capacity(output_samples * 2);
- let mut clean_payload = Vec::with_capacity(output_samples * 2);
+ let mut payload = Vec::with_capacity(output_samples * 2);
+ let mut pending = VecDeque::with_capacity(4);
let mut relay_terminal_timestamp_micros = 0_u64;
- loop {
- if microphone_frame.as_ref().is_some_and(|(header, payload)| {
- header.flags & AUDIO_FRAME_FLAG_END_OF_STREAM != 0
- && header.frame_count == 0
- && payload.is_empty()
- }) {
- relay_terminal_timestamp_micros = relay_terminal_timestamp_micros.max(
- microphone_frame
- .as_ref()
- .map(|(header, _)| header.timestamp_micros)
- .unwrap_or_default(),
+ let relay_result = (|| -> Result<(), String> {
+ loop {
+ if microphone_frame.as_ref().is_some_and(|(header, payload)| {
+ header.flags & AUDIO_FRAME_FLAG_END_OF_STREAM != 0
+ && header.frame_count == 0
+ && payload.is_empty()
+ }) {
+ relay_terminal_timestamp_micros = relay_terminal_timestamp_micros.max(
+ microphone_frame
+ .as_ref()
+ .map(|(header, _)| header.timestamp_micros)
+ .unwrap_or_default(),
+ );
+ microphone_done = true;
+ microphone_frame = None;
+ }
+ if system_frame.as_ref().is_some_and(|(header, payload)| {
+ header.flags & AUDIO_FRAME_FLAG_END_OF_STREAM != 0
+ && header.frame_count == 0
+ && payload.is_empty()
+ }) {
+ relay_terminal_timestamp_micros = relay_terminal_timestamp_micros.max(
+ system_frame
+ .as_ref()
+ .map(|(header, _)| header.timestamp_micros)
+ .unwrap_or_default(),
+ );
+ system_done = true;
+ system_frame = None;
+ }
+ if microphone_done && system_done {
+ break;
+ }
+ let microphone_timestamp = microphone_frame
+ .as_ref()
+ .map(|(header, _)| header.timestamp_micros);
+ let system_timestamp = system_frame
+ .as_ref()
+ .map(|(header, _)| header.timestamp_micros);
+ let Some(action) =
+ meeting_alignment_action(microphone_timestamp, system_timestamp)
+ else {
+ break;
+ };
+ if let (Some(microphone), Some(system)) =
+ (microphone_timestamp, system_timestamp)
+ {
+ stats.max_input_skew_micros =
+ stats.max_input_skew_micros.max(microphone.abs_diff(system));
+ }
+ let consume_microphone = matches!(
+ action,
+ MeetingAlignmentAction::Pair | MeetingAlignmentAction::MicrophoneOnly
);
- microphone_done = true;
- microphone_frame = None;
- }
- if system_frame.as_ref().is_some_and(|(header, payload)| {
- header.flags & AUDIO_FRAME_FLAG_END_OF_STREAM != 0
- && header.frame_count == 0
- && payload.is_empty()
- }) {
- relay_terminal_timestamp_micros = relay_terminal_timestamp_micros.max(
- system_frame
- .as_ref()
- .map(|(header, _)| header.timestamp_micros)
- .unwrap_or_default(),
+ let consume_system = matches!(
+ action,
+ MeetingAlignmentAction::Pair | MeetingAlignmentAction::SystemOnly
);
- system_done = true;
- system_frame = None;
- }
- if microphone_done && system_done {
- let terminal = AudioFrameHeader::new(
- 0,
+ let microphone_item = if consume_microphone {
+ microphone_frame.take()
+ } else {
+ stats.microphone_padding_frames =
+ stats.microphone_padding_frames.saturating_add(1);
+ None
+ };
+ let system_item = if consume_system {
+ system_frame.take()
+ } else {
+ stats.system_padding_frames = stats.system_padding_frames.saturating_add(1);
+ None
+ };
+ if let Some((_, payload)) = microphone_item.as_ref() {
+ pcm_i16_into(payload, &mut mic_samples)?;
+ } else {
+ mic_samples.clear();
+ mic_samples.resize(MEETING_AEC_FRAME_SAMPLES, 0);
+ }
+ if let Some((_, payload)) = system_item.as_ref() {
+ pcm_i16_into(payload, &mut system_samples)?;
+ } else {
+ system_samples.clear();
+ system_samples.resize(MEETING_AEC_FRAME_SAMPLES, 0);
+ }
+ if mic_samples.len() != MEETING_AEC_FRAME_SAMPLES
+ || system_samples.len() != MEETING_AEC_FRAME_SAMPLES
+ {
+ return Err("meeting LocalVQE received a non-10ms source frame".to_string());
+ }
+ downsample_meeting_48k_to_16k_into(&mic_samples, &mut microphone_16k)?;
+ downsample_meeting_48k_to_16k_into(&system_samples, &mut system_16k)?;
+ let timestamp_micros = match action {
+ MeetingAlignmentAction::Pair => microphone_timestamp
+ .unwrap_or_default()
+ .max(system_timestamp.unwrap_or_default()),
+ MeetingAlignmentAction::MicrophoneOnly => {
+ microphone_timestamp.unwrap_or_default()
+ }
+ MeetingAlignmentAction::SystemOnly => system_timestamp.unwrap_or_default(),
+ };
+ relay_terminal_timestamp_micros =
+ relay_terminal_timestamp_micros.max(timestamp_micros);
+ let microphone_eos = microphone_item.as_ref().is_some_and(|(header, _)| {
+ header.flags & AUDIO_FRAME_FLAG_END_OF_STREAM != 0
+ });
+ let system_eos = system_item.as_ref().is_some_and(|(header, _)| {
+ header.flags & AUDIO_FRAME_FLAG_END_OF_STREAM != 0
+ });
+ let mut combined_flags = microphone_item
+ .as_ref()
+ .map(|(header, _)| header.flags)
+ .unwrap_or_default()
+ | system_item
+ .as_ref()
+ .map(|(header, _)| header.flags)
+ .unwrap_or_default();
+ combined_flags &= !AUDIO_FRAME_FLAG_END_OF_STREAM;
+ let common_header = AudioFrameHeader::new(
+ (MEETING_OUTPUT_SAMPLES * 2) as u32,
relay_sequence,
- relay_terminal_timestamp_micros,
- 0,
+ timestamp_micros,
+ 160,
1,
- AUDIO_FRAME_FLAG_END_OF_STREAM,
+ combined_flags,
)
.map_err(|error| error.to_string())?;
- for handle in &outputs {
- stats.bytes_forwarded += write_meeting_frame(*handle, terminal, &[])?;
+ let frame = PendingMeetingFrame {
+ header: common_header,
+ microphone: microphone_16k
+ .as_slice()
+ .try_into()
+ .map_err(|_| "invalid microphone frame")?,
+ system: system_16k
+ .as_slice()
+ .try_into()
+ .map_err(|_| "invalid system frame")?,
+ };
+ if let Some(processor) = aec.as_mut() {
+ if pending.len() >= 4 {
+ return Err("LocalVQE exceeded bounded timeline buffer".into());
+ }
+ processor.push(&system_16k, µphone_16k)?;
+ pending.push_back(frame);
+ drain_meeting_enhancement(
+ processor,
+ &mut pending,
+ &outputs,
+ &mut stats,
+ &mut clean_16k,
+ &mut payload,
+ )?;
+ } else {
+ emit_meeting_frame(
+ &outputs,
+ frame,
+ µphone_16k,
+ false,
+ &mut stats,
+ &mut payload,
+ )?;
}
- break;
- }
- let microphone_timestamp = microphone_frame
- .as_ref()
- .map(|(header, _)| header.timestamp_micros);
- let system_timestamp = system_frame
- .as_ref()
- .map(|(header, _)| header.timestamp_micros);
- let Some(action) = meeting_alignment_action(microphone_timestamp, system_timestamp)
- else {
- break;
- };
- if let (Some(microphone), Some(system)) = (microphone_timestamp, system_timestamp) {
- stats.max_input_skew_micros =
- stats.max_input_skew_micros.max(microphone.abs_diff(system));
- }
- let consume_microphone = matches!(
- action,
- MeetingAlignmentAction::Pair | MeetingAlignmentAction::MicrophoneOnly
- );
- let consume_system = matches!(
- action,
- MeetingAlignmentAction::Pair | MeetingAlignmentAction::SystemOnly
- );
- let microphone_item = if consume_microphone {
- microphone_frame.take()
- } else {
- stats.microphone_padding_frames =
- stats.microphone_padding_frames.saturating_add(1);
- None
- };
- let system_item = if consume_system {
- system_frame.take()
- } else {
- stats.system_padding_frames = stats.system_padding_frames.saturating_add(1);
- None
- };
- if let Some((_, payload)) = microphone_item.as_ref() {
- pcm_i16_into(payload, &mut mic_samples)?;
- } else {
- mic_samples.clear();
- mic_samples.resize(MEETING_AEC_FRAME_SAMPLES, 0);
- }
- if let Some((_, payload)) = system_item.as_ref() {
- pcm_i16_into(payload, &mut system_samples)?;
- } else {
- system_samples.clear();
- system_samples.resize(MEETING_AEC_FRAME_SAMPLES, 0);
- }
- if mic_samples.len() != MEETING_AEC_FRAME_SAMPLES
- || system_samples.len() != MEETING_AEC_FRAME_SAMPLES
- {
- return Err("meeting AEC3 received a non-10ms source frame".to_string());
- }
- if let Some(processor) = aec.as_mut() {
- processor.process_into(&system_samples, &mic_samples, &mut clean_samples)?;
- } else {
- clean_samples.clear();
- clean_samples.extend_from_slice(&mic_samples);
- }
- let system_energy = meeting_frame_energy(&system_samples);
- if aec_enabled && system_energy >= meeting_render_active_energy_threshold() {
- stats.aec_render_active_frames =
- stats.aec_render_active_frames.saturating_add(1);
- stats.aec_render_energy += system_energy;
- stats.aec_raw_mic_energy += meeting_frame_energy(&mic_samples);
- stats.aec_clean_mic_energy += meeting_frame_energy(&clean_samples);
- }
- downsample_meeting_48k_to_16k_into(&mic_samples, &mut microphone_16k)?;
- downsample_meeting_48k_to_16k_into(&system_samples, &mut system_16k)?;
- downsample_meeting_48k_to_16k_into(&clean_samples, &mut clean_16k)?;
- pcm_bytes_into(µphone_16k, &mut microphone_payload);
- pcm_bytes_into(&system_16k, &mut system_payload);
- pcm_bytes_into(&clean_16k, &mut clean_payload);
- let timestamp_micros = match action {
- MeetingAlignmentAction::Pair => microphone_timestamp
- .unwrap_or_default()
- .max(system_timestamp.unwrap_or_default()),
- MeetingAlignmentAction::MicrophoneOnly => {
- microphone_timestamp.unwrap_or_default()
+ relay_sequence = relay_sequence.saturating_add(1);
+ if consume_microphone {
+ if microphone_eos {
+ microphone_done = true;
+ } else {
+ microphone_frame = Some(read_meeting_frame(microphone, &stop_rx)?);
+ }
}
- MeetingAlignmentAction::SystemOnly => system_timestamp.unwrap_or_default(),
- };
- relay_terminal_timestamp_micros =
- relay_terminal_timestamp_micros.max(timestamp_micros);
- let microphone_eos = microphone_item
- .as_ref()
- .is_some_and(|(header, _)| header.flags & AUDIO_FRAME_FLAG_END_OF_STREAM != 0);
- let system_eos = system_item
- .as_ref()
- .is_some_and(|(header, _)| header.flags & AUDIO_FRAME_FLAG_END_OF_STREAM != 0);
- let mut combined_flags = microphone_item
- .as_ref()
- .map(|(header, _)| header.flags)
- .unwrap_or_default()
- | system_item
- .as_ref()
- .map(|(header, _)| header.flags)
- .unwrap_or_default();
- combined_flags &= !AUDIO_FRAME_FLAG_END_OF_STREAM;
- let common_header = AudioFrameHeader::new(
- clean_payload.len() as u32,
- relay_sequence,
- timestamp_micros,
- 160,
- 1,
- combined_flags,
- )
- .map_err(|error| error.to_string())?;
- stats.bytes_forwarded +=
- write_meeting_frame(outputs[0], common_header, µphone_payload)?;
- stats.bytes_forwarded +=
- write_meeting_frame(outputs[1], common_header, &system_payload)?;
- let clean_header = common_header;
- stats.bytes_forwarded +=
- write_meeting_frame(outputs[2], clean_header, &clean_payload)?;
- stats.frames_processed += 1;
- relay_sequence = relay_sequence.saturating_add(1);
- if consume_microphone {
- if microphone_eos {
- microphone_done = true;
- } else {
- microphone_frame = Some(read_meeting_frame(microphone, &stop_rx)?);
+ if consume_system {
+ if system_eos {
+ system_done = true;
+ } else {
+ system_frame = Some(read_meeting_frame(system, &stop_rx)?);
+ }
}
}
- if consume_system {
- if system_eos {
- system_done = true;
- } else {
- system_frame = Some(read_meeting_frame(system, &stop_rx)?);
- }
+ Ok(())
+ })();
+ if let Err(error) = relay_result {
+ if error != "meetingRelayStopped" {
+ return Err(error);
}
}
+ // Explicit Stop/Pause interrupts the source read before source EOF.
+ // Flush accepted audio on that path too, before closing any output.
+ if let Some(processor) = aec.as_mut() {
+ processor.finish()?;
+ drain_meeting_enhancement(
+ processor,
+ &mut pending,
+ &outputs,
+ &mut stats,
+ &mut clean_16k,
+ &mut payload,
+ )?;
+ }
+ if !pending.is_empty() {
+ return Err("LocalVQE did not flush the complete capture timeline".into());
+ }
+ let terminal = AudioFrameHeader::new(
+ 0,
+ relay_sequence,
+ relay_terminal_timestamp_micros,
+ 0,
+ 1,
+ AUDIO_FRAME_FLAG_END_OF_STREAM,
+ )
+ .map_err(|error| error.to_string())?;
+ for handle in &outputs {
+ stats.bytes_forwarded += write_meeting_frame(*handle, terminal, &[])?;
+ }
Ok(())
})();
unsafe {
@@ -5510,9 +5605,21 @@ fn wide_null(value: &str) -> Vec {
}
fn self_test_payload() -> Value {
+ let enhancement = (|| -> Result<(), String> {
+ let mut processor = MeetingEnhancer::new()?;
+ processor.push(&[0; MEETING_OUTPUT_SAMPLES], &[0; MEETING_OUTPUT_SAMPLES])?;
+ processor.finish()?;
+ let mut output = Vec::with_capacity(MEETING_OUTPUT_SAMPLES);
+ if !processor.pop(&mut output) || output.len() != MEETING_OUTPUT_SAMPLES {
+ return Err("LocalVQE self-test did not produce a complete frame".into());
+ }
+ Ok(())
+ })();
json!({
"sidecar": SIDECAR_NAME,
- "ok": true,
+ "ok": enhancement.is_ok(),
+ "meetingEnhancementVerified": enhancement.is_ok(),
+ "error": enhancement.err(),
"workerVersion": env!("CARGO_PKG_VERSION"),
"protocolVersion": SIDECAR_PROTOCOL_VERSION,
"capabilities": capabilities_payload(),
@@ -5546,7 +5653,18 @@ fn capabilities_payload() -> Value {
"wasapiCaptureAvailable": wasapi_capture_enabled(),
"wasapiLoopbackAvailable": wasapi_capture_enabled(),
"meetingCaptureAvailable": wasapi_capture_enabled() || synthetic_capture_enabled(),
- "meetingAec3": {"available": true, "implementation": "aec3-rs", "version": "0.2.0"},
+ "meetingEnhancement": {
+ "available": scriber_localvqe::cpu_supported(),
+ "implementation": "LocalVQE",
+ "version": scriber_localvqe::MODEL_VERSION,
+ "modelSha256": scriber_localvqe::MODEL_SHA256,
+ "sampleRate": 16000,
+ "hopSamples": 256,
+ "echoCancellation": true,
+ "noiseSuppression": true,
+ "dereverberation": true,
+ "modelEmbedded": true
+ },
"wasapiCaptureEnv": WASAPI_CAPTURE_ENV,
"syntheticFramePipeAvailable": synthetic_capture_enabled(),
"syntheticFramePipeEnv": SYNTHETIC_CAPTURE_ENV,
@@ -6577,10 +6695,12 @@ mod tests {
assert_eq!(response["payload"]["aecActive"], true);
let sources = response["payload"]["sources"].as_array().unwrap();
assert_eq!(sources.len(), 3);
+ let (ready_tx, ready_rx) = mpsc::channel();
let readers: Vec<_> = sources
.iter()
.map(|source| {
let path = source["framePipe"].as_str().unwrap().to_string();
+ let ready = ready_tx.clone();
thread::spawn(move || {
let mut file = loop {
match std::fs::File::open(&path) {
@@ -6588,31 +6708,51 @@ mod tests {
Err(_) => thread::sleep(Duration::from_millis(5)),
}
};
- (0..8)
- .map(|_| {
- let mut header = [0u8; AUDIO_FRAME_HEADER_LEN];
- file.read_exact(&mut header).unwrap();
- let decoded = AudioFrameHeader::decode(&header).unwrap();
- let mut payload = vec![0u8; decoded.payload_len as usize];
- file.read_exact(&mut payload).unwrap();
- let peak = payload
- .chunks_exact(2)
- .map(|sample| {
- i16::from_le_bytes([sample[0], sample[1]]).unsigned_abs()
- })
- .max()
- .unwrap_or(0);
- (decoded, payload.len(), peak)
- })
- .collect::>()
+ let mut frames = Vec::new();
+ loop {
+ let mut header = [0u8; AUDIO_FRAME_HEADER_LEN];
+ file.read_exact(&mut header).unwrap();
+ let decoded = AudioFrameHeader::decode(&header).unwrap();
+ let mut payload = vec![0u8; decoded.payload_len as usize];
+ file.read_exact(&mut payload).unwrap();
+ let peak = payload
+ .chunks_exact(2)
+ .map(|sample| i16::from_le_bytes([sample[0], sample[1]]).unsigned_abs())
+ .max()
+ .unwrap_or(0);
+ frames.push((decoded, payload.len(), peak));
+ if frames.len() == 8 {
+ ready.send(()).unwrap();
+ }
+ if decoded.flags & AUDIO_FRAME_FLAG_END_OF_STREAM != 0 {
+ break;
+ }
+ }
+ frames
})
})
.collect();
+ for _ in 0..3 {
+ ready_rx.recv_timeout(Duration::from_secs(10)).unwrap();
+ }
+ let capture_id = response["payload"]["meetingCaptureId"].as_str().unwrap();
+ let stopped = state.stop_meeting_capture(capture_id, "meetingCaptureStop");
+ assert_eq!(stopped["stopped"], true);
+ assert_eq!(stopped["relay"]["relayError"], Value::Null, "{stopped}");
let mut received = Vec::new();
for reader in readers {
let frames = reader.join().unwrap();
- assert_eq!(frames.len(), 8);
- for (header, payload_len, _) in &frames {
+ assert!(frames.len() >= 9);
+ let (terminal, terminal_bytes, _) = frames.last().unwrap();
+ assert_eq!(*terminal_bytes, 0);
+ assert_eq!(terminal.frame_count, 0);
+ assert_ne!(terminal.flags & AUDIO_FRAME_FLAG_END_OF_STREAM, 0);
+ assert_eq!(terminal.sequence, (frames.len() - 1) as u64);
+ assert_eq!(
+ stopped["relay"]["framesProcessed"],
+ (frames.len() - 1) as u64
+ );
+ for (header, payload_len, _) in &frames[..frames.len() - 1] {
assert_eq!(header.frame_count, 160);
assert_eq!(*payload_len, 320);
}
@@ -6622,26 +6762,19 @@ mod tests {
);
received.push(frames);
}
- for frame_index in 0..8 {
+ assert!(received
+ .windows(2)
+ .all(|pair| pair[0].len() == pair[1].len()));
+ for frame_index in 0..received[0].len() {
assert!(received.windows(2).all(|pair| {
pair[0][frame_index].0.sequence == pair[1][frame_index].0.sequence
&& pair[0][frame_index].0.timestamp_micros
== pair[1][frame_index].0.timestamp_micros
}));
}
- let capture_id = response["payload"]["meetingCaptureId"].as_str().unwrap();
- let stop = json!({
- "protocolVersion": SIDECAR_PROTOCOL_VERSION,
- "requestId": "meeting-stop",
- "command": "meetingCaptureStop",
- "payload": {"meetingCaptureId": capture_id}
- });
- let stopped = state.handle_sidecar_request(&stop.to_string());
- assert_eq!(stopped["success"], true);
- assert_eq!(stopped["payload"]["stopped"], true);
- assert!(stopped["payload"]["relay"]["aecMetrics"].is_object());
+ assert!(stopped["relay"]["aecMetrics"].is_object());
assert_eq!(
- stopped["payload"]["relay"]["aecMetrics"]["measurement"],
+ stopped["relay"]["aecMetrics"]["measurement"],
"render-active-raw-to-clean-energy-ratio"
);
}
diff --git a/Frontend/src-tauri/src/meeting_aec.rs b/Frontend/src-tauri/src/meeting_aec.rs
index 36ad9974..9bf853b8 100644
--- a/Frontend/src-tauri/src/meeting_aec.rs
+++ b/Frontend/src-tauri/src/meeting_aec.rs
@@ -1,80 +1,148 @@
-//! Pinned pure-Rust WebRTC AEC3 adapter for 10 ms meeting frames.
+//! LocalVQE v1.3 streaming adapter. Transport is 10 ms; inference is 16 ms.
+//!
+//! Drop the native one-hop analysis pre-roll, retain original timestamps in the
+//! relay, and flush one zero hop at EOF. This avoids shifting clean speech or
+//! dropping the final partial hop. The model estimates acoustic delay itself.
+use scriber_localvqe::{Model, HOP};
+use std::collections::VecDeque;
-use aec3::{
- nodes::audio::AudioFormat,
- pipelines::linear::{self, LinearPipeline},
-};
-
-pub const MEETING_AEC_SAMPLE_RATE: u32 = 48_000;
pub const MEETING_AEC_FRAME_SAMPLES: usize = 480;
+pub const MEETING_OUTPUT_SAMPLES: usize = 160;
-pub struct MeetingAec3 {
- pipeline: LinearPipeline,
- render: Vec,
- capture: Vec,
- output: Vec,
+pub struct MeetingEnhancer {
+ model: Model,
+ stream: HopAdapter,
}
-impl MeetingAec3 {
- pub fn new(initial_delay_ms: i32) -> Result {
- let format = AudioFormat::ten_ms(MEETING_AEC_SAMPLE_RATE, 1);
- let pipeline = linear::builder(format, format)
- .initial_delay_ms(initial_delay_ms.clamp(0, 500))
- .enable_high_pass_filter(false)
- .enable_noise_suppression(false)
- .enable_gain_controller2(false)
- .build()
- .map_err(|error| format!("meeting AEC3 initialization failed: {error}"))?;
+impl MeetingEnhancer {
+ pub fn new() -> Result {
Ok(Self {
- pipeline,
- render: vec![0.0; MEETING_AEC_FRAME_SAMPLES],
- capture: vec![0.0; MEETING_AEC_FRAME_SAMPLES],
- output: vec![0.0; MEETING_AEC_FRAME_SAMPLES],
+ model: Model::new()?,
+ stream: HopAdapter::default(),
+ })
+ }
+
+ pub fn push(&mut self, render: &[i16], mic: &[i16]) -> Result<(), String> {
+ self.stream.push(render, mic, |mic, render, out| {
+ self.model.process(mic, render, out)
})
}
- #[cfg(test)]
- pub fn process(&mut self, render_pcm: &[i16], capture_pcm: &[i16]) -> Result, String> {
- let mut output = Vec::with_capacity(MEETING_AEC_FRAME_SAMPLES);
- self.process_into(render_pcm, capture_pcm, &mut output)?;
- Ok(output)
+ pub fn pop(&mut self, out: &mut Vec) -> bool {
+ self.stream.pop(out)
+ }
+
+ pub fn finish(&mut self) -> Result<(), String> {
+ self.stream
+ .finish(|mic, render, out| self.model.process(mic, render, out))
}
+}
+
+struct HopAdapter {
+ mic: [f32; HOP],
+ render: [f32; HOP],
+ used: usize,
+ primed: bool,
+ accepted: u64,
+ produced: u64,
+ output: VecDeque,
+ finished: bool,
+}
- pub fn process_into(
+impl Default for HopAdapter {
+ fn default() -> Self {
+ Self {
+ mic: [0.0; HOP],
+ render: [0.0; HOP],
+ used: 0,
+ primed: false,
+ accepted: 0,
+ produced: 0,
+ output: VecDeque::with_capacity(HOP * 3),
+ finished: false,
+ }
+ }
+}
+
+impl HopAdapter {
+ fn hop(
&mut self,
- render_pcm: &[i16],
- capture_pcm: &[i16],
- output_pcm: &mut Vec,
+ process: &mut impl FnMut(&[f32; HOP], &[f32; HOP], &mut [f32; HOP]) -> Result<(), String>,
) -> Result<(), String> {
- if render_pcm.len() != MEETING_AEC_FRAME_SAMPLES
- || capture_pcm.len() != MEETING_AEC_FRAME_SAMPLES
+ let mut output = [0.0; HOP];
+ process(&self.mic, &self.render, &mut output)?;
+ if self.primed {
+ let count = (self.accepted - self.produced).min(HOP as u64) as usize;
+ self.output.extend(output[..count].iter().map(|s| {
+ (s.clamp(-1.0, 1.0) * 32768.0)
+ .round()
+ .clamp(-32768.0, 32767.0) as i16
+ }));
+ self.produced += count as u64;
+ } else {
+ self.primed = true;
+ }
+ self.used = 0;
+ Ok(())
+ }
+
+ fn push(
+ &mut self,
+ render: &[i16],
+ mic: &[i16],
+ mut process: impl FnMut(&[f32; HOP], &[f32; HOP], &mut [f32; HOP]) -> Result<(), String>,
+ ) -> Result<(), String> {
+ if self.finished
+ || render.len() != MEETING_OUTPUT_SAMPLES
+ || mic.len() != MEETING_OUTPUT_SAMPLES
{
- return Err(format!(
- "meeting AEC3 requires {MEETING_AEC_FRAME_SAMPLES} samples per 10 ms frame"
- ));
+ return Err(
+ "LocalVQE requires an active stream and 160 samples per 10 ms frame".into(),
+ );
+ }
+ // A consumer must drain each push. Never grow with meeting duration.
+ if self.output.len() >= MEETING_OUTPUT_SAMPLES {
+ return Err("LocalVQE output was not drained".into());
+ }
+ self.accepted += mic.len() as u64;
+ for (&mic, &render) in mic.iter().zip(render) {
+ self.mic[self.used] = f32::from(mic) / 32768.0;
+ self.render[self.used] = f32::from(render) / 32768.0;
+ self.used += 1;
+ if self.used == HOP {
+ self.hop(&mut process)?;
+ }
+ }
+ Ok(())
+ }
+
+ fn pop(&mut self, out: &mut Vec) -> bool {
+ if self.output.len() < MEETING_OUTPUT_SAMPLES {
+ return false;
}
- for (target, sample) in self.render.iter_mut().zip(render_pcm) {
- *target = f32::from(*sample) / 32768.0;
+ out.clear();
+ out.extend(self.output.drain(..MEETING_OUTPUT_SAMPLES));
+ true
+ }
+
+ fn finish(
+ &mut self,
+ mut process: impl FnMut(&[f32; HOP], &[f32; HOP], &mut [f32; HOP]) -> Result<(), String>,
+ ) -> Result<(), String> {
+ if self.finished {
+ return Ok(());
}
- for (target, sample) in self.capture.iter_mut().zip(capture_pcm) {
- *target = f32::from(*sample) / 32768.0;
+ if self.used > 0 {
+ self.mic[self.used..].fill(0.0);
+ self.render[self.used..].fill(0.0);
+ self.hop(&mut process)?;
}
- self.pipeline
- .handle_render_frame(&self.render)
- .map_err(|error| format!("meeting AEC3 render processing failed: {error}"))?;
- let produced = self
- .pipeline
- .process_capture_frame(&self.capture, &mut self.output)
- .map_err(|error| format!("meeting AEC3 capture processing failed: {error}"))?;
- if !produced {
- return Err("meeting AEC3 did not produce a capture frame".to_string());
+ if self.produced < self.accepted {
+ self.mic.fill(0.0);
+ self.render.fill(0.0);
+ self.hop(&mut process)?;
}
- output_pcm.clear();
- output_pcm.extend(
- self.output
- .iter()
- .map(|sample| (sample.clamp(-1.0, 1.0) * 32767.0).round() as i16),
- );
+ self.finished = true;
Ok(())
}
}
@@ -82,125 +150,70 @@ impl MeetingAec3 {
#[cfg(test)]
mod tests {
use super::*;
- use std::collections::VecDeque;
-
- #[test]
- fn aec3_adapter_processes_exact_ten_ms_frames() {
- let mut processor = MeetingAec3::new(80).expect("AEC3 should initialize");
- let render = vec![0i16; MEETING_AEC_FRAME_SAMPLES];
- let capture = vec![1_000i16; MEETING_AEC_FRAME_SAMPLES];
- let output = processor
- .process(&render, &capture)
- .expect("AEC3 should produce output");
- assert_eq!(output.len(), MEETING_AEC_FRAME_SAMPLES);
- }
#[test]
- fn aec3_adapter_rejects_non_ten_ms_frames() {
- let mut processor = MeetingAec3::new(80).expect("AEC3 should initialize");
- assert!(processor.process(&[0; 10], &[0; 10]).is_err());
- }
-
- #[test]
- fn aec3_process_into_reuses_the_caller_buffer() {
- let mut processor = MeetingAec3::new(80).expect("AEC3 should initialize");
- let render = vec![0i16; MEETING_AEC_FRAME_SAMPLES];
- let capture = vec![1_000i16; MEETING_AEC_FRAME_SAMPLES];
- let mut output = Vec::with_capacity(MEETING_AEC_FRAME_SAMPLES);
- processor
- .process_into(&render, &capture, &mut output)
- .expect("first AEC3 frame");
- let allocation = output.as_ptr();
- for _ in 0..1_000 {
- processor
- .process_into(&render, &capture, &mut output)
- .expect("reused AEC3 frame");
- assert_eq!(output.as_ptr(), allocation);
- assert_eq!(output.len(), MEETING_AEC_FRAME_SAMPLES);
+ fn reblocking_preserves_every_sample_at_all_eof_offsets() {
+ // A one-hop delayed identity emulates native analysis/synthesis latency.
+ // Test every 10/16ms alignment, short streams, and a long recording.
+ for frames in (0..24).chain([10_003]) {
+ let mut stream = HopAdapter::default();
+ let mut previous = [0.0; HOP];
+ let mut process = |mic: &[f32; HOP], _: &[f32; HOP], out: &mut [f32; HOP]| {
+ *out = previous;
+ previous = *mic;
+ Ok(())
+ };
+ let mut expected = Vec::new();
+ let mut actual = Vec::new();
+ let mut output = Vec::new();
+ for frame in 0..frames {
+ let mic: Vec = (0..MEETING_OUTPUT_SAMPLES)
+ .map(|i| ((frame * 160 + i) % 60001) as i32 - 30000)
+ .map(|i| i as i16)
+ .collect();
+ expected.extend_from_slice(&mic);
+ stream
+ .push(&[0; MEETING_OUTPUT_SAMPLES], &mic, &mut process)
+ .unwrap();
+ while stream.pop(&mut output) {
+ actual.extend_from_slice(&output);
+ }
+ assert!(stream.output.len() < MEETING_OUTPUT_SAMPLES);
+ assert!(stream.accepted - stream.produced <= (HOP * 2) as u64);
+ }
+ stream.finish(&mut process).unwrap();
+ stream.finish(&mut process).unwrap();
+ while stream.pop(&mut output) {
+ actual.extend_from_slice(&output);
+ }
+ assert_eq!(actual, expected, "length/alignment at {frames} frames");
+ assert!(stream.output.is_empty());
+ assert!(stream.push(&[0; 160], &[0; 160], &mut process).is_err());
}
}
- fn deterministic_render(seed: &mut u32) -> Vec {
- (0..MEETING_AEC_FRAME_SAMPLES)
- .map(|_| {
- *seed = seed.wrapping_mul(1_664_525).wrapping_add(1_013_904_223);
- ((*seed >> 16) as i16) / 3
- })
- .collect()
- }
-
- fn energy(samples: &[i16]) -> f64 {
- samples
- .iter()
- .map(|sample| f64::from(*sample).powi(2))
- .sum::()
- / samples.len().max(1) as f64
- }
-
#[test]
- fn aec3_measurably_attenuates_delayed_render_echo() {
- let mut processor = MeetingAec3::new(80).expect("AEC3 should initialize");
- let mut history = VecDeque::from(vec![vec![0i16; MEETING_AEC_FRAME_SAMPLES]; 8]);
- let mut seed = 7u32;
- let mut input_energy = 0.0;
- let mut output_energy = 0.0;
- for frame_index in 0..900 {
- let render = deterministic_render(&mut seed);
- let delayed = history.pop_front().unwrap();
- history.push_back(render.clone());
- let capture: Vec = delayed
- .iter()
- .map(|sample| (*sample as f32 * 0.55) as i16)
- .collect();
- let output = processor.process(&render, &capture).expect("AEC3 frame");
- if frame_index >= 700 {
- input_energy += energy(&capture);
- output_energy += energy(&output);
- }
- }
- assert!(
- output_energy < input_energy * 0.35,
- "AEC3 residual ratio was {:.3}",
- output_energy / input_energy
- );
+ fn malformed_frames_and_processing_errors_are_visible() {
+ let mut stream = HopAdapter::default();
+ assert!(stream.push(&[0; 10], &[0; 160], |_, _, _| Ok(())).is_err());
+ stream.push(&[0; 160], &[0; 160], |_, _, _| Ok(())).unwrap();
+ assert!(stream
+ .push(&[0; 160], &[0; 160], |_, _, _| Err(
+ "inference failed".into()
+ ))
+ .is_err());
}
#[test]
- fn aec3_preserves_near_end_voice_during_double_talk() {
- let mut processor = MeetingAec3::new(80).expect("AEC3 should initialize");
- let mut history = VecDeque::from(vec![vec![0i16; MEETING_AEC_FRAME_SAMPLES]; 8]);
- let mut seed = 11u32;
- let mut local_energy = 0.0;
- let mut output_energy = 0.0;
- for frame_index in 0..900 {
- let render = deterministic_render(&mut seed);
- let delayed = history.pop_front().unwrap();
- history.push_back(render.clone());
- let local: Vec = (0..MEETING_AEC_FRAME_SAMPLES)
- .map(|sample| {
- let phase = ((frame_index * MEETING_AEC_FRAME_SAMPLES + sample) as f32)
- * 2.0
- * std::f32::consts::PI
- * 220.0
- / MEETING_AEC_SAMPLE_RATE as f32;
- (phase.sin() * 4_000.0) as i16
- })
- .collect();
- let capture: Vec = delayed
- .iter()
- .zip(&local)
- .map(|(echo, voice)| ((*echo as f32 * 0.55) as i16).saturating_add(*voice))
- .collect();
- let output = processor.process(&render, &capture).expect("AEC3 frame");
- if frame_index >= 700 {
- local_energy += energy(&local);
- output_energy += energy(&output);
- }
- }
- assert!(
- output_energy > local_energy * 0.25,
- "AEC3 removed too much near-end energy: {:.3}",
- output_energy / local_energy
- );
+ fn actual_model_streams_and_flushes_short_capture() {
+ let mut enhancer = MeetingEnhancer::new().unwrap();
+ let mut output = Vec::new();
+ enhancer.push(&[0; 160], &[0; 160]).unwrap();
+ assert!(!enhancer.pop(&mut output));
+ enhancer.finish().unwrap();
+ assert!(enhancer.pop(&mut output));
+ assert_eq!(output.len(), 160);
+ assert!(output.iter().all(|s| s.abs() <= 2));
+ assert!(!enhancer.pop(&mut output));
}
}
diff --git a/README.md b/README.md
index 7f73c65e..272ead78 100644
--- a/README.md
+++ b/README.md
@@ -29,7 +29,8 @@
## Latest highlights
- **Bot-free Meetings:** record microphone and Windows system audio on one
- timeline, remove speaker echo with AEC3, and create transcripts, summaries,
+ timeline, reduce echo, noise, and reverberation with LocalVQE v1.3, and create
+ transcripts, summaries,
decisions, action items, cited answers, and reusable exports.
- **Smarter speaker workflows:** use provider-native speaker turns or optional
offline Sherpa-ONNX diarization, then confirm names with Meeting-local labels,
@@ -159,8 +160,10 @@ Drop in audio, video, or several files at once. Scriber extracts audio, compress
### 👥 Capture meetings without a joining bot
The Meetings workspace records microphone and Windows system audio locally while
-the call is in progress. WebRTC AEC3 uses the system-audio reference to remove
-speaker echo from the microphone track; the raw source is still retained for
+the call is in progress. LocalVQE v1.3 uses the system-audio reference to reduce
+speaker echo, background noise, and reverberation in the microphone track. The
+single model runs locally on the CPU and is included in the app; the raw source
+is retained for
recovery. After stop, Scriber creates a timestamped canonical transcript,
summary, decisions, action items, cited chat answers, and reusable exports.
diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md
index 637b315c..4ab0f317 100644
--- a/THIRD_PARTY_NOTICES.md
+++ b/THIRD_PARTY_NOTICES.md
@@ -50,22 +50,226 @@ applicable Gemma terms, prohibited-use policy, model notice, and modification
notice beside the model artifacts. Use and redistribution remain subject to
those terms.
-## aec3 0.2.0
+## LocalVQE 1.3
-- Project: `aec3-rs`
-- Source: https://github.com/RubyBit/aec3-rs
-- Use in Scriber: WebRTC AEC3 processing in the crash-isolated meeting audio sidecar
-- License: MIT OR BSD-3-Clause; portions are derived from the WebRTC project
-- Pinned package: `aec3 = "=0.2.0"`
+- Project: LocalVQE by LocalAI
+- Source: https://github.com/localai-org/LocalVQE
+- Source revision: `f53063c9eb2a85f96479867d1dd911dc3bf6319b`
+- Model: `LocalAI-io/LocalVQE`, `localvqe-v1.3-4.8M-f32.gguf`
+- Model revision: `29ca38495cba9d6393a92a4dd890f28dd81f758d`
+- License: Apache-2.0 (code and model)
+- Use: CPU meeting echo cancellation, noise suppression, and dereverberation
+- Modifications: static CPU-only build, bounded Rust streaming adapter,
+ UTF-8 Windows file access, mandatory integrity checking, quiet initialization,
+ and disabled runtime backend discovery. Weights are embedded unchanged.
-Copyright (c) 2025 Angelos-Ermis Mangos.
+Apache License
+ Version 2.0, January 2004
+ http://www.apache.org/licenses/
-Permission is hereby granted, free of charge, to any person obtaining a copy of
-this software and associated documentation files (the "Software"), to deal in
-the Software without restriction, including without limitation the rights to
-use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
-the Software, and to permit persons to whom the Software is furnished to do so,
-subject to the following conditions:
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
+
+ 1. Definitions.
+
+ "License" shall mean the terms and conditions for use, reproduction,
+ and distribution as defined by Sections 1 through 9 of this document.
+
+ "Licensor" shall mean the copyright owner or entity authorized by
+ the copyright owner that is granting the License.
+
+ "Legal Entity" shall mean the union of the acting entity and all
+ other entities that control, are controlled by, or are under common
+ control with that entity. For the purposes of this definition,
+ "control" means (i) the power, direct or indirect, to cause the
+ direction or management of such entity, whether by contract or
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
+ outstanding shares, or (iii) beneficial ownership of such entity.
+
+ "You" (or "Your") shall mean an individual or Legal Entity
+ exercising permissions granted by this License.
+
+ "Source" form shall mean the preferred form for making modifications,
+ including but not limited to software source code, documentation
+ source, and configuration files.
+
+ "Object" form shall mean any form resulting from mechanical
+ transformation or translation of a Source form, including but
+ not limited to compiled object code, generated documentation,
+ and conversions to other media types.
+
+ "Work" shall mean the work of authorship, whether in Source or
+ Object form, made available under the License, as indicated by a
+ copyright notice that is included in or attached to the work
+ (an example is provided in the Appendix below).
+
+ "Derivative Works" shall mean any work, whether in Source or Object
+ form, that is based on (or derived from) the Work and for which the
+ editorial revisions, annotations, elaborations, or other modifications
+ represent, as a whole, an original work of authorship. For the purposes
+ of this License, Derivative Works shall not include works that remain
+ separable from, or merely link (or bind by name) to the interfaces of,
+ the Work and Derivative Works thereof.
+
+ "Contribution" shall mean any work of authorship, including
+ the original version of the Work and any modifications or additions
+ to that Work or Derivative Works thereof, that is intentionally
+ submitted to the Licensor for inclusion in the Work by the copyright owner
+ or by an individual or Legal Entity authorized to submit on behalf of
+ the copyright owner. For the purposes of this definition, "submitted"
+ means any form of electronic, verbal, or written communication sent
+ to the Licensor or its representatives, including but not limited to
+ communication on electronic mailing lists, source code control systems,
+ and issue tracking systems that are managed by, or on behalf of, the
+ Licensor for the purpose of discussing and improving the Work, but
+ excluding communication that is conspicuously marked or otherwise
+ designated in writing by the copyright owner as "Not a Contribution."
+
+ "Contributor" shall mean Licensor and any individual or Legal Entity
+ on behalf of whom a Contribution has been received by the Licensor and
+ subsequently incorporated within the Work.
+
+ 2. Grant of Copyright License. Subject to the terms and conditions of
+ this License, each Contributor hereby grants to You a perpetual,
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+ copyright license to reproduce, prepare Derivative Works of,
+ publicly display, publicly perform, sublicense, and distribute the
+ Work and such Derivative Works in Source or Object form.
+
+ 3. Grant of Patent License. Subject to the terms and conditions of
+ this License, each Contributor hereby grants to You a perpetual,
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+ (except as stated in this section) patent license to make, have made,
+ use, offer to sell, sell, import, and otherwise transfer the Work,
+ where such license applies only to those patent claims licensable
+ by such Contributor that are necessarily infringed by their
+ Contribution(s) alone or by combination of their Contribution(s)
+ with the Work to which such Contribution(s) was submitted. If You
+ institute patent litigation against any entity (including a
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
+ or a Contribution incorporated within the Work constitutes direct
+ or contributory patent infringement, then any patent licenses
+ granted to You under this License for that Work shall terminate
+ as of the date such litigation is filed.
+
+ 4. Redistribution. You may reproduce and distribute copies of the
+ Work or Derivative Works thereof in any medium, with or without
+ modifications, and in Source or Object form, provided that You
+ meet the following conditions:
+
+ (a) You must give any other recipients of the Work or
+ Derivative Works a copy of this License; and
+
+ (b) You must cause any modified files to carry prominent notices
+ stating that You changed the files; and
+
+ (c) You must retain, in the Source form of any Derivative Works
+ that You distribute, all copyright, patent, trademark, and
+ attribution notices from the Source form of the Work,
+ excluding those notices that do not pertain to any part of
+ the Derivative Works; and
+
+ (d) If the Work includes a "NOTICE" text file as part of its
+ distribution, then any Derivative Works that You distribute must
+ include a readable copy of the attribution notices contained
+ within such NOTICE file, excluding any notices that do not
+ pertain to any part of the Derivative Works, in at least one
+ of the following places: within a NOTICE text file distributed
+ as part of the Derivative Works; within the Source form or
+ documentation, if provided along with the Derivative Works; or,
+ within a display generated by the Derivative Works, if and
+ wherever such third-party notices normally appear. The contents
+ of the NOTICE file are for informational purposes only and
+ do not modify the License. You may add Your own attribution
+ notices within Derivative Works that You distribute, alongside
+ or as an addendum to the NOTICE text from the Work, provided
+ that such additional attribution notices cannot be construed
+ as modifying the License.
+
+ You may add Your own copyright statement to Your modifications and
+ may provide additional or different license terms and conditions
+ for use, reproduction, or distribution of Your modifications, or
+ for any such Derivative Works as a whole, provided Your use,
+ reproduction, and distribution of the Work otherwise complies with
+ the conditions stated in this License.
+
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
+ any Contribution intentionally submitted for inclusion in the Work
+ by You to the Licensor shall be under the terms and conditions of
+ this License, without any additional terms or conditions.
+ Notwithstanding the above, nothing herein shall supersede or modify
+ the terms of any separate license agreement you may have executed
+ with Licensor regarding such Contributions.
+
+ 6. Trademarks. This License does not grant permission to use the trade
+ names, trademarks, service marks, or product names of the Licensor,
+ except as required for reasonable and customary use in describing the
+ origin of the Work and reproducing the content of the NOTICE file.
+
+ 7. Disclaimer of Warranty. Unless required by applicable law or
+ agreed to in writing, Licensor provides the Work (and each
+ Contributor provides its Contributions) on an "AS IS" BASIS,
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
+ implied, including, without limitation, any warranties or conditions
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
+ PARTICULAR PURPOSE. You are solely responsible for determining the
+ appropriateness of using or redistributing the Work and assume any
+ risks associated with Your exercise of permissions under this License.
+
+ 8. Limitation of Liability. In no event and under no legal theory,
+ whether in tort (including negligence), contract, or otherwise,
+ unless required by applicable law (such as deliberate and grossly
+ negligent acts) or agreed to in writing, shall any Contributor be
+ liable to You for damages, including any direct, indirect, special,
+ incidental, or consequential damages of any character arising as a
+ result of this License or out of the use or inability to use the
+ Work (including but not limited to damages for loss of goodwill,
+ work stoppage, computer failure or malfunction, or any and all
+ other commercial damages or losses), even if such Contributor
+ has been advised of the possibility of such damages.
+
+ 9. Accepting Warranty or Additional Liability. While redistributing
+ the Work or Derivative Works thereof, You may choose to offer,
+ and charge a fee for, acceptance of support, warranty, indemnity,
+ or other liability obligations and/or rights consistent with this
+ License. However, in accepting such obligations, You may act only
+ on Your own behalf and on Your sole responsibility, not on behalf
+ of any other Contributor, and only if You agree to indemnify,
+ defend, and hold each Contributor harmless for any liability
+ incurred by, or claims asserted against, such Contributor by reason
+ of your accepting any such warranty or additional liability.
+
+ END OF TERMS AND CONDITIONS
+
+ Copyright 2024-2026 Richard Sherwood Palethorpe
+
+ Licensed under the Apache License, Version 2.0 (the "License");
+ you may not use this file except in compliance with the License.
+ You may obtain a copy of the License at
+
+ http://www.apache.org/licenses/LICENSE-2.0
+
+ Unless required by applicable law or agreed to in writing, software
+ distributed under the License is distributed on an "AS IS" BASIS,
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ See the License for the specific language governing permissions and
+ limitations under the License.
+
+## GGML (LocalVQE runtime)
+
+- Source: https://github.com/ggml-org/ggml
+- Revision: `c044a8eeae2591faa0950c8b5e514cbc4bbfc4ca`
+- Use: statically linked CPU inference for LocalVQE v1.3
+
+MIT License
+
+Copyright (c) 2023-2026 The ggml authors
+
+Permission is hereby granted, free of charge, to any person obtaining a copy
+of this software and associated documentation files (the "Software"), to deal
+in the Software without restriction, including without limitation the rights
+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+copies of the Software, and to permit persons to whom the Software is
+furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
@@ -78,36 +282,6 @@ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.
-WebRTC-derived portions:
-
-Copyright (c) 2011, The WebRTC project authors. All rights reserved.
-
-Redistribution and use in source and binary forms, with or without
-modification, are permitted provided that the following conditions are met:
-
-1. Redistributions of source code must retain the above copyright notice, this
- list of conditions and the following disclaimer.
-2. Redistributions in binary form must reproduce the above copyright notice,
- this list of conditions and the following disclaimer in the documentation
- and/or other materials provided with the distribution.
-3. Neither the name of Google nor the names of its contributors may be used to
- endorse or promote products derived from this software without specific
- prior written permission.
-
-THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
-AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
-IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
-DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
-FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
-DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
-SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
-CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
-OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
-OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
-
-The upstream patent grant is published at:
-https://github.com/RubyBit/aec3-rs/blob/v0.2.0/PATENT
-
## Optional WeSpeaker speaker-embedding model
- Model: `talatapp/wespeaker-voxceleb-resnet34-LM-onnx`
diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md
index 7130aea5..e856f88d 100644
--- a/docs/ARCHITECTURE.md
+++ b/docs/ARCHITECTURE.md
@@ -216,9 +216,14 @@ Meetings:
final pass; `final_only` records locally and opens no live STT connection.
2. One crash-isolated Rust audio-sidecar process opens WASAPI microphone and
loopback sources at 48 kHz in shared mode; it never opens a camera/video
- device. Pinned `aec3-rs` consumes the loopback render
- reference and produces a cleaned microphone stream before all three tracks
- are downsampled to 16 kHz and stamped on one monotonic timeline.
+ device. The microphone and loopback frames are downsampled to 16 kHz before
+ the single pinned LocalVQE v1.3 model reduces echo, noise, and reverberation.
+ Its 256-sample inference hops are adapted to the 160-sample transport frames;
+ the relay holds at most four frames, removes the native one-hop pre-roll,
+ and flushes the final partial hop before EOF. All three tracks retain the
+ original capture timestamps and sample counts. The old `aecEnabled` switch
+ controls this joint enhancement; the legacy `aecDelayMs` input is accepted
+ for compatibility but LocalVQE estimates delay internally.
Its private endpoint inventory covers capture and render flows; the
token-protected Meeting API exposes only friendly labels and hashed IDs for
explicit route selection. Its explicit local device test reuses the same
@@ -331,7 +336,7 @@ Meetings:
Audio format is intentionally tiered rather than conflated with model input:
-- AEC3, Silero, Smart Turn, live STT, and the durable capture boundary consume
+- LocalVQE v1.3, Silero, Smart Turn, live STT, and the durable capture boundary consume
PCM frames. Compression is never inserted into this latency-sensitive branch.
- Recoverable 30-second work chunks remain 16-kHz mono PCM until their two-phase
publication and the canonical transcript commit are proven durable.
@@ -2019,7 +2024,7 @@ is enabled, session teardown schedules a background refill with brand-new
instances so later hotkeys retain the warm-start benefit.
The Settings page has a dedicated Meetings section. It snapshots the selected
-final STT provider, analysis model, Smart Turn, AEC3, automatic-analysis, and
+final STT provider, analysis model, Smart Turn, LocalVQE v1.3, automatic-analysis, and
audio-retention defaults into each new meeting. Final STT choices expose the
exact model plus native timestamp and diarization capability; changing a
default never mutates an existing meeting's pipeline snapshot.
diff --git a/docs/PERFORMANCE_AND_PACKAGING.md b/docs/PERFORMANCE_AND_PACKAGING.md
index 1ddf85af..6c1cad48 100644
--- a/docs/PERFORMANCE_AND_PACKAGING.md
+++ b/docs/PERFORMANCE_AND_PACKAGING.md
@@ -1763,11 +1763,13 @@ route remains below its documented 2-GB boundary and has no audio-duration cap.
**Implemented allocation-reuse slice**
- Caller-owned `Vec` buffers are reused for microphone, system, and clean
- AEC samples. AEC3 writes into the caller's existing output buffer.
-- Three 48-to-16-kHz downsample buffers and their three PCM byte buffers are
- allocated once per relay session and cleared without releasing capacity.
+ audio samples. LocalVQE v1.3 uses bounded hop buffers; the relay retains at
+ most four transport frames while waiting for aligned enhancement output.
+- Two 48-to-16-kHz downsample buffers, the clean output buffer, and one shared
+ PCM byte buffer are allocated once per relay session and reused.
- A 10,000-frame regression test verifies stable backing allocations and exact
- output sizes; the AEC adapter has an independent 1,000-frame reuse test.
+ output sizes; the enhancement adapter separately tests all 10/16-ms tail
+ alignments, short captures, and a 10,003-frame stream without sample loss.
- Upstream frame payload ownership and `WasapiPcmConverter` compaction remain
candidates for a future measured pass. The long installed CPU/jitter gate is
still required before claiming a device-level percentage improvement.
@@ -3480,7 +3482,7 @@ Free-threaded CPython is intentionally outside this profile: it excludes the
3.14 JIT and would require a separate complete `cp314t` native-wheel graph.
The installed Meeting hot-path qualification includes an exact 60-second
-physical WASAPI/AEC3 soak. Its Python level probe no longer loops over every
+physical WASAPI/LocalVQE v1.3 soak. Its Python level probe no longer loops over every
signed-16 sample with `int.from_bytes`; shared `audioop-lts` PCM metrics perform
RMS and peak work in C while preserving the existing signed little-endian and
trailing-partial-byte semantics. The gate additionally reports payload bytes,
@@ -3489,9 +3491,30 @@ growth, three-source frame continuity, and zero persisted/provider audio.
## Meeting Audio Packaging
-- `aec3 = 0.2.0` is compiled into the existing crash-isolated Rust audio
- sidecar; meeting capture does not add another executable or a GStreamer/Clang
- runtime dependency.
+- `native/scriber-localvqe` statically links LocalVQE v1.3 and its pinned GGML
+ CPU backend into the existing crash-isolated audio sidecar. Its only model is
+ the Apache-2.0 4.8M F32 model (about 19 MB), embedded in the executable.
+ `inputs.json` locks both source archives and the model by revision and SHA-256;
+ CMake fetches them only at build time. No account or runtime download is needed.
+ The file-only model loader uses a temporary copy of the embedded public weights
+ during construction and deletes it after weights are loaded into memory.
+ The native build disables runtime backend discovery, hash-bypass overrides,
+ OpenMP, GPU backends, and host-specific instruction tuning. It supports UTF-8
+ Windows model paths and keeps model initialization off the JSON stdout channel.
+ No additional executable or inference DLL is installed. Build prerequisites
+ now include CMake 3.24+ and a C++17 compiler alongside the Windows SDK.
+ The fixed x64 inference target requires AVX2, FMA, and F16C; Rust checks these
+ before starting enhanced capture and reports unavailability on older CPUs.
+ The C++ standard library and exception support are linked statically so no
+ new inference or C++ redistributable DLL is required.
+- Local validation on 2026-09-28: the Windows release build passes the pinned
+ upstream v1.3 F32 regression fixture (maximum absolute error about 2.05e-7).
+ On a Ryzen AI 9 HX 370, the 992-ms fixture took about 191 ms for CPU inference
+ with up to four threads, excluding model initialization. This short synthetic
+ reference check is not a physical-meeting quality or long-duration CPU gate.
+ The release binary's self-test loads the embedded model and performs inference;
+ Rust pipe tests cover synchronized three-track output and Stop-tail flushing.
+ Physical Teams/Zoom/Meet qualification remains required before release.
- `meeting_aec.rs` is part of the Rust-audio-sidecar cache key, so AEC changes
invalidate that focused artifact without unnecessarily invalidating the
Python backend cache.
diff --git a/docs/ROADMAP_AND_KNOWN_ISSUES.md b/docs/ROADMAP_AND_KNOWN_ISSUES.md
index b78b1bd6..4e216d7d 100644
--- a/docs/ROADMAP_AND_KNOWN_ISSUES.md
+++ b/docs/ROADMAP_AND_KNOWN_ISSUES.md
@@ -235,7 +235,7 @@ Meetings:
Meeting notes use a serialized, coalescing save lane with retry and
page-teardown flushing.
- Native meeting capture uses one Rust audio sidecar for mic plus loopback,
- pinned `aec3-rs` echo cancellation, a shared monotonic timeline, three durable
+ pinned LocalVQE v1.3 joint audio enhancement, a shared monotonic timeline, three durable
tracks, health monitoring, pause/resume gaps, and checksum-validated chunks.
- Pause, stop, cleanup, and device reconnect now arm the recorder before native
pipes close, so Windows `OSError` disconnects commit the valid partial chunk
@@ -917,7 +917,7 @@ routes, sleep/resume, long meetings, network loss/recovery, Outlook tenant
types, and installer upgrade/uninstall retention. The optional WeSpeaker model
also remains behind a commercial/legal review because of its VoxCeleb training
data terms. These are release evidence gates, not missing fallback capture
-paths; the normal Live Mic workflow intentionally does not enable AEC3 without
+paths; the normal Live Mic workflow intentionally does not enable LocalVQE v1.3 without
a render reference. `scripts\run_meeting_release_matrix.ps1` now prepares 19
atomic non-passing operator drafts and
`scripts\validate_meeting_release_matrix.py` validates completed evidence; the
@@ -947,7 +947,7 @@ screen-reader conformance, 200% zoom, localization, or release readiness.
#### Do not rebuild the existing baseline
-The current product already has Mic/System/AEC3 capture, the explicit route
+The current product already has Mic/System/LocalVQE v1.3 capture, the explicit route
test, pause/resume/stop, 30-second checkpoints, reconnect health, durable import
and recovery, Outlook connect/status, transparent live/final/analysis model
labels, Voice Library controls, Overview/Decisions/Actions/Questions/Notes/Ask
@@ -1304,7 +1304,7 @@ of code, prompts, assets, or schemas.
| Axis | Meetily evidence | Scriber evidence and decision |
| --- | --- | --- |
| Product boundary | Local Whisper/Parakeet transcription, optional local or cloud summary providers, import/retranscription, transcript recovery, templates, and a compact two-pane Meeting view | Scriber already covers bot-free capture, local/cloud transcription, import/reprocessing, recovery, speaker review, Outlook context, notes, Ask, exports, email drafts, and delivery. Preserve that broader workflow; borrow only interactions that shorten review. |
-| Audio capture | One in-process Rust `RecordingManager`, process-global `Mutex