From 7b7406178876652b5320116625218f5e90a27803 Mon Sep 17 00:00:00 2001 From: MythEclipse Date: Tue, 9 Jun 2026 01:07:35 +0700 Subject: [PATCH] perf(voice): reduce transmit latency MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - FFmpeg: frame_duration 10ms, compression_level 0, nobuffer/low_delay flags, minimal probe - Frontend: buffer 4096→1024 samples (~170ms→~42ms), faster base64 with chunking - Remove processor.connect(destination) to avoid audio feedback loop Co-Authored-By: Claude Opus 4.8 --- .../src/modules/voice-recording/transmitter.ts | 10 +++++++++- .../src/shared/hooks/useAudioTransmit.ts | 18 +++++++++--------- 2 files changed, 18 insertions(+), 10 deletions(-) diff --git a/services/discord-gateway/src/modules/voice-recording/transmitter.ts b/services/discord-gateway/src/modules/voice-recording/transmitter.ts index ca4f9cf..8bea79e 100644 --- a/services/discord-gateway/src/modules/voice-recording/transmitter.ts +++ b/services/discord-gateway/src/modules/voice-recording/transmitter.ts @@ -50,8 +50,16 @@ export class VoiceTransmitter { "-ar", "48000", // Output sample rate: 48kHz "-ac", "2", // Output channels: stereo "-application", "lowdelay", // Low delay mode for real-time - "-frame_duration", "20", // 20ms frames + "-frame_duration", "10", // 10ms frames (smaller = less latency) "-packet_loss", "0", // No packet loss expected + "-compression_level", "0", // Fastest compression (lowest CPU latency) + "-cutoff", "20000", // Audio cutoff frequency + "-vsync", "0", // No video sync + "-fflags", "nobuffer", // No buffering + "-flags", "low_delay", // Low delay processing + "-strict", "experimental", + "-analyzeduration", "0", // No analysis duration + "-probesize", "32", // Minimal probe size "pipe:1", // Write to stdout ], { stdio: ["pipe", "pipe", "pipe"], diff --git a/services/frontend/src/shared/hooks/useAudioTransmit.ts b/services/frontend/src/shared/hooks/useAudioTransmit.ts index 91f839e..d80076b 100644 --- a/services/frontend/src/shared/hooks/useAudioTransmit.ts +++ b/services/frontend/src/shared/hooks/useAudioTransmit.ts @@ -58,10 +58,10 @@ export function useAudioTransmit(socketRef: { const audioContext = new AudioContextCtor({ sampleRate: SAMPLE_RATE }); audioContextRef.current = audioContext; const source = audioContext.createMediaStreamSource(stream); - const processor = audioContext.createScriptProcessor(4096, 1, 1); + const processor = audioContext.createScriptProcessor(1024, 1, 1); processorRef.current = processor; source.connect(processor); - processor.connect(audioContext.destination); + // Don't connect processor to destination (no monitoring feedback) processor.onaudioprocess = (event) => { if (!socketRef.current || socketRef.current.readyState !== WebSocket.OPEN) return; @@ -70,15 +70,15 @@ export function useAudioTransmit(socketRef: { for (let i = 0; i < inputData.length; i++) pcmData[i] = Math.max(-1, Math.min(1, inputData[i])) * 32767; - // Convert to base64 - const bytes = new Uint8Array(pcmData.buffer); - let binary = ''; - for (let i = 0; i < bytes.length; i++) { - binary += String.fromCharCode(bytes[i]); + // Fast base64 using Uint8Array + btoa + const uint8 = new Uint8Array(pcmData.buffer); + let base64 = ''; + const chunkSize = 8192; + for (let i = 0; i < uint8.length; i += chunkSize) { + const chunk = uint8.subarray(i, i + chunkSize); + base64 += btoa(String.fromCharCode(...chunk)); } - const base64 = btoa(binary); - // Send as JSON for backend to forward to Redis socketRef.current.send(JSON.stringify({ type: 'voice_transmit', buffer: base64