From bd082a26bf73de601faba82a63f14d13405eba19 Mon Sep 17 00:00:00 2001 From: evgeny Date: Fri, 2 Oct 2026 10:22:34 +0300 Subject: [PATCH] =?UTF-8?q?=D0=A0=D0=B0=D0=B7=D0=B4=D0=B5=D0=BB=D0=B8=20?= =?UTF-8?q?=D0=B0=D1=83=D0=B4=D0=B8=D0=BE=D0=BF=D0=BE=D1=82=D0=BE=D0=BA?= =?UTF-8?q?=D0=B8=20=D0=B8=20=D0=BF=D1=80=D0=B8=D0=B2=D1=8F=D0=B6=D0=B8=20?= =?UTF-8?q?AEC=20=D0=BA=20=D1=84=D0=B0=D0=BA=D1=82=D0=B8=D1=87=D0=B5=D1=81?= =?UTF-8?q?=D0=BA=D0=BE=D0=BC=D1=83=20=D0=BE=D0=B1=D1=89=D0=B5=D0=BC=D1=83?= =?UTF-8?q?=20=D0=B2=D1=8B=D1=85=D0=BE=D0=B4=D1=83?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- lib/miniaudio.h | 31 +- lib/speex_aec.c | 38 +- lib/speex_aec.h | 11 +- src/call/call_audio.c | 49 ++- src/call/call_audio.h | 13 +- src/radio/radio_audio.c | 284 +++++++++------ src/radio/radio_audio.h | 15 +- tests/test_speex_aec.c | 39 ++- tools/chatgui/CMakeLists.txt | 35 +- tools/chatgui/src/audio_device.cpp | 35 ++ tools/chatgui/src/audio_device.h | 11 + tools/chatgui/src/audio_event_queue.h | 54 +++ tools/chatgui/src/audiodevicesettingspage.cpp | 2 +- tools/chatgui/src/audiorecorder.cpp | 86 +++-- tools/chatgui/src/audiorecorder.h | 14 +- tools/chatgui/src/call_audio_engine.cpp | 103 +++--- tools/chatgui/src/call_audio_engine.h | 24 +- tools/chatgui/src/radio_audio_engine.cpp | 95 ++--- tools/chatgui/src/radio_audio_engine.h | 29 +- tools/chatgui/src/sound_manager.cpp | 304 +++++++++++++--- tools/chatgui/src/sound_manager.h | 29 +- tools/chatgui/src/voice_audio_io.cpp | 331 ++++++++++++++++++ tools/chatgui/src/voice_audio_io.h | 61 ++++ tools/chatgui/tests/run_pulse_recovery.py | 25 +- .../chatgui/tests/test_audio_event_queue.cpp | 50 +++ tools/chatgui/tests/test_audio_recovery.cpp | 56 ++- tools/chatgui/tests/test_audio_timing.c | 57 +++ tools/chatgui/tests/test_voice_audio_io.cpp | 106 ++++++ tools/chatgui/tests/test_voice_tx.cpp | 143 ++++++++ tools/chatgui/transport/miniaudio_impl.c | 119 +++++++ 30 files changed, 1838 insertions(+), 411 deletions(-) create mode 100644 tools/chatgui/src/audio_event_queue.h create mode 100644 tools/chatgui/src/voice_audio_io.cpp create mode 100644 tools/chatgui/src/voice_audio_io.h create mode 100644 tools/chatgui/tests/test_audio_event_queue.cpp create mode 100644 tools/chatgui/tests/test_audio_timing.c create mode 100644 tools/chatgui/tests/test_voice_audio_io.cpp create mode 100644 tools/chatgui/tests/test_voice_tx.cpp diff --git a/lib/miniaudio.h b/lib/miniaudio.h index 49b8c5e7..6ac3795f 100644 --- a/lib/miniaudio.h +++ b/lib/miniaudio.h @@ -7565,6 +7565,7 @@ struct ma_context ma_proc pa_stream_get_buffer_attr; ma_proc pa_stream_set_buffer_attr; ma_proc pa_stream_get_device_name; + ma_proc pa_stream_get_latency; ma_proc pa_stream_set_write_callback; ma_proc pa_stream_set_read_callback; ma_proc pa_stream_set_suspended_callback; @@ -30959,6 +30960,7 @@ typedef const ma_pa_sample_spec* (* ma_pa_stream_get_sample_spec_proc) ( typedef const ma_pa_channel_map* (* ma_pa_stream_get_channel_map_proc) (ma_pa_stream* s); typedef const ma_pa_buffer_attr* (* ma_pa_stream_get_buffer_attr_proc) (ma_pa_stream* s); typedef ma_pa_operation* (* ma_pa_stream_set_buffer_attr_proc) (ma_pa_stream* s, const ma_pa_buffer_attr* attr, ma_pa_stream_success_cb_t cb, void* userdata); +typedef int (* ma_pa_stream_get_latency_proc)(const ma_pa_stream* s, ma_uint64* usec, int* negative); typedef const char* (* ma_pa_stream_get_device_name_proc) (const ma_pa_stream* s); typedef void (* ma_pa_stream_set_write_callback_proc) (ma_pa_stream* s, ma_pa_stream_request_cb_t cb, void* userdata); typedef void (* ma_pa_stream_set_read_callback_proc) (ma_pa_stream* s, ma_pa_stream_request_cb_t cb, void* userdata); @@ -31840,6 +31842,9 @@ static void ma_device_on_read__pulse(ma_pa_stream* pStream, size_t byteCount, vo ma_silence_pcm_frames(silence, capacity, pDevice->capture.internalFormat, pDevice->capture.internalChannels); while (remaining > 0 && ma_device_get_state(pDevice) == ma_device_state_started) { ma_uint32 n = (ma_uint32)ma_min(remaining, capacity); +#ifdef MA_AUDIO_CAPTURE_HOLE + MA_AUDIO_CAPTURE_HOLE(framesMapped - remaining); +#endif ma_device_handle_backend_data_callback(pDevice, NULL, silence, n); remaining -= n; } @@ -32144,7 +32149,7 @@ static ma_result ma_device_init__pulse(ma_device* pDevice, const ma_device_confi if (pDescriptorCapture->sampleRate != 0) { ss.rate = pDescriptorCapture->sampleRate; } - streamFlags = MA_PA_STREAM_START_CORKED | MA_PA_STREAM_ADJUST_LATENCY; + streamFlags = MA_PA_STREAM_START_CORKED | MA_PA_STREAM_ADJUST_LATENCY | MA_PA_STREAM_AUTO_TIMING_UPDATE | MA_PA_STREAM_INTERPOLATE_TIMING; if (ma_format_from_pulse(ss.format) == ma_format_unknown) { if (ma_is_little_endian()) { @@ -32303,7 +32308,7 @@ static ma_result ma_device_init__pulse(ma_device* pDevice, const ma_device_confi ss.rate = pDescriptorPlayback->sampleRate; } - streamFlags = MA_PA_STREAM_START_CORKED | MA_PA_STREAM_ADJUST_LATENCY; + streamFlags = MA_PA_STREAM_START_CORKED | MA_PA_STREAM_ADJUST_LATENCY | MA_PA_STREAM_AUTO_TIMING_UPDATE | MA_PA_STREAM_INTERPOLATE_TIMING; if (ma_format_from_pulse(ss.format) == ma_format_unknown) { if (ma_is_little_endian()) { ss.format = MA_PA_SAMPLE_FLOAT32LE; @@ -32729,6 +32734,7 @@ static ma_result ma_context_init__pulse(ma_context* pContext, const ma_context_c pContext->pulse.pa_stream_get_channel_map = (ma_proc)ma_dlsym(ma_context_get_log(pContext), pContext->pulse.pulseSO, "pa_stream_get_channel_map"); pContext->pulse.pa_stream_get_buffer_attr = (ma_proc)ma_dlsym(ma_context_get_log(pContext), pContext->pulse.pulseSO, "pa_stream_get_buffer_attr"); pContext->pulse.pa_stream_set_buffer_attr = (ma_proc)ma_dlsym(ma_context_get_log(pContext), pContext->pulse.pulseSO, "pa_stream_set_buffer_attr"); + pContext->pulse.pa_stream_get_latency = (ma_proc)ma_dlsym(ma_context_get_log(pContext), pContext->pulse.pulseSO, "pa_stream_get_latency"); pContext->pulse.pa_stream_get_device_name = (ma_proc)ma_dlsym(ma_context_get_log(pContext), pContext->pulse.pulseSO, "pa_stream_get_device_name"); pContext->pulse.pa_stream_set_write_callback = (ma_proc)ma_dlsym(ma_context_get_log(pContext), pContext->pulse.pulseSO, "pa_stream_set_write_callback"); pContext->pulse.pa_stream_set_read_callback = (ma_proc)ma_dlsym(ma_context_get_log(pContext), pContext->pulse.pulseSO, "pa_stream_set_read_callback"); @@ -32795,6 +32801,7 @@ static ma_result ma_context_init__pulse(ma_context* pContext, const ma_context_c ma_pa_stream_get_channel_map_proc _pa_stream_get_channel_map = pa_stream_get_channel_map; ma_pa_stream_get_buffer_attr_proc _pa_stream_get_buffer_attr = pa_stream_get_buffer_attr; ma_pa_stream_set_buffer_attr_proc _pa_stream_set_buffer_attr = pa_stream_set_buffer_attr; + ma_pa_stream_get_latency_proc _pa_stream_get_latency = pa_stream_get_latency; ma_pa_stream_get_device_name_proc _pa_stream_get_device_name = pa_stream_get_device_name; ma_pa_stream_set_write_callback_proc _pa_stream_set_write_callback = pa_stream_set_write_callback; ma_pa_stream_set_read_callback_proc _pa_stream_set_read_callback = pa_stream_set_read_callback; @@ -32860,6 +32867,7 @@ static ma_result ma_context_init__pulse(ma_context* pContext, const ma_context_c pContext->pulse.pa_stream_get_channel_map = (ma_proc)_pa_stream_get_channel_map; pContext->pulse.pa_stream_get_buffer_attr = (ma_proc)_pa_stream_get_buffer_attr; pContext->pulse.pa_stream_set_buffer_attr = (ma_proc)_pa_stream_set_buffer_attr; + pContext->pulse.pa_stream_get_latency = (ma_proc)_pa_stream_get_latency; pContext->pulse.pa_stream_get_device_name = (ma_proc)_pa_stream_get_device_name; pContext->pulse.pa_stream_set_write_callback = (ma_proc)_pa_stream_set_write_callback; pContext->pulse.pa_stream_set_read_callback = (ma_proc)_pa_stream_set_read_callback; @@ -44639,6 +44647,14 @@ MA_API ma_result ma_device_handle_backend_data_callback(ma_device* pDevice, void if (frameCount == 0) { return MA_INVALID_ARGS; } + if (((pDevice->type == ma_device_type_capture || pDevice->type == ma_device_type_loopback) && pInput == NULL) || + (pDevice->type == ma_device_type_playback && pOutput == NULL)) { + return MA_INVALID_ARGS; + } + +#ifdef MA_AUDIO_TIMING_BEGIN + MA_AUDIO_TIMING_BEGIN(pDevice, frameCount, pInput != NULL); +#endif if (pDevice->type == ma_device_type_duplex) { if (pInput != NULL) { @@ -44650,22 +44666,17 @@ MA_API ma_result ma_device_handle_backend_data_callback(ma_device* pDevice, void } } else { if (pDevice->type == ma_device_type_capture || pDevice->type == ma_device_type_loopback) { - if (pInput == NULL) { - return MA_INVALID_ARGS; - } - ma_device__send_frames_to_client(pDevice, frameCount, pInput); } if (pDevice->type == ma_device_type_playback) { - if (pOutput == NULL) { - return MA_INVALID_ARGS; - } - ma_device__read_frames_from_client(pDevice, frameCount, pOutput); } } +#ifdef MA_AUDIO_TIMING_END + MA_AUDIO_TIMING_END(); +#endif return MA_SUCCESS; } diff --git a/lib/speex_aec.c b/lib/speex_aec.c index 42e7d7cf..4278e927 100644 --- a/lib/speex_aec.c +++ b/lib/speex_aec.c @@ -24,6 +24,7 @@ struct speex_aec { SpeexEchoState* st; + int mic_channels, render_channels; int frame_samples; /* сэмплов в кадре (960 @48кГц) */ int depth; /* глубина линии задержки в кадрах (>=1) */ int16_t* line; /* кольцо depth*frame_samples */ @@ -42,17 +43,25 @@ static void aec_enqueue(speex_aec_t* a, const int16_t* frame) { a->count--; a->overruns++; } - memcpy(a->line + (size_t)((a->head + a->count) % a->depth) * a->frame_samples, - frame, (size_t)a->frame_samples * sizeof(int16_t)); + memcpy(a->line + (size_t)((a->head + a->count) % a->depth) * a->frame_samples * a->render_channels, + frame, (size_t)a->frame_samples * a->render_channels * sizeof(int16_t)); a->count++; } speex_aec_t* speex_aec_create(int sample_rate, int frame_samples, int filter_samples, int delay_frames) { + return speex_aec_create_mc(sample_rate, frame_samples, filter_samples, delay_frames, 1, 1); +} + +/* Независимые каналы референса: противофазное стерео не усредняется. */ +speex_aec_t* speex_aec_create_mc(int sample_rate, int frame_samples, int filter_samples, int delay_frames, + int mic_channels, int render_channels) { speex_aec_t* a; int rate; if (sample_rate <= 0 || frame_samples <= 0 || filter_samples < frame_samples || - (delay_frames > 0 && delay_frames > INT_MAX / frame_samples)) { + mic_channels < 1 || mic_channels > 2 || render_channels < 1 || render_channels > 2 || + frame_samples > INT_MAX / render_channels || frame_samples > INT_MAX / mic_channels || + (delay_frames > 0 && delay_frames > INT_MAX / frame_samples / render_channels)) { DEBUG_ERROR(DEBUG_CATEGORY_AEC, "%s: bad args rate=%d frame=%d filter=%d delay=%d", AEC_ID, sample_rate, frame_samples, filter_samples, delay_frames); return NULL; @@ -64,7 +73,7 @@ speex_aec_t* speex_aec_create(int sample_rate, int frame_samples, int filter_sam return NULL; } - a->st = speex_echo_state_init(frame_samples, filter_samples); + a->st = speex_echo_state_init_mc(frame_samples, filter_samples, mic_channels, render_channels); if (!a->st) { DEBUG_ERROR(DEBUG_CATEGORY_AEC, "%s: echo state allocation failed", AEC_ID); u_free(a); @@ -78,10 +87,11 @@ speex_aec_t* speex_aec_create(int sample_rate, int frame_samples, int filter_sam return NULL; } + a->mic_channels = mic_channels; a->render_channels = render_channels; a->frame_samples = frame_samples; a->depth = delay_frames > 0 ? delay_frames : 1; - a->line = (int16_t*)u_calloc((uint32_t)(a->depth * frame_samples), sizeof(int16_t)); - a->play_acc = (int16_t*)u_calloc((uint32_t)frame_samples, sizeof(int16_t)); + a->line = (int16_t*)u_calloc((uint32_t)(a->depth * frame_samples * render_channels), sizeof(int16_t)); + a->play_acc = (int16_t*)u_calloc((uint32_t)(frame_samples * render_channels), sizeof(int16_t)); if (!a->line || !a->play_acc) { DEBUG_ERROR(DEBUG_CATEGORY_AEC, "%s: OOM for buffers", AEC_ID); if (a->line) u_free(a->line); @@ -91,8 +101,8 @@ speex_aec_t* speex_aec_create(int sample_rate, int frame_samples, int filter_sam return NULL; } - DEBUG_INFO(DEBUG_CATEGORY_AEC, "%s: created rate=%d frame=%d filter=%d delay=%d frames", - AEC_ID, sample_rate, frame_samples, filter_samples, a->depth); + DEBUG_INFO(DEBUG_CATEGORY_AEC, "%s: created rate=%d frame=%d filter=%d delay=%d frames mic=%d render=%d", + AEC_ID, sample_rate, frame_samples, filter_samples, a->depth, mic_channels, render_channels); return a; } @@ -117,7 +127,7 @@ void speex_aec_feed_playback(speex_aec_t* a, const int16_t* pcm, int count) { int fs, take; if (!a || !pcm || count <= 0) return; - fs = a->frame_samples; + fs = a->frame_samples * a->render_channels; while (count > 0) { take = fs - a->play_len; @@ -139,23 +149,23 @@ int speex_aec_process_capture(speex_aec_t* a, const int16_t* pcm, int count, int if (!a || !pcm || !out || count <= 0) return 0; fs = a->frame_samples; - if (count != fs) { - DEBUG_ERROR(DEBUG_CATEGORY_AEC, "%s: capture requires one frame samples=%d expected=%d", AEC_ID, count, fs); + if (count != fs * a->mic_channels) { + DEBUG_ERROR(DEBUG_CATEGORY_AEC, "%s: capture requires one frame samples=%d expected=%d", AEC_ID, count, fs * a->mic_channels); return 0; } if (a->count == a->depth) { - const int16_t* ref = a->line + (size_t)a->head * fs; + const int16_t* ref = a->line + (size_t)a->head * fs * a->render_channels; speex_echo_cancellation(a->st, pcm, ref, out); a->head = (a->head + 1) % a->depth; a->count--; } else { /* линия не наполнена (старт или рендер отстаёт): passthrough без канселлера */ - memcpy(out, pcm, (size_t)fs * sizeof(int16_t)); + memcpy(out, pcm, (size_t)count * sizeof(int16_t)); a->underruns++; } - return fs; + return count; } int speex_aec_delay_fill(const speex_aec_t* a) { diff --git a/lib/speex_aec.h b/lib/speex_aec.h index 10954f99..d5aa8025 100644 --- a/lib/speex_aec.h +++ b/lib/speex_aec.h @@ -2,7 +2,7 @@ * speex_aec.h — акустическое эхоподавление (AEC), C-обёртка над SpeexDSP mdf.c. * * Подавляет эхо дальнего конца (то, что играем в динамик) в сигнале микрофона - * перед кодированием. Работает на int16 PCM, один канал, частота 48000 (можно + * перед кодированием. Работает на interleaved int16 PCM, 1/2 канала, частота 48000 (можно * 8000/16000/32000/48000), кадр 20 мс (960 сэмплов @48 кГц). * * Модель использования (duplex-контур звонка): @@ -44,6 +44,11 @@ typedef struct speex_aec speex_aec_t; */ speex_aec_t* speex_aec_create(int sample_rate, int frame_samples, int filter_samples, int delay_frames); +/* Многоканальный AEC: frame_samples на канал, count в feed/process — все interleaved отсчёты. + * mic_channels и render_channels — 1/2; возвращаемый count захвата равен frame_samples*mic_channels. */ +speex_aec_t* speex_aec_create_mc(int sample_rate, int frame_samples, int filter_samples, int delay_frames, + int mic_channels, int render_channels); + /* Освободить канселлер. NULL безопасен. */ void speex_aec_destroy(speex_aec_t* aec); @@ -57,8 +62,8 @@ void speex_aec_reset(speex_aec_t* aec); void speex_aec_feed_playback(speex_aec_t* aec, const int16_t* pcm, int count); /** - * Обработать ровно один кадр захвата (count == frame_samples). - * Возвращает frame_samples при успехе, 0 при ошибке. out вмещает полный кадр + * Обработать ровно один кадр захвата (count == frame_samples*mic_channels). + * Возвращает count при успехе, 0 при ошибке. out вмещает полный interleaved кадр * и не должен алиаситься с pcm. Накопитель аппаратных чанков принадлежит I/O-движку. */ int speex_aec_process_capture(speex_aec_t* aec, const int16_t* pcm, int count, int16_t* out); diff --git a/src/call/call_audio.c b/src/call/call_audio.c index 6a179059..5edfbd27 100644 --- a/src/call/call_audio.c +++ b/src/call/call_audio.c @@ -44,6 +44,7 @@ static struct call_jitter* g_vj = NULL; static struct call_tones* g_tones = NULL; static speex_aec_t* g_aec = NULL; static int g_aec_delay_frames = CALL_AEC_DELAY_FRAMES; +static int g_prepared_capture = 0; static int g_aec_enabled = 1; /* по умолчанию: настройка aec_enabled */ static struct audio_compressor* g_compressor = NULL; static uint64_t g_tx_generation; @@ -124,7 +125,7 @@ void call_audio_on_media(struct UTUN_INSTANCE* inst, uint64_t call_id, /* ── GUI-поток: lifecycle ── */ -int call_audio_start(uint64_t call_id) { +static int call_audio_start_impl(uint64_t call_id, int prepared) { if (call_id == 0) { DEBUG_ERROR(DEBUG_CATEGORY_CALL, "%s: start bad call_id=0", CALL_AUDIO_ID); return -1; @@ -132,8 +133,10 @@ int call_audio_start(uint64_t call_id) { pthread_mutex_lock(&g_mtx); if (g_active && g_call_id == call_id) { + int same_mode = g_prepared_capture == prepared; pthread_mutex_unlock(&g_mtx); - return 0; + if (!same_mode) DEBUG_ERROR(DEBUG_CATEGORY_CALL, "%s: capture policy cannot change within active call", CALL_AUDIO_ID); + return same_mode ? 0 : -1; } if (g_active) { pthread_mutex_unlock(&g_mtx); @@ -179,7 +182,7 @@ int call_audio_start(uint64_t call_id) { g_aec_enabled = g_inst ? chat_setting_get_int(g_inst, "aec_enabled", 1) : 1; speex_aec_t* aec = NULL; - if (g_aec_enabled) { + if (g_aec_enabled && !prepared) { aec = call_audio_aec_create(); if (!aec) { g_aec_enabled = 0; @@ -210,7 +213,7 @@ int call_audio_start(uint64_t call_id) { g_decoder = dec; g_vj = j; g_tones = tones; - g_aec = aec; + g_aec = aec; g_prepared_capture = prepared; g_compressor = ac; g_call_id = call_id; g_active = 1; @@ -231,6 +234,10 @@ int call_audio_start(uint64_t call_id) { return 0; } +/* Платформа с собственным голосовым захватом явно передаёт PCM после AEC. */ +int call_audio_start_prepared(uint64_t call_id) { return call_audio_start_impl(call_id, 1); } +int call_audio_start(uint64_t call_id) { return call_audio_start_impl(call_id, 0); } + void call_audio_begin_end(uint64_t call_id) { pthread_mutex_lock(&g_mtx); if (g_active && call_id == g_call_id && !g_ending) { @@ -292,7 +299,7 @@ void call_audio_reset_io(uint64_t call_id, int playback) { pthread_mutex_lock(&g_mtx); if (g_active && g_call_id == call_id) { if (g_aec) speex_aec_reset(g_aec); - if (g_compressor) audio_compressor_reset(g_compressor); + if (!playback && g_compressor) audio_compressor_reset(g_compressor); if (playback) { call_jitter_reset(g_vj); g_gap_samples = 0; @@ -317,7 +324,7 @@ static speex_aec_t* call_audio_aec_create(void) { void call_audio_set_aec_enabled(int enabled) { pthread_mutex_lock(&g_mtx); g_aec_enabled = enabled ? 1 : 0; - if (g_active) { + if (g_active && !g_prepared_capture) { if (g_aec_enabled && !g_aec) { g_aec = call_audio_aec_create(); if (!g_aec) DEBUG_ERROR(DEBUG_CATEGORY_CALL, "%s: AEC create failed (enabled)", CALL_AUDIO_ID); @@ -335,7 +342,7 @@ void call_audio_set_aec_delay_frames(int frames) { if (frames < 1) frames = 1; pthread_mutex_lock(&g_mtx); g_aec_delay_frames = frames; - if (g_active && g_aec_enabled && g_aec) { + if (g_active && !g_prepared_capture && g_aec_enabled && g_aec) { speex_aec_destroy(g_aec); g_aec = call_audio_aec_create(); if (!g_aec) DEBUG_ERROR(DEBUG_CATEGORY_CALL, "%s: AEC recreate failed (delay=%d)", CALL_AUDIO_ID, frames); @@ -346,8 +353,11 @@ void call_audio_set_aec_delay_frames(int frames) { /* ── аудио-поток: TX / RX ── */ -int call_audio_feed_pcm_muted(uint64_t call_id, const int16_t* pcm, int count, int muted) { - if (!pcm || count != CALL_AUDIO_FRAME_SAMPLES) return -1; +static int call_audio_feed_impl(uint64_t call_id, const int16_t* pcm, int count, int muted, const uint8_t* mute_mask, int prepared) { + if (!pcm || count != CALL_AUDIO_FRAME_SAMPLES || (prepared && !mute_mask)) { + DEBUG_ERROR(DEBUG_CATEGORY_CALL, "%s: invalid capture frame count=%d prepared=%d", CALL_AUDIO_ID, count, prepared); + return -1; + } uint8_t opus[256]; int16_t cleaned[CALL_AUDIO_FRAME_SAMPLES]; @@ -360,6 +370,11 @@ int call_audio_feed_pcm_muted(uint64_t call_id, const int16_t* pcm, int count, i pthread_mutex_unlock(&g_mtx); return -1; } + if (g_prepared_capture != prepared) { + pthread_mutex_unlock(&g_mtx); + DEBUG_ERROR(DEBUG_CATEGORY_CALL, "%s: capture policy mismatch prepared=%d", CALL_AUDIO_ID, prepared); + return -1; + } const int16_t* tx = pcm; if (g_aec) { if (speex_aec_process_capture(g_aec, pcm, count, cleaned) == count) @@ -374,7 +389,11 @@ int call_audio_feed_pcm_muted(uint64_t call_id, const int16_t* pcm, int count, i } tx = amplified; } - if (muted) { memset(amplified, 0, sizeof(amplified)); tx = amplified; } + if (mute_mask) { + if (tx != amplified) memcpy(amplified, tx, sizeof(amplified)); + for (int i = 0; i < count; ++i) if (mute_mask[i]) amplified[i] = 0; + tx = amplified; + } else if (muted) { memset(amplified, 0, sizeof(amplified)); tx = amplified; } len = opus_codec_encode(g_encoder, tx, count, opus, (int)sizeof(opus)); ua = g_inst ? g_inst->ua : NULL; inst = g_inst; @@ -405,9 +424,17 @@ int call_audio_feed_pcm_muted(uint64_t call_id, const int16_t* pcm, int count, i return 0; } +/* Маска соответствует границам mute в том же упорядоченном потоке, что и PCM. */ +int call_audio_feed_prepared_pcm(uint64_t call_id, const int16_t* pcm, const uint8_t* mute_mask) { + return call_audio_feed_impl(call_id, pcm, CALL_AUDIO_FRAME_SAMPLES, 0, mute_mask, 1); +} +int call_audio_feed_pcm_muted(uint64_t call_id, const int16_t* pcm, int count, int muted) { + return call_audio_feed_impl(call_id, pcm, count, muted, NULL, 0); +} + /* Незамьюченный захват для платформ без программного mute. */ int call_audio_feed_pcm(uint64_t call_id, const int16_t* pcm, int count) { - return call_audio_feed_pcm_muted(call_id, pcm, count, 0); + return call_audio_feed_impl(call_id, pcm, count, 0, NULL, 0); } int call_audio_pull_pcm(uint64_t call_id, int16_t* out, int max_samples) { diff --git a/src/call/call_audio.h b/src/call/call_audio.h index 5bfe5ce7..a74fdc1d 100644 --- a/src/call/call_audio.h +++ b/src/call/call_audio.h @@ -35,8 +35,11 @@ extern "C" { int call_audio_init(struct UTUN_INSTANCE* inst); void call_audio_destroy(struct UTUN_INSTANCE* inst); -/* GUI-поток: создать codec + jitter + tones для звонка. 0 = успех. */ +/* GUI-поток: создать codec + jitter + tones. Старый API принимает raw PCM и использует core AEC. + * Для desktop нужен start_prepared: AEC и stereo reference принадлежат VoiceAudioIo. 0 = успех. */ int call_audio_start(uint64_t call_id); +/* AEC принадлежит платформенному захвату; pull больше не подаёт собственный референс. */ +int call_audio_start_prepared(uint64_t call_id); /* GUI-поток: по CALL_ENDED — начать тон завершения (pull_pcm выдаст 450мс тона). */ void call_audio_begin_end(uint64_t call_id); @@ -50,6 +53,8 @@ void call_audio_stop(void); /* аудио-поток: закодировать PCM-кадр (ровно CALL_AUDIO_FRAME_SAMPLES) и отправить пиру. */ int call_audio_feed_pcm(uint64_t call_id, const int16_t* pcm, int count); /* Mute после AEC/AGC: фильтр получает настоящий микрофон, Opus — нулевой PCM. */ +/* Подготовленный PCM и маска mute (960 элементов); AEC уже выполнен до этого API. */ +int call_audio_feed_prepared_pcm(uint64_t call_id, const int16_t* pcm, const uint8_t* mute_mask); int call_audio_feed_pcm_muted(uint64_t call_id, const int16_t* pcm, int count, int muted); /* аудио-поток: выдать до max_samples финального PCM (jitter + stretch + тоны). @@ -70,12 +75,12 @@ int call_audio_get_stats(uint64_t call_id, int* buffer_ms, int* tempo_x100, * Release допустим только после остановки обоих. */ void call_audio_reset_io(uint64_t call_id, int playback); -/* Включить/выключить AEC (SpeexDSP) в рантайме. desktop — по настройке GUI, - * Android — по спикерфону (route == SPEAKER). Работает и в активном звонке. */ +/* Core AEC для raw capture: Android — по спикерфону (route == SPEAKER). + * Prepared capture не создаёт core AEC; desktop передаёт настройку в VoiceAudioIo. */ void call_audio_set_aec_enabled(int enabled); /* Задать задержку рендер→захват AEC в кадрах (20 мс). Android измеряет через - * AudioTrack/AudioRecord getTimestamp, desktop использует дефолт. Если AEC уже + * AudioTrack/AudioRecord getTimestamp. Prepared desktop выравнивает референс по времени. Если core AEC уже * активен — пересоздаётся с новой задержкой. */ void call_audio_set_aec_delay_frames(int frames); diff --git a/src/radio/radio_audio.c b/src/radio/radio_audio.c index a45842d6..bb1af580 100644 --- a/src/radio/radio_audio.c +++ b/src/radio/radio_audio.c @@ -73,6 +73,10 @@ struct radio_source { }; static pthread_mutex_t g_mtx = PTHREAD_MUTEX_INITIALIZER; +/* TX → общий mutex. На время VAD общий mutex отпускается; TX сохраняет lifetime и порядок PCM/PTT. */ +static pthread_mutex_t g_tx_mtx = PTHREAD_MUTEX_INITIALIZER; +static void radio_tx_lock(void) { pthread_mutex_lock(&g_tx_mtx); pthread_mutex_lock(&g_mtx); } +static void radio_tx_unlock(void) { pthread_mutex_unlock(&g_mtx); pthread_mutex_unlock(&g_tx_mtx); } static struct UTUN_INSTANCE* g_inst = NULL; static int g_active = 0; static uint64_t g_group_id = 0; @@ -99,14 +103,26 @@ static uint64_t g_tx_stats_tb = 0; static uint64_t g_vad_stats_tb = 0; static uint32_t g_tx_frames = 0; /* кадров за интервал */ static uint32_t g_tx_bytes = 0; /* суммарный размер opus за интервал */ -static uint32_t g_tx_tail = 0; /* дропнутых хвостовых сэмплов за интервал */ - +static uint32_t g_tx_tail = 0; /* нулевых отсчётов дополнения хвоста за интервал */ + +/* Один TX-автомат и отдельный накопитель сетевого кадра; AEC выполняется до этого API. */ +static int g_transmitting, g_tx_pcm_frames, g_vad_decimation; +static int16_t g_tx_pcm[RADIO_AUDIO_MIX_SAMPLES]; +static float g_vad_sum; +static uint64_t g_tx_burst; +/* Только uasync: идентичность открытой сетевой серии, включая поколение capture. */ +static uint64_t g_net_generation, g_net_burst, g_net_group; static uint64_t g_tx_generation; +static uint64_t g_listen_generation; +static uint64_t g_instance_generation; +static unsigned int g_control_pending; +static int g_control_warned; +static struct radio_audio_talk* g_reserved_fin; static unsigned int g_tx_pending, g_tx_dropped, g_tx_reported; struct radio_audio_tx { struct posted_task task; struct UTUN_INSTANCE* inst; - uint64_t group_id, generation, created_tb; + uint64_t group_id, generation, listen_generation, instance_generation, burst, created_tb; int len; uint8_t opus[RADIO_MAX_OPUS]; }; @@ -115,8 +131,10 @@ static void radio_audio_send(void* arg) { struct radio_audio_tx* tx = arg; pthread_mutex_lock(&g_mtx); int current = tx->generation == g_tx_generation; - if (g_tx_pending) --g_tx_pending; - int send = current && g_active && g_inst == tx->inst && g_group_id == tx->group_id; + int owner = tx->instance_generation == g_instance_generation && g_inst == tx->inst; + if (owner && g_tx_pending) --g_tx_pending; + int send = owner && tx->listen_generation == g_listen_generation && g_active && g_group_id == tx->group_id && + g_net_generation == tx->generation && g_net_burst == tx->burst && g_net_group == tx->group_id; if (get_time_tb() - tx->created_tb > 2000) { send = 0; if (current) ++g_tx_dropped; } unsigned int dropped = g_tx_dropped, pending = g_tx_pending; int changed = dropped != g_tx_reported; @@ -132,7 +150,7 @@ static void radio_audio_send(void* arg) { DEBUG_WARN(DEBUG_CATEGORY_RADIO, "%s: TX dropped=%u pending=%u limit=8 max_age=200ms", RADIO_AUDIO_ID, dropped, pending); } - if (frames) DEBUG_DEBUG(DEBUG_CATEGORY_RADIO, "%s: TX frames=%u bytes=%u tail_samples=%u", RADIO_AUDIO_ID, frames, bytes, tail); + if (frames) DEBUG_DEBUG(DEBUG_CATEGORY_RADIO, "%s: TX frames=%u bytes=%u tail_padding_samples=%u", RADIO_AUDIO_ID, frames, bytes, tail); if (send) radio_talk_send(tx->inst, tx->group_id, tx->opus, tx->len); } @@ -142,7 +160,9 @@ static int radio_audio_post_frame(struct UTUN_INSTANCE* inst, uint64_t group_id, struct radio_audio_tx* task = u_calloc(1, sizeof(*task)); if (!task) { ++g_tx_dropped; return -1; } task->inst = inst; task->group_id = group_id; - task->generation = g_tx_generation; task->created_tb = get_time_tb(); + task->generation = g_tx_generation; task->burst = g_tx_burst; task->created_tb = get_time_tb(); + task->listen_generation = g_listen_generation; + task->instance_generation = g_instance_generation; task->len = len; memcpy(task->opus, opus, (size_t)len); task->task.callback = radio_audio_send; task->task.arg = task; @@ -307,7 +327,9 @@ static struct radio_source* radio_src_acquire(uint64_t src, uint16_t stream) { return NULL; } -static void radio_audio_post_talk(struct UASYNC* ua, uint64_t group_id, int begin); +static int radio_audio_post_talk(struct UASYNC* ua, uint64_t group_id, int begin); +static void radio_audio_update_transmission(void); +static void radio_audio_flush_tx(void); /* ── lifecycle ── */ @@ -315,7 +337,7 @@ int radio_audio_init(struct UTUN_INSTANCE* inst) { if (!inst) return -1; pthread_mutex_lock(&g_mtx); if (g_inst == inst) { pthread_mutex_unlock(&g_mtx); return 0; } - g_inst = inst; g_tx_pending = 0; + g_inst = inst; g_tx_pending = g_control_pending = 0; ++g_listen_generation; ++g_instance_generation; pthread_mutex_unlock(&g_mtx); radio_set_frame_cb(inst, radio_audio_on_frame, NULL); DEBUG_INFO(DEBUG_CATEGORY_RADIO, "%s: initialized", RADIO_AUDIO_ID); @@ -327,6 +349,7 @@ void radio_audio_destroy(struct UTUN_INSTANCE* inst) { radio_set_frame_cb(inst, NULL, NULL); pthread_mutex_lock(&g_mtx); g_inst = NULL; + ++g_instance_generation; pthread_mutex_unlock(&g_mtx); DEBUG_INFO(DEBUG_CATEGORY_RADIO, "%s: destroyed", RADIO_AUDIO_ID); } @@ -339,6 +362,7 @@ int radio_audio_start(uint64_t group_id, int capture_channels) { radio_audio_stop(); pthread_mutex_lock(&g_mtx); g_group_id = group_id; + ++g_listen_generation; g_active = 1; g_last_rx_tb = 0; memset(g_sources, 0, sizeof(g_sources)); @@ -355,14 +379,14 @@ int radio_audio_capture_start(uint64_t group_id, int capture_channels, int vad_e DEBUG_ERROR(DEBUG_CATEGORY_RADIO, "%s: capture invalid channels=%d", RADIO_AUDIO_ID, capture_channels); return -1; } - pthread_mutex_lock(&g_mtx); + radio_tx_lock(); if (!g_active || g_group_id != group_id || g_encoder) { - pthread_mutex_unlock(&g_mtx); + radio_tx_unlock(); DEBUG_ERROR(DEBUG_CATEGORY_RADIO, "%s: capture start invalid state grp=%016llx", RADIO_AUDIO_ID, (unsigned long long)group_id); return -1; } opus_codec_encoder_t* enc = opus_codec_encoder_create(RADIO_AUDIO_SAMPLE_RATE, capture_channels); - if (!enc) { pthread_mutex_unlock(&g_mtx); DEBUG_ERROR(DEBUG_CATEGORY_RADIO, "%s: encoder create failed", RADIO_AUDIO_ID); return -1; } + if (!enc) { radio_tx_unlock(); DEBUG_ERROR(DEBUG_CATEGORY_RADIO, "%s: encoder create failed", RADIO_AUDIO_ID); return -1; } int opus_mode = chat_setting_get_int(g_inst, "radio_opus_mode", 0); int opus_bitrate_kbps = chat_setting_get_int(g_inst, "radio_opus_bitrate", RADIO_AUDIO_BITRATE / 1000); opus_codec_encoder_application_set(enc, opus_mode ? OPUS_CODEC_APP_AUDIO : OPUS_CODEC_APP_VOIP); @@ -387,21 +411,24 @@ int radio_audio_capture_start(uint64_t group_id, int capture_channels, int vad_e if (!ac) DEBUG_WARN(DEBUG_CATEGORY_RADIO, "%s: compressor unavailable", RADIO_AUDIO_ID); /* VAD анализирует текущие кадры без истории захвата. */ - g_vad_mode = vad_enabled ? 1 : 0; - if (g_vad_mode) { + g_vad_mode = 0; + if (vad_enabled) { g_vad_threshold = (float)chat_setting_get_int(g_inst, "radio_vad_threshold", 50) / 100.0f; g_vad_hangover_ms = chat_setting_get_int(g_inst, "radio_vad_hangover_ms", 200); - g_vad = silero_vad_create_default(); + pthread_mutex_unlock(&g_mtx); + silero_vad_t* vad = silero_vad_create_default(); + pthread_mutex_lock(&g_mtx); + g_vad = vad; g_vad_mode = vad != NULL; if (!g_vad) { DEBUG_WARN(DEBUG_CATEGORY_RADIO, "%s: silero_vad_create_default failed — VAD auto-PTT disabled", RADIO_AUDIO_ID); - g_vad_mode = 0; } } if (g_vad_mode) { g_vad_win_len = 0; radio_vad_fsm_reset(&g_vad_fsm); } - g_vad_manual_ptt = 0; + g_vad_manual_ptt = g_transmitting = g_tx_pcm_frames = g_vad_decimation = 0; + g_vad_sum = 0; g_tx_burst = 0; ++g_tx_generation; g_tx_dropped = g_tx_reported = 0; g_encoder = enc; @@ -410,16 +437,19 @@ int radio_audio_capture_start(uint64_t group_id, int capture_channels, int vad_e DEBUG_INFO(DEBUG_CATEGORY_RADIO, "%s: capture started grp=%016llx ch=%d opus=%s bitrate=%dkbps compressor=%s vad=%d threshold=%.2f hangover=%dms", RADIO_AUDIO_ID, (unsigned long long)group_id, capture_channels, opus_mode ? "AUDIO" : "VOIP", opus_bitrate_kbps, ac ? (audio_compressor_is_enabled(ac) ? "on" : "off") : "unavailable", g_vad_mode, g_vad_threshold, g_vad_hangover_ms); - pthread_mutex_unlock(&g_mtx); + radio_tx_unlock(); return 0; } void radio_audio_capture_stop(void) { - pthread_mutex_lock(&g_mtx); + radio_tx_lock(); if (g_encoder) { - ++g_tx_generation; - if (g_vad_manual_ptt || g_vad_fsm.talking) + if (g_transmitting) { + radio_audio_flush_tx(); radio_audio_post_talk(g_inst ? g_inst->ua : NULL, g_group_id, 0); + } + ++g_tx_generation; + g_transmitting = g_tx_pcm_frames = 0; opus_codec_encoder_destroy(g_encoder); g_encoder = NULL; if (g_compressor) { audio_compressor_destroy(g_compressor); g_compressor = NULL; } if (g_vad) { silero_vad_destroy(g_vad); g_vad = NULL; } @@ -429,7 +459,7 @@ void radio_audio_capture_stop(void) { radio_vad_fsm_reset(&g_vad_fsm); DEBUG_INFO(DEBUG_CATEGORY_RADIO, "%s: capture stopped", RADIO_AUDIO_ID); } - pthread_mutex_unlock(&g_mtx); + radio_tx_unlock(); } void radio_audio_stop(void) { @@ -478,7 +508,7 @@ int radio_audio_vad_mode(void) { int radio_audio_transmitting(void) { pthread_mutex_lock(&g_mtx); - int talking = g_active && (g_vad_manual_ptt || g_vad_fsm.talking); + int talking = g_active && g_transmitting; pthread_mutex_unlock(&g_mtx); return talking; } @@ -488,55 +518,82 @@ int radio_audio_transmitting(void) { struct radio_audio_talk { struct posted_task task; struct UTUN_INSTANCE* inst; - uint64_t group_id; + uint64_t group_id, generation, listen_generation, instance_generation, burst, created_tb; int begin; }; static void radio_audio_talk(void* arg) { struct radio_audio_talk* talk = arg; - if (talk->begin) radio_talk_begin(talk->inst, talk->group_id); - else radio_talk_end(talk->inst, talk->group_id); + pthread_mutex_lock(&g_mtx); + int owner = g_inst == talk->inst && g_instance_generation == talk->instance_generation; + if (owner && g_control_pending) --g_control_pending; + int current = owner && talk->listen_generation == g_listen_generation && g_active && g_group_id == talk->group_id; + pthread_mutex_unlock(&g_mtx); + if (talk->begin) { + if (current && get_time_tb() - talk->created_tb <= 2000 && radio_talk_begin(talk->inst, talk->group_id) == 0) { + g_net_generation = talk->generation; g_net_burst = talk->burst; g_net_group = talk->group_id; + } + } else if (g_net_generation == talk->generation && g_net_burst == talk->burst && g_net_group == talk->group_id) { + if (owner) radio_talk_end(talk->inst, talk->group_id); + g_net_generation = g_net_burst = g_net_group = 0; + } } -static void radio_audio_post_talk(struct UASYNC* ua, uint64_t group_id, int begin) { - if (!ua) { DEBUG_ERROR(DEBUG_CATEGORY_RADIO, "%s: talk post without uasync", RADIO_AUDIO_ID); return; } - struct radio_audio_talk* talk = u_calloc(1, sizeof(*talk)); - if (!talk) { DEBUG_ERROR(DEBUG_CATEGORY_RADIO, "%s: talk post allocation failed", RADIO_AUDIO_ID); return; } +/* Не более 16 control-задач, включая заранее выделенный FIN активной серии. */ +static int radio_audio_post_talk(struct UASYNC* ua, uint64_t group_id, int begin) { + if (!ua) { DEBUG_ERROR(DEBUG_CATEGORY_RADIO, "%s: talk post without uasync", RADIO_AUDIO_ID); return -1; } + struct radio_audio_talk* talk; + if (begin) { + if (g_control_pending > 14) { + if (!g_control_warned) DEBUG_WARN(DEBUG_CATEGORY_RADIO, "%s: TX BEGIN deferred control_pending=%u limit=16", + RADIO_AUDIO_ID, g_control_pending); + g_control_warned = 1; return -1; + } + talk = u_calloc(1, sizeof(*talk)); + struct radio_audio_talk* fin = u_calloc(1, sizeof(*fin)); + if (!talk || !fin) { + u_free(talk); u_free(fin); + DEBUG_ERROR(DEBUG_CATEGORY_RADIO, "%s: BEGIN/FIN allocation failed", RADIO_AUDIO_ID); return -1; + } + g_reserved_fin = fin; g_control_pending += 2; g_control_warned = 0; + } else { + talk = g_reserved_fin; g_reserved_fin = NULL; + if (!talk) { DEBUG_ERROR(DEBUG_CATEGORY_RADIO, "%s: FIN without reserved task", RADIO_AUDIO_ID); return -1; } + } talk->inst = g_inst; talk->group_id = group_id; talk->begin = begin; + talk->generation = g_tx_generation; talk->burst = g_tx_burst; + talk->listen_generation = g_listen_generation; talk->created_tb = get_time_tb(); + talk->instance_generation = g_instance_generation; talk->task.callback = radio_audio_talk; talk->task.arg = talk; uasync_post_reserved(ua, &talk->task); + return 0; } void radio_audio_talk_begin(uint64_t group_id) { if (!g_inst || !g_inst->ua) return; - pthread_mutex_lock(&g_mtx); + radio_tx_lock(); if (!g_active || group_id != g_group_id || !g_encoder) { - pthread_mutex_unlock(&g_mtx); + radio_tx_unlock(); DEBUG_WARN(DEBUG_CATEGORY_RADIO, "%s: talk_begin without capture grp=%016llx", RADIO_AUDIO_ID, (unsigned long long)group_id); return; } g_vad_manual_ptt = 1; - if (g_vad_fsm.talking) { - /* VAD вела передачу — закрываем её FIN, затем открываем ручной burst */ - radio_audio_post_talk(g_inst->ua, group_id, 0); - radio_vad_fsm_reset(&g_vad_fsm); - } - radio_audio_post_talk(g_inst->ua, group_id, 1); - pthread_mutex_unlock(&g_mtx); + radio_audio_update_transmission(); + radio_tx_unlock(); } void radio_audio_talk_end(uint64_t group_id) { if (!g_inst || !g_inst->ua) return; - pthread_mutex_lock(&g_mtx); + radio_tx_lock(); if (!g_active || group_id != g_group_id || !g_vad_manual_ptt) { - pthread_mutex_unlock(&g_mtx); + radio_tx_unlock(); DEBUG_WARN(DEBUG_CATEGORY_RADIO, "%s: talk_end without manual transmission grp=%016llx", RADIO_AUDIO_ID, (unsigned long long)group_id); return; } g_vad_manual_ptt = 0; - radio_audio_post_talk(g_inst->ua, group_id, 0); - pthread_mutex_unlock(&g_mtx); + radio_audio_update_transmission(); + radio_tx_unlock(); } /* ── VAD авто-PTT: helpers (под g_mtx) ── */ @@ -561,87 +618,86 @@ static void radio_audio_encode_post(uint64_t group_id, const int16_t* frame, if (radio_audio_post_frame(inst, group_id, opus, l) == 0) { ++g_tx_frames; g_tx_bytes += (uint32_t)l; } } -/* VAD-ветка feed_pcm: ресемпл → Silero → фильтр → старт/стоп/передача (держит g_mtx). */ -static int radio_vad_feed(uint64_t group_id, const int16_t* pcm, int count) { - float out16k[RADIO_VAD_FRAME_SAMPLES_16K]; - struct UASYNC* ua; - struct UTUN_INSTANCE* inst; - - pthread_mutex_lock(&g_mtx); - if (!g_active || group_id != g_group_id || !g_encoder || !g_vad) { - pthread_mutex_unlock(&g_mtx); - return -1; - } - ua = g_inst ? g_inst->ua : NULL; - inst = g_inst; - - if (g_vad_manual_ptt) { - /* Ручной PTT: передаём текущий кадр, VAD не вмешивается. */ - radio_audio_encode_post(group_id, pcm, ua, inst); - pthread_mutex_unlock(&g_mtx); - return 0; - } +/* Хвост относится только к своей серии. Тишина добавляется исключительно перед Opus. */ +static void radio_audio_flush_tx(void) { + if (!g_tx_pcm_frames) return; + int count = g_tx_pcm_frames * g_capture_channels; + g_tx_tail += RADIO_AUDIO_FRAME_SAMPLES * g_capture_channels - count; + memset(g_tx_pcm + count, 0, (size_t)(RADIO_AUDIO_FRAME_SAMPLES * g_capture_channels - count) * sizeof(int16_t)); + radio_audio_encode_post(g_group_id, g_tx_pcm, g_inst ? g_inst->ua : NULL, g_inst); + g_tx_pcm_frames = 0; +} - radio_vad_resample(pcm, g_capture_channels, out16k); - int off = 0; - while (off < RADIO_VAD_FRAME_SAMPLES_16K) { - int space = RADIO_VAD_WINDOW_16K - g_vad_win_len; - int take = (RADIO_VAD_FRAME_SAMPLES_16K - off < space) ? (RADIO_VAD_FRAME_SAMPLES_16K - off) : space; - memcpy(g_vad_win + g_vad_win_len, out16k + off, (size_t)take * sizeof(float)); - g_vad_win_len += take; - off += take; - if (g_vad_win_len == RADIO_VAD_WINDOW_16K) { - float prob = 0.0f; - if (silero_vad_process(g_vad, g_vad_win, &prob) != 0) { - DEBUG_ERROR(DEBUG_CATEGORY_RADIO, "%s: VAD processing failed", RADIO_AUDIO_ID); - prob = 0.0f; - } - g_vad_win_len = 0; - uint64_t now = get_time_tb(); - int busy = (g_last_rx_tb != 0 && (now - g_last_rx_tb) < RADIO_VAD_BUSY_TB) ? 1 : 0; - if (!g_vad_stats_tb || now - g_vad_stats_tb >= RADIO_AUDIO_STATS_TB) { - DEBUG_DEBUG(DEBUG_CATEGORY_RADIO, "%s: vad prob=%.3f threshold=%.2f busy=%d talking=%d confirm=%d", - RADIO_AUDIO_ID, prob, g_vad_threshold, busy, g_vad_fsm.talking, g_vad_fsm.confirm_count); - g_vad_stats_tb = now; - } - int action = radio_vad_fsm_update(&g_vad_fsm, prob, g_vad_threshold, - RADIO_VAD_CONFIRM_WINDOWS, - (uint64_t)g_vad_hangover_ms * 10, busy, now); - if (action == 1) { - DEBUG_INFO(DEBUG_CATEGORY_RADIO, "%s: vad START prob=%.2f grp=%016llx", - RADIO_AUDIO_ID, prob, (unsigned long long)group_id); - radio_audio_post_talk(ua, group_id, 1); - } else if (action == -1) { - DEBUG_INFO(DEBUG_CATEGORY_RADIO, "%s: vad STOP grp=%016llx", - RADIO_AUDIO_ID, (unsigned long long)group_id); - radio_audio_post_talk(ua, group_id, 0); - } - } - } +/* Ручной PTT имеет приоритет; смена manual/VAD при открытой серии не создаёт второй BEGIN. */ +static void radio_audio_update_transmission(void) { + int desired = g_vad_manual_ptt || (g_vad_mode && g_vad_fsm.talking); + if (desired == g_transmitting) return; + if (!desired) radio_audio_flush_tx(); + else { ++g_tx_burst; g_tx_pcm_frames = 0; } + if (radio_audio_post_talk(g_inst ? g_inst->ua : NULL, g_group_id, desired) != 0) return; + g_transmitting = desired; + DEBUG_INFO(DEBUG_CATEGORY_RADIO, "%s: TX %s generation=%llu burst=%llu manual=%d vad=%d", RADIO_AUDIO_ID, + desired ? "BEGIN" : "FIN", (unsigned long long)g_tx_generation, (unsigned long long)g_tx_burst, + g_vad_manual_ptt, g_vad_fsm.talking); +} - if (g_vad_fsm.talking) { - radio_audio_encode_post(group_id, pcm, ua, inst); +void radio_audio_capture_discontinuity(uint64_t group_id) { + radio_tx_lock(); + if (g_active && g_group_id == group_id && g_encoder) { + DEBUG_WARN(DEBUG_CATEGORY_RADIO, "%s: capture discontinuity tail=%d transmitting=%d", RADIO_AUDIO_ID, + g_tx_pcm_frames, g_transmitting); + radio_audio_flush_tx(); + g_vad_win_len = g_vad_decimation = 0; g_vad_sum = 0; + if (g_vad) silero_vad_reset(g_vad); + radio_vad_fsm_reset(&g_vad_fsm); + radio_audio_update_transmission(); } + radio_tx_unlock(); +} +/* VAD читает очищенный сигнал до AGC, накопление 48k→16k непрерывно между PCM-чанками. */ +static void radio_audio_vad_sample(const int16_t* frame) { + float mono = g_capture_channels == 2 ? ((float)frame[0] + frame[1]) * 0.5f : frame[0]; + g_vad_sum += mono; + if (++g_vad_decimation != 3) return; + g_vad_win[g_vad_win_len++] = g_vad_sum / (3.0f * 32768.0f); + g_vad_decimation = 0; g_vad_sum = 0; + if (g_vad_win_len != RADIO_VAD_WINDOW_16K) return; + float probability = 0; pthread_mutex_unlock(&g_mtx); - return 0; + int result = silero_vad_process(g_vad, g_vad_win, &probability); + pthread_mutex_lock(&g_mtx); + if (result != 0) + DEBUG_ERROR(DEBUG_CATEGORY_RADIO, "%s: VAD process failed", RADIO_AUDIO_ID); + g_vad_win_len = 0; + uint64_t now = get_time_tb(); + int busy = g_last_rx_tb && now - g_last_rx_tb < RADIO_VAD_BUSY_TB; + radio_vad_fsm_update(&g_vad_fsm, probability, g_vad_threshold, RADIO_VAD_CONFIRM_WINDOWS, + (uint64_t)g_vad_hangover_ms * 10, busy, now); + radio_audio_update_transmission(); + if (!g_vad_stats_tb || now - g_vad_stats_tb >= RADIO_AUDIO_STATS_TB) { + DEBUG_DEBUG(DEBUG_CATEGORY_RADIO, "%s: VAD probability=%.3f busy=%d manual=%d transmitting=%d", + RADIO_AUDIO_ID, probability, busy, g_vad_manual_ptt, g_transmitting); + g_vad_stats_tb = now; + } } -/* ── TX (аудио-поток) ── */ - +/* Очищенные interleaved отсчёты произвольной длины; границы PTT ставит единственный TX-writer. */ int radio_audio_feed_pcm(uint64_t group_id, const int16_t* pcm, int count) { - pthread_mutex_lock(&g_mtx); - int frame_total = RADIO_AUDIO_FRAME_SAMPLES * g_capture_channels; - if (!pcm || count != frame_total) { pthread_mutex_unlock(&g_mtx); return -1; } - if (g_vad_mode) { pthread_mutex_unlock(&g_mtx); return radio_vad_feed(group_id, pcm, count); } - - if (!g_active || group_id != g_group_id || !g_encoder) { - pthread_mutex_unlock(&g_mtx); - DEBUG_WARN(DEBUG_CATEGORY_RADIO, "%s: frame without capture", RADIO_AUDIO_ID); + radio_tx_lock(); + if (!g_active || group_id != g_group_id || !g_encoder || !pcm || count <= 0 || count % g_capture_channels) { + radio_tx_unlock(); + DEBUG_WARN(DEBUG_CATEGORY_RADIO, "%s: invalid capture frame count=%d", RADIO_AUDIO_ID, count); return -1; } - if (g_vad_manual_ptt) radio_audio_encode_post(group_id, pcm, g_inst ? g_inst->ua : NULL, g_inst); - pthread_mutex_unlock(&g_mtx); + radio_audio_update_transmission(); // Повторить BEGIN после освобождения bounded control-очереди. + for (int offset = 0; offset < count; offset += g_capture_channels) { + if (g_vad_mode) radio_audio_vad_sample(pcm + offset); + if (!g_transmitting) continue; + memcpy(g_tx_pcm + g_tx_pcm_frames * g_capture_channels, pcm + offset, g_capture_channels * sizeof(int16_t)); + if (++g_tx_pcm_frames == RADIO_AUDIO_FRAME_SAMPLES) radio_audio_flush_tx(); + } + radio_tx_unlock(); return 0; } diff --git a/src/radio/radio_audio.h b/src/radio/radio_audio.h index 12164692..992b26d1 100644 --- a/src/radio/radio_audio.h +++ b/src/radio/radio_audio.h @@ -1,15 +1,15 @@ // radio_audio.h — сервисный аудио-движок рации (единый для desktop/Android/headless). // // Владеет Opus encoder (TX), per-source декодерами и микшером (RX). Приём: -// radio.c → radio_set_frame_cb → radio_audio_on_frame (uasync) декодирует Opus-кадр -// и кладёт PCM в ring-буфер источника; аудио-поток radio_audio_pull_pcm микширует +// radio.c → radio_set_frame_cb → radio_audio_on_frame (uasync) сохраняет Opus-кадр +// в jitter источника; декодирование и time-stretch выполняются на pull; аудио-поток radio_audio_pull_pcm микширует // PCM всех источников. Передача: аудио-поток radio_audio_feed_pcm кодирует PCM и // постит radio_talk_send в uasync. Микшер избыточных источников — насыщающее // суммирование (soft-clip). Полудуплекс: радио включено — слушаем, PTT — говорим. // // Потоки (как call_audio): -// - radio_audio_start / talk_begin / talk_end / stop — GUI-поток; -// - radio_audio_feed_pcm / pull_pcm — аудио-поток; +// - radio_audio_start / capture_start / capture_stop / stop — GUI после остановки workers; +// - radio_audio_feed_pcm / talk_begin / talk_end — один TX-writer, pull_pcm — отдельный RX-writer; // - radio_audio_on_frame — uasync-поток (из radio.c). // Общее состояние — под мьютексом. Синглтон на один активный group_id. @@ -42,17 +42,20 @@ int radio_audio_start(uint64_t group_id, int capture_channels); * capture_stop отправляет FIN активной передачи и освобождает TX, сохраняя RX. */ int radio_audio_capture_start(uint64_t group_id, int capture_channels, int vad_enabled); void radio_audio_capture_stop(void); +/* Единственный TX-writer: завершить старый PCM-хвост, сбросить VAD после потери захвата. */ +void radio_audio_capture_discontinuity(uint64_t group_id); /* GUI-поток: прекратить прослушивание (освободить всё). */ void radio_audio_stop(void); /* Playback остановлен: удалить устаревшие источники, сохранив TX и подписку. */ void radio_audio_reset_playback(uint64_t group_id); -/* GUI-поток: PTT зажата/отпущена — постит radio_talk_begin/end в uasync. */ +/* Единственный TX-writer: PTT упорядочен с PCM; сетевые BEGIN/FIN исполняются в uasync. */ void radio_audio_talk_begin(uint64_t group_id); void radio_audio_talk_end(uint64_t group_id); -/* аудио-поток: закодировать PCM-кадр (ровно RADIO_AUDIO_FRAME_SAMPLES) и отправить. */ +/* Единственный TX-writer: очищенные interleaved отсчёты произвольной длины (кратно числу каналов). + * VAD до AGC; один автомат manual/VAD; хвост каждой серии дополняется тишиной до FIN. */ int radio_audio_feed_pcm(uint64_t group_id, const int16_t* pcm, int count); /* аудио-поток: выдать до max_samples микшированного PCM (источники микшируются, diff --git a/tests/test_speex_aec.c b/tests/test_speex_aec.c index 9a8ca661..3f8b0637 100644 --- a/tests/test_speex_aec.c +++ b/tests/test_speex_aec.c @@ -56,7 +56,9 @@ static int test_lifecycle_and_null(void) { CHECK(speex_aec_create(0, FS, FILTER, DELAY_FRAMES) == NULL, "create(rate=0) should fail"); CHECK(speex_aec_create(RATE, 0, FILTER, DELAY_FRAMES) == NULL, "create(frame=0) should fail"); CHECK(speex_aec_create(RATE, FS, FS - 1, DELAY_FRAMES) == NULL, "create(filter= 160 && frame < 180; + for (int i = 0; i < FS; ++i) { + render[2*i] = aec_noise(); render[2*i+1] = aec_noise(); + int16_t speech = (int16_t)(3000 * sin(2 * M_PI * 440 * (frame * FS + i) / RATE)); + for (int ch = 0; ch < 2; ++ch) { + int16_t echo = (int16_t)(render[2*i] * (ch ? -0.3 : 0.5) + render[2*i+1] * (ch ? 0.45 : 0.2)); + input[2*i+ch] = echo + (double_talk ? speech : 0); + if (frame >= 180) echo_power += (double)echo * echo; + } + } + speex_aec_feed_playback(a, render, FS * 2); + CHECK(speex_aec_process_capture(a, input, FS * 2, output) == FS * 2, "MC capture count"); + for (int i = 0; i < FS * 2; ++i) { + if (frame >= 180) residual_power += (double)output[i] * output[i]; + if (double_talk) { + double speech = 3000 * sin(2 * M_PI * 440 * (frame * FS + i / 2) / RATE); + speech_power += speech * speech; output_power += (double)output[i] * output[i]; cross += speech * output[i]; + } + } + } + printf(" MC residual=%.6f double-talk correlation=%.3f\n", residual_power / echo_power, cross / sqrt(speech_power * output_power)); + CHECK(residual_power < echo_power * 0.1, "MC echo not suppressed"); + CHECK(cross / sqrt(speech_power * output_power) > 0.7, "MC near-end speech distorted"); + speex_aec_destroy(a); +} + int main(void) { debug_config_init(); debug_set_level(DEBUG_LEVEL_INFO); @@ -208,6 +244,7 @@ int main(void) { test_lifecycle_and_null(); test_delay_line(); test_echo_suppression(); + test_stereo_microphones(); if (g_failures == 0) { printf("TEST PASSED\n"); diff --git a/tools/chatgui/CMakeLists.txt b/tools/chatgui/CMakeLists.txt index 1a9d7ad0..afef7e09 100644 --- a/tools/chatgui/CMakeLists.txt +++ b/tools/chatgui/CMakeLists.txt @@ -130,6 +130,7 @@ add_executable(vibechat src/nodespage.cpp src/soundsettingspage.cpp src/sound_manager.cpp + src/voice_audio_io.cpp src/audio_device.cpp src/audiorecorder.cpp src/voicemessageencoder.cpp @@ -241,11 +242,35 @@ target_include_directories(test_radio_audio PRIVATE ${CMAKE_SOURCE_DIR}/../../sr target_link_libraries(test_radio_audio PRIVATE utun_voice utun pthread) add_test(NAME test_radio_audio COMMAND test_radio_audio) +add_executable(test_audio_event_queue tests/test_audio_event_queue.cpp) +target_link_libraries(test_audio_event_queue PRIVATE pthread) +add_test(NAME test_audio_event_queue COMMAND test_audio_event_queue) +add_executable(test_voice_audio_io tests/test_voice_audio_io.cpp src/voice_audio_io.cpp) +target_include_directories(test_voice_audio_io PRIVATE ${UTUN_INCLUDE_DIRS}) +target_link_libraries(test_voice_audio_io PRIVATE utun pthread) +add_test(NAME test_voice_audio_io COMMAND test_voice_audio_io) +set_tests_properties(test_audio_event_queue test_voice_audio_io PROPERTIES TIMEOUT 15) +if(UNIX AND NOT APPLE) + add_executable(test_audio_timing tests/test_audio_timing.c transport/audio_diagnostics.cpp) + target_link_libraries(test_audio_timing PRIVATE utun pthread dl m) + target_compile_options(test_audio_timing PRIVATE -UNDEBUG) + add_test(NAME test_audio_timing COMMAND test_audio_timing) + add_executable(test_voice_tx tests/test_voice_tx.cpp) + target_include_directories(test_voice_tx PRIVATE ${UTUN_INCLUDE_DIRS}) + target_link_libraries(test_voice_tx PRIVATE utun_voice utun pthread) + target_link_options(test_voice_tx PRIVATE "-Wl,--wrap=radio_talk_begin" "-Wl,--wrap=radio_talk_end" + "-Wl,--wrap=radio_talk_send" "-Wl,--wrap=call_send_media" "-Wl,--wrap=chat_setting_get_int" + "-Wl,--wrap=silero_vad_create_default" "-Wl,--wrap=silero_vad_destroy" "-Wl,--wrap=silero_vad_reset" + "-Wl,--wrap=silero_vad_process") + add_test(NAME test_voice_tx COMMAND test_voice_tx) +endif() + # smoke-тест видео-декодера (FFmpeg). Ручной запуск: test_video_engine add_executable(test_video_engine tests/test_video_engine.cpp src/videoplayer_engine.cpp src/sound_manager.cpp + src/voice_audio_io.cpp src/audio_device.cpp transport/miniaudio_impl.c transport/audio_diagnostics.cpp @@ -257,22 +282,22 @@ if(FFMPEG_FOUND) endif() set_source_files_properties(transport/miniaudio_impl.c PROPERTIES LANGUAGE C) if(WIN32) - target_link_libraries(test_video_engine PRIVATE ${QT_LIBS} ${QT_CORE} utun pthread ${FFMPEG_TARGET} ${FFMPEG_LIBRARIES}) + target_link_libraries(test_video_engine PRIVATE ${QT_LIBS} ${QT_CORE} utun_voice utun pthread ${FFMPEG_TARGET} ${FFMPEG_LIBRARIES}) else() - target_link_libraries(test_video_engine PRIVATE ${QT_LIBS} ${QT_CORE} utun pthread dl ${FFMPEG_TARGET} ${FFMPEG_LIBRARIES}) + target_link_libraries(test_video_engine PRIVATE ${QT_LIBS} ${QT_CORE} utun_voice utun pthread dl ${FFMPEG_TARGET} ${FFMPEG_LIBRARIES}) endif() add_executable(test_ptt_key_chord tests/test_ptt_key_chord.cpp) target_link_libraries(test_ptt_key_chord PRIVATE ${QT_CORE}) add_test(NAME test_ptt_key_chord COMMAND test_ptt_key_chord) -add_executable(test_voice_file tests/test_voice_file.cpp src/voiceplayback.cpp src/sound_manager.cpp src/audio_device.cpp transport/miniaudio_impl.c transport/audio_diagnostics.cpp) +add_executable(test_voice_file tests/test_voice_file.cpp src/voiceplayback.cpp src/sound_manager.cpp src/voice_audio_io.cpp src/audio_device.cpp transport/miniaudio_impl.c transport/audio_diagnostics.cpp) target_include_directories(test_voice_file PRIVATE ${UTUN_INCLUDE_DIRS}) -target_link_libraries(test_voice_file PRIVATE ${QT_LIBS} utun pthread ${CMAKE_DL_LIBS}) +target_link_libraries(test_voice_file PRIVATE ${QT_LIBS} utun_voice utun pthread ${CMAKE_DL_LIBS}) add_test(NAME test_voice_file COMMAND test_voice_file) if(CMAKE_SYSTEM_NAME STREQUAL "Linux") - add_executable(test_audio_recovery tests/test_audio_recovery.cpp src/audio_device.cpp src/sound_manager.cpp + add_executable(test_audio_recovery tests/test_audio_recovery.cpp src/audio_device.cpp src/sound_manager.cpp src/voice_audio_io.cpp src/audiorecorder.cpp src/call_audio_engine.cpp src/radio_audio_engine.cpp transport/miniaudio_impl.c transport/audio_diagnostics.cpp) target_include_directories(test_audio_recovery PRIVATE ${UTUN_INCLUDE_DIRS}) diff --git a/tools/chatgui/src/audio_device.cpp b/tools/chatgui/src/audio_device.cpp index c76445fc..ec2918ee 100644 --- a/tools/chatgui/src/audio_device.cpp +++ b/tools/chatgui/src/audio_device.cpp @@ -3,6 +3,11 @@ #include "debug_config.h" #include #include +#include +extern "C" int64_t audio_stream_frame_time(ma_device* device, unsigned int frames, int* valid); +extern "C" uint64_t audio_stream_route(ma_device* device, int capture); +extern "C" int audio_stream_capture_missing(ma_device* device); +extern "C" void audio_stream_log_route(ma_device* device, int capture); AudioDevice::AudioDevice() { m_clock.start(); } AudioDevice::~AudioDevice() { close(); } @@ -14,6 +19,7 @@ bool AudioDevice::init(ma_context* context, ma_device_config config, const char* if (!context) { retryLater(); return false; } auto* device = new ma_device{}; config.dataCallback = data; + config.noFixedSizedCallback = MA_TRUE; // Аппаратные чанки не требуют скрытого накопителя miniaudio. config.notificationCallback = notification; config.pUserData = this; ma_result result = ma_device_init(context, &config, device); @@ -25,6 +31,8 @@ bool AudioDevice::init(ma_context* context, ma_device_config config, const char* } m_device = device; m_frames = 0; + m_route = 0; m_reportedRoute = 0; + m_timingMeasured = false; m_reportedTiming = -1; m_seenFrames = 0; m_suspended = false; audio_diag_label(device, session, role); @@ -56,10 +64,30 @@ void AudioDevice::close() { DEBUG_INFO(DEBUG_CATEGORY_CALL, "audio: %s closed", m_role); } m_callback = {}; + m_route = 0; +} + +void AudioDevice::pause() { + if (!m_device) return; + ma_result result = ma_device_stop(m_device); + if (result != MA_SUCCESS) { + DEBUG_ERROR(DEBUG_CATEGORY_CALL, "audio: %s pause failed result=%d", m_role, result); + close(); + } } void AudioDevice::data(ma_device* device, void* out, const void* in, ma_uint32 frames) { auto* self = static_cast(device->pUserData); + int valid = 0; + bool capture = device->type == ma_device_type_capture; + int64_t time = audio_stream_frame_time(device, frames, &valid); + int64_t now = std::chrono::duration_cast(std::chrono::steady_clock::now().time_since_epoch()).count(); + if (!time) time = capture ? now - (int64_t)frames * 1000000 / device->sampleRate : now; + self->m_callbackTime.store(time, std::memory_order_relaxed); + self->m_timingMeasured.store(valid, std::memory_order_relaxed); + uint64_t route = audio_stream_route(device, capture); + if (route != self->m_route.exchange(route, std::memory_order_relaxed)) audio_stream_log_route(device, capture); + self->m_captureMissing.store(audio_stream_capture_missing(device), std::memory_order_relaxed); if (self->m_callback) self->m_callback(in, out, frames); self->m_frames.fetch_add(frames, std::memory_order_relaxed); } @@ -73,6 +101,13 @@ void AudioDevice::notification(const ma_device_notification* event) { bool AudioDevice::needsRecovery() { if (!m_device) return false; uint64_t frames = m_frames.load(std::memory_order_relaxed); + uint64_t route = m_route.load(); + int measured = m_timingMeasured.load(); + if (frames && (route != m_reportedRoute || measured != m_reportedTiming)) { + DEBUG_INFO(DEBUG_CATEGORY_AEC, "audio: %s timing=%s physical_route=%016llx", m_role, + measured ? "backend-snapshot" : "callback-estimate", (unsigned long long)route); + m_reportedRoute = route; m_reportedTiming = measured; + } qint64 now = m_clock.elapsed(); if (frames != m_seenFrames || m_suspended.load()) { m_seenFrames = frames; diff --git a/tools/chatgui/src/audio_device.h b/tools/chatgui/src/audio_device.h index f20ebae5..ea74ab46 100644 --- a/tools/chatgui/src/audio_device.h +++ b/tools/chatgui/src/audio_device.h @@ -16,6 +16,7 @@ public: AudioDevice& operator=(const AudioDevice&) = delete; bool init(ma_context* context, ma_device_config config, const char* role, uint64_t session, Callback callback); bool start(); + void pause(); // GUI: завершает текущий callback, сохраняя backend stream и его физический route. void close(); bool needsRecovery(); bool retryReady() const; @@ -23,6 +24,10 @@ public: void retryNow(); bool initialized() const { return m_device != nullptr; } ma_device* device() const { return m_device; } + int64_t callbackTimeUs() const { return m_callbackTime.load(std::memory_order_relaxed); } + bool timingMeasured() const { return m_timingMeasured.load(std::memory_order_relaxed); } + uint64_t route() const { return m_route.load(std::memory_order_relaxed); } + bool captureMissing() const { return m_captureMissing.load(std::memory_order_relaxed); } private: static void data(ma_device* device, void* out, const void* in, ma_uint32 frames); static void notification(const ma_device_notification* event); @@ -31,6 +36,12 @@ private: const char* m_role = "audio"; QElapsedTimer m_clock; std::atomic m_frames{0}; + std::atomic m_callbackTime{0}; + std::atomic m_timingMeasured{false}; + std::atomic m_route{0}; + std::atomic m_captureMissing{false}; + uint64_t m_reportedRoute = 0; + int m_reportedTiming = -1; std::atomic m_suspended{false}; uint64_t m_seenFrames = 0; qint64 m_progressAt = 0, m_retryAt = 0; diff --git a/tools/chatgui/src/audio_event_queue.h b/tools/chatgui/src/audio_event_queue.h new file mode 100644 index 00000000..658a7507 --- /dev/null +++ b/tools/chatgui/src/audio_event_queue.h @@ -0,0 +1,54 @@ +#pragma once +#include +#include +#include +#include + +/* Ограниченная очередь с несколькими producers. Producer не меняет позицию consumer. + * Резервирование и публикация разделены sequence каждого слота; callback не ждёт занятый слот. + * reset допустим только до запуска всех участников. Последние 8 слотов зарезервированы для команд. */ +template class AudioEventQueue { + static_assert(Capacity > 8); + struct Slot { std::atomic sequence{0}; T value{}; }; + std::array m_slots; + alignas(64) std::atomic m_write{0}; + alignas(64) std::atomic m_read{0}; +public: + AudioEventQueue() { reset(); } + void reset() { + m_write = 0; m_read = 0; + for (size_t i = 0; i < Capacity; ++i) m_slots[i].sequence = i; + } + bool push(const T& value, bool control = false) { + uint64_t position = m_write.load(std::memory_order_relaxed); + for (int attempt = 0; attempt < 8; ++attempt) { + uint64_t read = m_read.load(std::memory_order_acquire); + if (read > position) { position = m_write.load(std::memory_order_relaxed); continue; } + if (position - read >= Capacity - (control ? 0 : 8)) return false; + auto& slot = m_slots[position % Capacity]; + int64_t difference = (int64_t)(slot.sequence.load(std::memory_order_acquire) - position); + if (difference < 0) return false; + if (difference == 0 && m_write.compare_exchange_weak(position, position + 1, std::memory_order_relaxed)) { + slot.value = value; + slot.sequence.store(position + 1, std::memory_order_release); + return true; + } + position = m_write.load(std::memory_order_relaxed); + } + return false; + } + bool pop(T& value) { + uint64_t position = m_read.load(std::memory_order_relaxed); + auto& slot = m_slots[position % Capacity]; + if (slot.sequence.load(std::memory_order_acquire) != position + 1) return false; + value = slot.value; + slot.sequence.store(position + Capacity, std::memory_order_release); + m_read.store(position + 1, std::memory_order_release); + return true; + } + size_t size() const { + uint64_t read = m_read.load(std::memory_order_acquire); + uint64_t count = m_write.load(std::memory_order_acquire) - read; + return (size_t)(count > Capacity ? Capacity : count); // Приблизительный снимок, включая ещё не опубликованные слоты. + } +}; diff --git a/tools/chatgui/src/audiodevicesettingspage.cpp b/tools/chatgui/src/audiodevicesettingspage.cpp index abe29758..a72e9489 100644 --- a/tools/chatgui/src/audiodevicesettingspage.cpp +++ b/tools/chatgui/src/audiodevicesettingspage.cpp @@ -363,7 +363,7 @@ void AudioDeviceSettingsPage::onTestMicToggled() { m_testRecorder->init(); QString id = m_inputCombo->currentData().toString(); m_testRecorder->setCaptureDevice(id); - m_testRecorder->startRecording(); + m_testRecorder->startRecording(true); m_micTestActive = true; m_levelTimer->start(); m_testMicBtn->setText("Stop Test"); diff --git a/tools/chatgui/src/audiorecorder.cpp b/tools/chatgui/src/audiorecorder.cpp index 68115bcf..34376ce7 100644 --- a/tools/chatgui/src/audiorecorder.cpp +++ b/tools/chatgui/src/audiorecorder.cpp @@ -12,29 +12,50 @@ extern "C" { #include #include #include - -void AudioRecorder::captureCallback(const void* pInput, unsigned int frameCount) { - if (!pInput || !frameCount || !m_recording.load()) return; - const int16_t* src = (const int16_t*)pInput; - size_t add = (size_t)frameCount * (size_t)m_channels; - - if (m_compressor && m_captureCompressorEnabled) { - audio_compressor_push(m_compressor, src, add); - } else { - size_t cur = m_pcmBuffer.size(); - m_pcmBuffer.resize(cur + add); - std::memcpy(m_pcmBuffer.data() + cur, src, add * sizeof(int16_t)); +#include +#include +#include +#include + +/* Аппаратный callback не выделяет память и не обрабатывает компрессор. */ +void AudioRecorder::captureCallback(const void* input, unsigned frames) { + if (!input || !m_recording.load() || m_captureFailed.load()) return; + auto* pcm = static_cast(input); + for (unsigned offset = 0; offset < frames;) { + CaptureBlock block{}; block.frames = std::min(frames - offset, 960u); + memcpy(block.pcm, pcm + offset * m_channels, block.frames * m_channels * sizeof(int16_t)); + if (!m_captureQueue.push(block)) { m_captureFailed = true; break; } + offset += block.frames; } + m_wake.notify_one(); +} - float sumSq = 0.0f; - for (unsigned int i = 0; i < frameCount * (unsigned int)m_channels; i++) { - float v = (float)src[i] / 32768.0f; - sumSq += v * v; +/* Только worker изменяет PCM и compressor; GUI получает атомарные метрики. */ +void AudioRecorder::recordLoop() { + try { + while (m_workerRunning || m_captureQueue.size()) { + CaptureBlock block{}; + if (!m_captureQueue.pop(block)) { + std::unique_lock lock(m_waitMutex); + m_wake.wait_for(lock, std::chrono::milliseconds(5)); + continue; + } + if (m_captureFailed) continue; + size_t count = (size_t)block.frames * m_channels; + if (!m_meterOnly) { + if (m_compressor && m_captureCompressorEnabled) { + if (audio_compressor_push(m_compressor, block.pcm, count) != 0) { m_captureFailed = true; continue; } + } else m_pcmBuffer.insert(m_pcmBuffer.end(), block.pcm, block.pcm + count); + } + double sum = 0; + for (size_t i = 0; i < count; ++i) { double value = block.pcm[i] / 32768.0; sum += value * value; } + m_peakLevel = (float)std::sqrt(sum / count); + m_recordedFrames.fetch_add(block.frames, std::memory_order_relaxed); + } + } catch (const std::exception& error) { + m_captureFailed = true; + DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "recorder: worker failed reason=%s", error.what()); } - float rms = sqrtf(sumSq / (float)(frameCount * m_channels)); - m_peakLevel = rms; - - m_recordedFrames.fetch_add(frameCount, std::memory_order_relaxed); } AudioRecorder::AudioRecorder(QObject* parent) @@ -90,17 +111,26 @@ void AudioRecorder::shutdown() { m_initialized = false; } -void AudioRecorder::startRecording() { +void AudioRecorder::startRecording(bool meterOnly) { if (m_recording) return; if (!m_initialized) { DEBUG_ERROR(DEBUG_CATEGORY_CALL, "recorder: not initialized"); return; } - m_pcmBuffer.clear(); + m_pcmBuffer.clear(); m_meterOnly = meterOnly; + m_captureQueue.reset(); m_captureFailed = false; m_recordedFrames = 0; m_elapsedMs = 0; m_peakLevel = 0; - m_captureCompressorEnabled = m_compressorEnabled; + m_captureCompressorEnabled = !meterOnly && m_compressorEnabled; if (m_compressor) { - audio_compressor_configure(m_compressor, &m_compressorConfig); + if (audio_compressor_configure(m_compressor, &m_compressorConfig) != 0) { + DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "recorder: cannot configure compressor"); return; + } audio_compressor_set_enabled(m_compressor, m_captureCompressorEnabled); audio_compressor_reset(m_compressor); } + m_workerRunning = true; + try { m_recordWorker = std::thread(&AudioRecorder::recordLoop, this); } + catch (const std::system_error& error) { + m_workerRunning = false; + DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "recorder: worker start failed reason=%s", error.what()); return; + } m_recording = true; m_capture.retryNow(); recoverCapture(); @@ -111,6 +141,10 @@ void AudioRecorder::startRecording() { void AudioRecorder::recoverCapture() { if (!m_recording) return; + if (m_captureFailed) { + DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "recorder: capture loss or processing failure; recording rejected"); + stopRecording(); return; + } if (m_capture.needsRecovery()) { m_capture.close(); m_durationTimer.stop(); m_capture.retryLater(); } if (!m_capture.retryReady()) return; ma_context* context = SoundManager::instance()->context(); @@ -137,6 +171,8 @@ void AudioRecorder::stopRecording() { if (!m_recording.exchange(false)) return; m_recoveryTimer.stop(); m_durationTimer.stop(); m_capture.close(); + m_workerRunning = false; m_wake.notify_one(); + if (m_recordWorker.joinable()) m_recordWorker.join(); m_elapsedMs = (int)(m_recordedFrames.load() * 1000 / m_sampleRate); int duration = m_elapsedMs; size_t samples = m_compressor && m_captureCompressorEnabled ? audio_compressor_output_size(m_compressor) : m_pcmBuffer.size(); @@ -151,7 +187,7 @@ int AudioRecorder::durationMs() const { /* Worker получает владельца буфера; новая запись имеет отдельное состояние. */ bool AudioRecorder::takeRecording(struct attachment_send_req* request) { - if (!request || m_recording) { DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "recorder: cannot detach active recording"); return false; } + if (!request || m_recording || m_meterOnly || m_captureFailed) { DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "recorder: cannot detach active recording"); return false; } if (m_compressor && m_captureCompressorEnabled) { auto* replacement = audio_compressor_clone_config(m_compressor); if (!replacement) return false; diff --git a/tools/chatgui/src/audiorecorder.h b/tools/chatgui/src/audiorecorder.h index e4012672..b6ea30fe 100644 --- a/tools/chatgui/src/audiorecorder.h +++ b/tools/chatgui/src/audiorecorder.h @@ -6,6 +6,10 @@ #include #include #include "audio_device.h" +#include "audio_event_queue.h" +#include +#include +#include extern "C" { #include "audio_compressor.h" } @@ -26,7 +30,7 @@ public: void setCaptureDevice(const QString& id) { m_captureId = id; m_customDevice = true; } - void startRecording(); + void startRecording(bool meterOnly = false); void stopRecording(); bool isRecording() const { return m_recording; } @@ -49,6 +53,14 @@ private: QTimer m_recoveryTimer; void recoverCapture(); void captureCallback(const void* input, unsigned int frames); + void recordLoop(); + struct CaptureBlock { unsigned frames = 0; int16_t pcm[1920]{}; }; + AudioEventQueue m_captureQueue; + std::thread m_recordWorker; + std::mutex m_waitMutex; + std::condition_variable m_wake; + std::atomic m_workerRunning{false}, m_captureFailed{false}; + bool m_meterOnly = false; bool m_initialized = false; std::atomic m_recording{false}; int m_sampleRate = 48000; diff --git a/tools/chatgui/src/call_audio_engine.cpp b/tools/chatgui/src/call_audio_engine.cpp index 35b3b0c5..3d537a2b 100644 --- a/tools/chatgui/src/call_audio_engine.cpp +++ b/tools/chatgui/src/call_audio_engine.cpp @@ -4,6 +4,7 @@ #include "../transport/audio_diagnostics.h" extern "C" { #include "call/call_audio.h" +#include "chat/chat_setting.h" #include "debug_config.h" } #include @@ -24,7 +25,7 @@ CallAudioEngine::CallAudioEngine() { connect(&m_recoveryTimer, &QTimer::timeout, this, &CallAudioEngine::recoverDevices); connect(SoundManager::instance(), &SoundManager::contextAboutToReset, this, &CallAudioEngine::pauseDevices); connect(SoundManager::instance(), &SoundManager::contextRestored, this, [this]() { - m_capture.retryNow(); m_playback.retryNow(); + m_capture.retryNow(); if (m_active) recoverDevices(); }); } @@ -32,100 +33,90 @@ CallAudioEngine::CallAudioEngine() { int CallAudioEngine::start(quint64 callId) { if (m_active && m_callId == callId && !m_ending) return 0; stop(); - if (!gui_bridge_get_inst() || call_audio_start(callId) != 0) { + if (!gui_bridge_get_inst() || call_audio_start_prepared(callId) != 0) { DEBUG_ERROR(DEBUG_CATEGORY_CALL, "CallAudio: cannot start call=%016llx", (unsigned long long)callId); return -1; } ++m_generation; m_callId = callId; - m_captureBuf.assign(CALL_AUDIO_FRAME_SAMPLES, 0); - m_capturePos = 0; + if (!m_voice.start(VoiceAudioIo::Kind::Call, callId)) { call_audio_release(callId); m_callId = 0; return -1; } m_muted = false; m_ending = false; m_active = true; - m_capture.retryNow(); m_playback.retryNow(); + m_capture.retryNow(); recoverDevices(); m_recoveryTimer.start(); DEBUG_INFO(DEBUG_CATEGORY_CALL, "CallAudio: session started call=%016llx", (unsigned long long)callId); return 0; } -bool CallAudioEngine::openDevice(bool playback) { - AudioDevice& device = playback ? m_playback : m_capture; - ma_context* context = SoundManager::instance()->context(); +bool CallAudioEngine::openCapture() { + auto* context = SoundManager::instance()->context(); if (!context) return false; - ma_device_config config = ma_device_config_init(playback ? ma_device_type_playback : ma_device_type_capture); - config.capture.format = config.playback.format = ma_format_s16; - config.capture.channels = config.playback.channels = 1; - config.sampleRate = CALL_AUDIO_SAMPLE_RATE; + ma_device_config config = ma_device_config_init(ma_device_type_capture); + config.capture.format = ma_format_s16; config.capture.channels = 1; config.sampleRate = CALL_AUDIO_SAMPLE_RATE; ma_device_id id; - bool selected = SoundManager::deviceId(playback ? m_playbackId : m_captureId, &id); - if (selected) { - if (playback) config.playback.pDeviceID = &id; - else config.capture.pDeviceID = &id; - } - auto callback = [this](const void* in, void* out, unsigned int frames) { dataCallback(in, out, frames); }; - const char* role = playback ? "call-playback" : "call-capture"; + bool selected = SoundManager::deviceId(m_captureId, &id); + if (selected) config.capture.pDeviceID = &id; + auto callback = [this](const void* in, void*, unsigned frames) { dataCallback(in, nullptr, frames); }; auto attempt = [&]() { - if (!playback) m_capturePos = 0; - bool ready = device.init(context, config, role, m_callId.load(), callback) && device.start(); - if (!ready) call_audio_reset_io(m_callId.load(), playback); - return ready; + if (!m_capture.init(context, config, "call-capture", m_callId.load(), callback)) return false; + bool aec = chat_setting_get_int(gui_bridge_get_inst(), "aec_enabled", 1) != 0; + if (!m_voice.startCapture(1, aec)) { m_capture.close(); return false; } + m_voice.setMuted(m_muted.load()); + if (m_capture.start()) return true; + m_voice.stopCapture(); call_audio_reset_io(m_callId.load(), 0); + return false; }; if (attempt()) return true; if (!selected) return false; - DEBUG_WARN(DEBUG_CATEGORY_CALL, "CallAudio: %s selected device unavailable, trying default", role); - config.capture.pDeviceID = config.playback.pDeviceID = nullptr; + DEBUG_WARN(DEBUG_CATEGORY_CALL, "CallAudio: selected input unavailable, trying default"); + config.capture.pDeviceID = nullptr; return attempt(); } void CallAudioEngine::recoverDevices() { if (!m_active) return; - if (m_playback.needsRecovery()) { - m_playback.close(); - call_audio_reset_io(m_callId.load(), 1); - m_playback.retryLater(); - } - if (!m_ending && m_capture.needsRecovery()) { - m_capture.close(); - m_capturePos = 0; - call_audio_reset_io(m_callId.load(), 0); - m_capture.retryLater(); + auto* manager = SoundManager::instance(); + if (!m_outputAttached && manager->context()) m_outputAttached = manager->attachVoice(&m_voice, m_playbackId); + if (!m_ending && ((m_capture.initialized() && !m_voice.captureRunning()) || m_capture.needsRecovery())) { + m_capture.close(); m_voice.stopCapture(); + call_audio_reset_io(m_callId.load(), 0); m_capture.retryLater(); } - if (m_playback.retryReady()) openDevice(true); - if (!m_ending && m_capture.retryReady()) openDevice(false); + if (!m_ending && m_capture.retryReady()) openCapture(); + if (!m_ending) m_voice.setAecEnabled(chat_setting_get_int(gui_bridge_get_inst(), "aec_enabled", 1) != 0); } void CallAudioEngine::pauseDevices() { - m_capture.close(); m_playback.close(); - m_capturePos = 0; - if (m_active) call_audio_reset_io(m_callId.load(), 1); + m_capture.close(); m_voice.stopCapture(); + SoundManager::instance()->detachVoice(&m_voice); m_outputAttached = false; + m_voice.resetPlayback(); } void CallAudioEngine::setMuted(bool muted) { if (m_muted.exchange(muted) == muted) return; + m_voice.setMuted(muted); DEBUG_INFO(DEBUG_CATEGORY_CALL, "CallAudio: mute=%d", muted); } void CallAudioEngine::setPlaybackDevice(const QString& id) { if (id == m_playbackId) return; m_playbackId = id; - m_playback.close(); m_playback.retryNow(); - if (m_active) { call_audio_reset_io(m_callId.load(), 1); recoverDevices(); } + SoundManager::instance()->detachVoice(&m_voice); m_outputAttached = false; + if (m_active) recoverDevices(); } void CallAudioEngine::setCaptureDevice(const QString& id) { if (id == m_captureId) return; m_captureId = id; - m_capture.close(); m_capture.retryNow(); - m_capturePos = 0; + m_capture.close(); m_voice.stopCapture(); m_capture.retryNow(); if (m_active) { call_audio_reset_io(m_callId.load(), 0); recoverDevices(); } } void CallAudioEngine::beginEnd(quint64 callId) { if (!m_active || m_callId != callId || m_ending) return; m_ending = true; - m_capture.close(); + m_capture.close(); m_voice.stopCapture(); call_audio_begin_end(callId); uint64_t generation = m_generation; QTimer::singleShot(500, this, [this, generation]() { if (m_generation == generation) stop(); }); @@ -135,7 +126,7 @@ void CallAudioEngine::stop() { ++m_generation; m_recoveryTimer.stop(); m_active = false; - pauseDevices(); + pauseDevices(); m_voice.stop(); uint64_t id = m_callId.exchange(0); if (id) call_audio_release(id); m_muted = false; @@ -143,22 +134,6 @@ void CallAudioEngine::stop() { if (id) DEBUG_INFO(DEBUG_CATEGORY_CALL, "CallAudio: session stopped call=%016llx", (unsigned long long)id); } -void CallAudioEngine::dataCallback(const void* in, void* out, unsigned int frames) { - if (in) { - const auto* pcm = static_cast(in); - bool muted = m_muted.load(); - for (unsigned int i = 0; i < frames; ++i) { - m_captureBuf[m_capturePos++] = pcm[i]; - if (m_capturePos == CALL_AUDIO_FRAME_SAMPLES) { - int result = call_audio_feed_pcm_muted(m_callId.load(), m_captureBuf.data(), CALL_AUDIO_FRAME_SAMPLES, muted); - if (result != 0) audio_diag_event(m_capture.device(), AUDIO_DIAG_ERROR, AUDIO_DIAG_FEED, 1, result); - m_capturePos = 0; - } - } - } - if (out) { - auto* pcm = static_cast(out); - int n = call_audio_pull_pcm(m_callId.load(), pcm, (int)frames); - if (n < (int)frames) std::memset(pcm + n, 0, (frames - n) * sizeof(*pcm)); - } +void CallAudioEngine::dataCallback(const void* in, void*, unsigned int frames) { + if (in) m_voice.capture(static_cast(in), frames, m_capture.callbackTimeUs(), m_capture.route(), m_capture.captureMissing()); } diff --git a/tools/chatgui/src/call_audio_engine.h b/tools/chatgui/src/call_audio_engine.h index 05dc89c7..102d55ad 100644 --- a/tools/chatgui/src/call_audio_engine.h +++ b/tools/chatgui/src/call_audio_engine.h @@ -6,16 +6,16 @@ #include #include #include "audio_device.h" +#include "voice_audio_io.h" struct ma_device; -/* Аудио-контур звонка (desktop): только miniaudio I/O + захват/воспроизведение. - * Весь аудиопроцессинг (Opus, джиттер-буфер, time-stretch, тоны) — в сервисном - * слое src/call/call_audio.c; сюда приходит/отсюда уходит финальный PCM. - * - * Потоки: - * - независимые callbacks capture → call_audio_feed_pcm и call_audio_pull_pcm → playback; - * - GUI-поток: start()/beginEnd()/stop() по ACCEPTED/ENDED. */ +/* Desktop-контур звонка: GUI владеет capture device и жизненным циклом. + * Hardware callback копирует PCM в ограниченную очередь VoiceAudioIo. + * Capture worker выполняет AEC и упорядочивает PCM/PTT/mute; независимый RX worker + * получает готовый голос из сервисного слоя. SoundManager смешивает его со всеми + * звуками приложения на том же выходе и возвращает итоговый stereo reference. + * Stop: close capture → join capture worker → detach output → join RX → release core. */ class CallAudioEngine : public QObject { Q_OBJECT public: @@ -42,12 +42,12 @@ public: QString playbackDevice() const { return m_playbackId; } void setCaptureDevice(const QString& id); - /* аудио-поток: miniaudio data callback (захват + воспроизведение) */ + /* Hardware capture callback: без DSP и растущих буферов. */ void dataCallback(const void* in, void* out, unsigned int frames); private: CallAudioEngine(); - bool openDevice(bool playback); + bool openCapture(); void recoverDevices(); void pauseDevices(); @@ -58,11 +58,11 @@ private: QString m_playbackId; /* только GUI-поток */ QString m_captureId; - AudioDevice m_capture, m_playback; + AudioDevice m_capture; + VoiceAudioIo m_voice; + bool m_outputAttached = false; QTimer m_recoveryTimer; uint64_t m_generation = 0; bool m_ending = false; - std::vector m_captureBuf; /* 960 samples, аудио-поток */ - int m_capturePos = 0; }; diff --git a/tools/chatgui/src/radio_audio_engine.cpp b/tools/chatgui/src/radio_audio_engine.cpp index 3d608bd7..16841a36 100644 --- a/tools/chatgui/src/radio_audio_engine.cpp +++ b/tools/chatgui/src/radio_audio_engine.cpp @@ -26,11 +26,11 @@ RadioAudioEngine::RadioAudioEngine() { auto* manager = SoundManager::instance(); if (m_captureId != manager->captureDevice()) { m_captureId = manager->captureDevice(); - m_capture.close(); if (m_active) radio_audio_capture_stop(); m_capture.retryNow(); + m_capture.close(); m_voice.stopCapture(); if (m_active) radio_audio_capture_stop(); m_capture.retryNow(); } if (m_playbackId != manager->playbackDevice()) { m_playbackId = manager->playbackDevice(); - m_playback.close(); if (m_active) radio_audio_reset_playback(m_groupId.load()); m_playback.retryNow(); + manager->detachVoice(&m_voice); m_outputAttached = false; } recoverDevices(); }); @@ -38,7 +38,7 @@ RadioAudioEngine::RadioAudioEngine() { connect(&m_recoveryTimer, &QTimer::timeout, this, &RadioAudioEngine::recoverDevices); connect(SoundManager::instance(), &SoundManager::contextAboutToReset, this, &RadioAudioEngine::pauseDevices); connect(SoundManager::instance(), &SoundManager::contextRestored, this, [this]() { - m_capture.retryNow(); m_playback.retryNow(); + m_capture.retryNow(); recoverDevices(); }); } @@ -52,9 +52,10 @@ int RadioAudioEngine::start(quint64 groupId) { return -1; } m_groupId = groupId; + if (!m_voice.start(VoiceAudioIo::Kind::Radio, groupId)) { radio_audio_stop(); m_groupId = 0; return -1; } m_wantStereo = chat_setting_get_int(inst, "radio_capture_stereo", 0) != 0; m_active = true; - m_capture.retryNow(); m_playback.retryNow(); + m_capture.retryNow(); recoverDevices(); m_recoveryTimer.start(); DEBUG_INFO(DEBUG_CATEGORY_RADIO, "RadioAudio: listening group=%016llx", (unsigned long long)groupId); @@ -77,65 +78,44 @@ void RadioAudioEngine::openCapture() { if (attempt) DEBUG_WARN(DEBUG_CATEGORY_RADIO, "RadioAudio: selected input unavailable, trying default ch=%d", channels); if (!m_capture.init(context, config, "radio-capture", m_groupId.load(), callback)) continue; m_captureChannels = channels; - m_captureBuf.assign(RADIO_AUDIO_FRAME_SAMPLES * channels, 0); - m_capturePos = 0; + auto* inst = gui_bridge_get_inst(); int vad = chat_setting_get_int(inst, "radio_vad_enabled", 0); if (radio_audio_capture_start(m_groupId.load(), channels, vad) != 0) { m_capture.close(); m_capture.retryLater(); return; } - m_vadMode = radio_audio_vad_mode() != 0; - if (!m_capture.start()) { radio_audio_capture_stop(); continue; } - if (m_talking) radio_audio_talk_begin(m_groupId.load()); + if (!m_voice.startCapture(channels, chat_setting_get_int(inst, "aec_enabled", 1) != 0)) { + m_capture.close(); radio_audio_capture_stop(); return; + } + if (m_talking) m_voice.setPtt(true); + if (!m_capture.start()) { m_voice.stopCapture(); radio_audio_capture_stop(); continue; } if (channels == 1 && m_wantStereo) DEBUG_WARN(DEBUG_CATEGORY_RADIO, "RadioAudio: stereo input unavailable, using mono"); return; } } } -void RadioAudioEngine::openPlayback() { - auto* context = SoundManager::instance()->context(); - if (!context) return; - ma_device_config config = ma_device_config_init(ma_device_type_playback); - config.playback.format = ma_format_s16; - config.playback.channels = 2; - config.sampleRate = RADIO_AUDIO_SAMPLE_RATE; - auto callback = [this](const void*, void* out, unsigned int frames) { playbackCallback(out, frames); }; - ma_device_id id; - bool selected = SoundManager::deviceId(m_playbackId, &id); - if (selected) config.playback.pDeviceID = &id; - if (m_playback.init(context, config, "radio-playback", m_groupId.load(), callback) && m_playback.start()) return; - if (!selected) return; - DEBUG_WARN(DEBUG_CATEGORY_RADIO, "RadioAudio: selected output unavailable, trying default"); - config.playback.pDeviceID = nullptr; - if (m_playback.init(context, config, "radio-playback", m_groupId.load(), callback)) m_playback.start(); -} - void RadioAudioEngine::recoverDevices() { if (!m_active) return; - if (m_capture.needsRecovery()) { - m_capture.close(); radio_audio_capture_stop(); - m_capturePos = 0; m_capture.retryLater(); - } - if (m_playback.needsRecovery()) { - m_playback.close(); radio_audio_reset_playback(m_groupId.load()); m_playback.retryLater(); + auto* manager = SoundManager::instance(); + if (!m_outputAttached && manager->context()) m_outputAttached = manager->attachVoice(&m_voice, m_playbackId); + if ((m_capture.initialized() && !m_voice.captureRunning()) || m_capture.needsRecovery()) { + m_capture.close(); m_voice.stopCapture(); radio_audio_capture_stop(); m_capture.retryLater(); } - if (m_playback.retryReady()) openPlayback(); if (m_capture.retryReady()) openCapture(); + m_voice.setAecEnabled(chat_setting_get_int(gui_bridge_get_inst(), "aec_enabled", 1) != 0); } void RadioAudioEngine::pauseDevices() { - m_capture.close(); m_playback.close(); - m_capturePos = 0; - if (m_active) { - radio_audio_capture_stop(); - radio_audio_reset_playback(m_groupId.load()); - } + m_capture.close(); m_voice.stopCapture(); + SoundManager::instance()->detachVoice(&m_voice); m_outputAttached = false; + if (m_active) radio_audio_capture_stop(); + m_voice.resetPlayback(); } void RadioAudioEngine::stop() { m_recoveryTimer.stop(); - m_capture.close(); m_playback.close(); + pauseDevices(); m_voice.stop(); uint64_t group = m_groupId.exchange(0); bool talking = m_talking.exchange(false); m_buttonTalking = m_hotkeyTalking = false; @@ -164,41 +144,12 @@ void RadioAudioEngine::setHotkeyTalking(bool pressed) { void RadioAudioEngine::updateTalking() { bool talking = m_active && (m_buttonTalking || m_hotkeyTalking); if (m_talking.exchange(talking) == talking || !m_active) return; - if (talking) radio_audio_talk_begin(m_groupId.load()); - else radio_audio_talk_end(m_groupId.load()); + m_voice.setPtt(talking); DEBUG_DEBUG(DEBUG_CATEGORY_RADIO, "RadioAudio: PTT %s group=%016llx button=%d hotkey=%d", talking ? "started" : "stopped", (unsigned long long)m_groupId.load(), m_buttonTalking, m_hotkeyTalking); emit talkingChanged(talking); } void RadioAudioEngine::captureCallback(const void* in, unsigned int frames) { - const int16_t* in16 = (const int16_t*)in; - if (!in16) return; - /* VAD-режим: захват фидится постоянно (решение «говорить» — в radio_audio). - * Обычный PTT: только когда зажата кнопка. */ - if (!m_vadMode.load() && !m_talking.load()) return; - - /* захват: копим FRAME_SAMPLES * captureChannels → encode+send. */ - int totalIn = (int)frames * m_captureChannels; - for (int i = 0; i < totalIn; i++) { - m_captureBuf[m_capturePos++] = in16[i]; - if (m_capturePos >= (int)m_captureBuf.size()) { - int result = radio_audio_feed_pcm(m_groupId.load(), m_captureBuf.data(), (int)m_captureBuf.size()); - if (result != 0) audio_diag_event(m_capture.device(), AUDIO_DIAG_ERROR, AUDIO_DIAG_FEED, 1, result); - m_capturePos = 0; - } - } -} - -void RadioAudioEngine::playbackCallback(void* out, unsigned int frames) { - int16_t* out16 = (int16_t*)out; - - /* воспроизведение: микшированный PCM из сервисного слоя (всегда стерео) */ - unsigned int offset = 0; - while (offset < frames) { - unsigned int count = std::min(frames - offset, (unsigned int)RADIO_AUDIO_FRAME_SAMPLES); - int n = radio_audio_pull_pcm(m_groupId.load(), out16 + offset * 2, count * 2); - if (n < (int)count * 2) std::memset(out16 + offset * 2 + n, 0, (count * 2 - n) * sizeof(*out16)); - offset += count; - } + if (in) m_voice.capture(static_cast(in), frames, m_capture.callbackTimeUs(), m_capture.route(), m_capture.captureMissing()); } diff --git a/tools/chatgui/src/radio_audio_engine.h b/tools/chatgui/src/radio_audio_engine.h index d3ddcb40..b879b19d 100644 --- a/tools/chatgui/src/radio_audio_engine.h +++ b/tools/chatgui/src/radio_audio_engine.h @@ -5,18 +5,17 @@ #include #include #include "audio_device.h" +#include "voice_audio_io.h" #include struct ma_device; -/* Аудио-контур рации (desktop): только miniaudio I/O + захват/воспроизведение. - * Весь аудиопроцессинг (Opus, per-source декодеры, микшер) — в сервисном слое - * src/radio/radio_audio.c; сюда приходит/отсюда уходит финальный PCM. - * - * Потоки: - * - аудио-поток (miniaudio callback): capture → radio_audio_feed_pcm, - * radio_audio_pull_pcm → playback; - * - GUI-поток: start()/stop() по включению/выключению рации канала. */ +/* Desktop-контур рации: GUI владеет capture device и жизненным циклом. + * Hardware callback копирует PCM в ограниченную очередь VoiceAudioIo. + * Capture worker выполняет AEC и упорядочивает PCM/PTT/mute; независимый RX worker + * получает готовый голос из сервисного слоя. SoundManager смешивает его со всеми + * звуками приложения на том же выходе и возвращает итоговый stereo reference. + * Stop: close capture → join capture worker → detach output → join RX → release core. */ class RadioAudioEngine : public QObject { Q_OBJECT public: @@ -37,9 +36,8 @@ public: bool isActive() const { return m_active.load(); } quint64 groupId() const { return m_groupId.load(); } - /* аудио-поток: miniaudio callbacks (захват и воспроизведение раздельно) */ + /* Hardware capture callback: без DSP и растущих буферов. */ void captureCallback(const void* in, unsigned int frames); - void playbackCallback(void* out, unsigned int frames); signals: void talkingChanged(bool talking); // GUI-поток, суммарное состояние кнопки и хоткея @@ -51,19 +49,18 @@ private: void pauseDevices(); void recoverDevices(); void openCapture(); - void openPlayback(); + std::atomic m_active{false}; std::atomic m_talking{false}; - std::atomic m_vadMode{false}; /* VAD авто-PTT: фидим PCM постоянно */ std::atomic m_groupId{0}; - AudioDevice m_capture, m_playback; + AudioDevice m_capture; + VoiceAudioIo m_voice; + bool m_outputAttached = false; QTimer m_recoveryTimer; bool m_wantStereo = false; QString m_captureId, m_playbackId; - std::vector m_captureBuf; /* FRAME_SAMPLES * m_captureChannels, аудио-поток */ - int m_capturePos = 0; - int m_captureChannels = 1; /* 1=моно, 2=стерео — по настройке/возможности */ + int m_captureChannels = 1; }; diff --git a/tools/chatgui/src/sound_manager.cpp b/tools/chatgui/src/sound_manager.cpp index 29884c13..bc5a6e4a 100644 --- a/tools/chatgui/src/sound_manager.cpp +++ b/tools/chatgui/src/sound_manager.cpp @@ -1,11 +1,13 @@ #include "miniaudio.h" #include "sound_manager.h" +#include "voice_audio_io.h" #include "../transport/audio_diagnostics.h" #include #include #include #include +#include #include "../../lib/debug_config.h" SoundManager* SoundManager::instance() { @@ -20,6 +22,14 @@ SoundManager::~SoundManager() { extern "C" int audio_context_failed(ma_context* context); +static uint64_t pulse_output_key(const QString& key) { + ma_device_id id{}; + if (!SoundManager::deviceId(key, &id)) return 0; + uint64_t hash = UINT64_C(14695981039346656037); + for (const char* name = id.pulse; *name; ++name) { hash ^= (unsigned char)*name; hash *= UINT64_C(1099511628211); } + return hash; +} + SoundManager::SoundManager() { m_clock.start(); m_recoveryTimer.setInterval(200); @@ -65,11 +75,15 @@ bool SoundManager::openContext() { m_backendSelected = true; m_contextRetryDelay = 250; DEBUG_INFO(DEBUG_CATEGORY_CALL, "SoundManager: context ready backend=%s", ma_get_backend_name(m_backend)); + m_resetting = false; emit contextRestored(); return true; } void SoundManager::openOutput() { + for (auto* voice : m_voices) if (voice) voice->resetPlayback(); + if (m_output.initialized() && m_output.start()) return; + m_restoreSelections = true; ma_device_config config = ma_device_config_init(ma_device_type_playback); config.playback.format = ma_format_f32; config.playback.channels = 2; @@ -77,33 +91,213 @@ void SoundManager::openOutput() { ma_device_id id; bool selected = deviceId(m_playbackId, &id); if (selected) config.playback.pDeviceID = &id; - auto callback = [this](const void*, void* out, unsigned int frames) { - ma_engine_read_pcm_frames(m_engine, out, frames, nullptr); + auto callback = [this, previousRoute = uint64_t(0)](const void*, void* out, unsigned int frames) mutable { + uint64_t route = m_output.route(); + if (previousRoute && route != previousRoute) for (auto* voice : m_voices) if (voice) voice->resetPlayback(); + previousRoute = route; + m_renderErrors.fetch_add(renderOutput(m_engine, m_voices, static_cast(out), frames, m_output.callbackTimeUs())); }; - if (m_output.init(m_context, config, "sound-engine", 0, callback) && m_output.start()) return; + if (m_output.init(m_context, config, "sound-engine", 0, callback) && m_output.start()) { + m_actualOutput = resolveOutput(m_playbackId); return; + } if (!selected) return; DEBUG_WARN(DEBUG_CATEGORY_CALL, "SoundManager: selected output unavailable, trying default"); config.playback.pDeviceID = nullptr; - if (m_output.init(m_context, config, "sound-engine", 0, callback)) m_output.start(); + if (m_output.init(m_context, config, "sound-engine", 0, callback) && m_output.start()) m_actualOutput = resolveOutput({}); } void SoundManager::recoverAudio() { if (!m_initialized) return; + unsigned errors = m_renderErrors.exchange(0); + if (errors) DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "audio: mixer read failures=%u; output replaced with silence", errors); for (auto it = m_pcmSessions.begin(); it != m_pcmSessions.end();) { if (it->second->transient && !isPcmPlaying(it->first)) it = m_pcmSessions.erase(it); else ++it; } if (m_context && audio_context_failed(m_context)) { DEBUG_ERROR(DEBUG_CATEGORY_CALL, "SoundManager: context failed, closing dependent devices"); + m_resetting = true; emit contextAboutToReset(); m_output.close(); + for (auto& [key, route] : m_voiceRoutes) route->device.close(); ma_context_uninit(m_context); delete m_context; m_context = nullptr; m_contextRetryAt = m_clock.elapsed() + 250; } if (!m_context && m_clock.elapsed() >= m_contextRetryAt) openContext(); if (!m_context) return; - if (m_output.needsRecovery()) { m_output.close(); m_output.retryLater(); } - if (m_output.retryReady()) openOutput(); + if (m_output.needsRecovery()) { + m_output.close(); + for (auto* voice : m_voices) if (voice) voice->resetPlayback(); + m_output.retryLater(); + } + bool reopened = m_output.retryReady(); + if (reopened) openOutput(); + for (auto& [key, route] : m_voiceRoutes) { + if (route->device.needsRecovery()) { + route->device.close(); + for (auto* voice : route->voices) if (voice) voice->resetPlayback(); + route->device.retryLater(); + } + if (route->device.retryReady()) openVoiceRoute(*route); + } + mergeVoiceRoutes(); + if (m_restoreSelections && m_output.initialized() && (m_backend != ma_backend_pulseaudio || m_output.route())) { + m_restoreSelections = false; + auto voices = m_voices; + for (auto* voice : voices) if (voice) { + QString wanted = m_voiceSelections.at(voice); + bool same = m_backend == ma_backend_pulseaudio ? pulse_output_key(wanted) == m_output.route() : wanted == m_actualOutput; + if (!wanted.isEmpty() && wanted == resolveOutput(wanted) && !same) attachVoice(voice, wanted); + } + } +} + +/* Последнее смешивание и ограничение: тот же PCM передаётся устройству и всем AEC этого выхода. */ +unsigned SoundManager::renderOutput(ma_engine* engine, const std::array& voices, + float* out, unsigned frames, int64_t firstTimeUs) { + unsigned errors = 0; + for (unsigned offset = 0; offset < frames;) { + unsigned count = std::min(frames - offset, 960u); + float* destination = out + offset * 2; + std::fill_n(destination, count * 2, 0.0f); + if (engine && ma_engine_read_pcm_frames(engine, destination, count, nullptr) != MA_SUCCESS) { + ++errors; std::fill_n(destination, count * 2, 0.0f); + } + for (auto* voice : voices) if (voice) voice->mix(destination, count); + for (unsigned i = 0; i < count * 2; ++i) { + float value = destination[i]; + destination[i] = std::isfinite(value) ? std::clamp(value, -1.0f, 1.0f) : 0.0f; + } + for (auto* voice : voices) if (voice) + voice->reference(destination, count, firstTimeUs + (int64_t)offset * 1000000 / 48000); + offset += count; + } + return errors; +} + +/* Default и явный ID одного устройства получают один микшер. */ +QString SoundManager::resolveOutput(const QString& requested) { + auto devices = enumeratePlaybackDevices(); + for (const auto& device : devices) if (device.id == requested) return requested; + for (const auto& device : devices) if (device.isDefault) return device.id; + return {}; +} + +bool SoundManager::openVoiceRoute(VoiceRoute& route) { + if (!m_context || m_resetting) return false; + for (auto* voice : route.voices) if (voice) voice->resetPlayback(); + if (route.device.initialized() && route.device.start()) return true; + ma_device_config config = ma_device_config_init(ma_device_type_playback); + config.playback.format = ma_format_f32; config.playback.channels = 2; config.sampleRate = 48000; + ma_device_id id; + bool selected = deviceId(route.wanted, &id); + if (selected) config.playback.pDeviceID = &id; + auto callback = [this, &route, previousRoute = uint64_t(0)](const void*, void* out, unsigned frames) mutable { + uint64_t physical = route.device.route(); + if (previousRoute && physical != previousRoute) for (auto* voice : route.voices) if (voice) voice->resetPlayback(); + previousRoute = physical; + if (physical && physical == m_output.route()) { + std::fill_n(static_cast(out), frames * 2, 0.0f); return; // GUI объединит совпавшие маршруты. + } + renderOutput(nullptr, route.voices, static_cast(out), frames, route.device.callbackTimeUs()); + }; + if (route.device.init(m_context, config, "voice-output", 0, callback) && route.device.start()) { + route.actual = resolveOutput(route.wanted); return true; + } + if (!selected) return false; + DEBUG_WARN(DEBUG_CATEGORY_CALL, "audio: voice output unavailable, trying default"); + config.playback.pDeviceID = nullptr; + if (route.device.init(m_context, config, "voice-output", 0, callback) && route.device.start()) { + route.actual = resolveOutput({}); return true; + } + return false; +} + +bool SoundManager::attachVoice(VoiceAudioIo* voice, const QString& output) { + detachVoice(voice); + if (!voice || !m_initialized || !m_context || m_resetting) return false; + size_t attached = std::count_if(m_voices.begin(), m_voices.end(), [](auto* v) { return v != nullptr; }); + for (const auto& [key, route] : m_voiceRoutes) + attached += std::count_if(route->voices.begin(), route->voices.end(), [](auto* v) { return v != nullptr; }); + if (attached >= 2) { DEBUG_ERROR(DEBUG_CATEGORY_CALL, "audio: only one call and one radio source are supported"); return false; } + QString key = resolveOutput(output); + QString root = m_actualOutput.isEmpty() ? resolveOutput(m_playbackId) : m_actualOutput; + bool same = key == root; + if (m_backend == ma_backend_pulseaudio && m_output.route()) same = pulse_output_key(key) == m_output.route(); + if (same) { + m_output.pause(); + auto slot = std::find(m_voices.begin(), m_voices.end(), nullptr); + if (slot == m_voices.end()) { DEBUG_ERROR(DEBUG_CATEGORY_CALL, "audio: too many voice sources on output"); openOutput(); return false; } + *slot = voice; voice->resetPlayback(); + openOutput(); + } else { + auto& route = m_voiceRoutes[key]; + if (!route) { route = std::make_unique(); route->wanted = key; } + route->device.pause(); + auto slot = std::find(route->voices.begin(), route->voices.end(), nullptr); + if (slot == route->voices.end()) { DEBUG_ERROR(DEBUG_CATEGORY_CALL, "audio: too many voice sources on separate output"); openVoiceRoute(*route); return false; } + *slot = voice; voice->resetPlayback(); + openVoiceRoute(*route); + mergeVoiceRoutes(); + } + DEBUG_INFO(DEBUG_CATEGORY_CALL, "audio: voice attached to shared output default=%d", output.isEmpty()); + m_voiceSelections[voice] = output; + return true; +} + +void SoundManager::detachVoice(VoiceAudioIo* voice) { + if (!voice) return; + m_voiceSelections.erase(voice); + auto slot = std::find(m_voices.begin(), m_voices.end(), voice); + if (slot != m_voices.end()) { + m_output.pause(); *slot = nullptr; + if (m_initialized && m_context && !m_resetting) openOutput(); + } + for (auto it = m_voiceRoutes.begin(); it != m_voiceRoutes.end();) { + auto& route = *it->second; + auto entry = std::find(route.voices.begin(), route.voices.end(), voice); + if (entry == route.voices.end()) { ++it; continue; } + route.device.pause(); *entry = nullptr; + if (std::all_of(route.voices.begin(), route.voices.end(), [](auto* v) { return !v; })) it = m_voiceRoutes.erase(it); + else { if (!m_resetting) openVoiceRoute(route); ++it; } + } +} + +/* Fallback на уже используемый выход не оставляет два независимых референса одного динамика. */ +void SoundManager::mergeVoiceRoutes() { + if (m_resetting || !m_output.initialized()) return; + for (auto first = m_voiceRoutes.begin(); first != m_voiceRoutes.end(); ++first) { + auto& target = *first->second; + for (auto next = std::next(first); next != m_voiceRoutes.end();) { + auto& source = *next->second; + uint64_t physical = target.device.route(); + if (!physical || physical != source.device.route()) { ++next; continue; } + target.device.pause(); source.device.close(); + for (auto* voice : source.voices) if (voice) { + auto slot = std::find(target.voices.begin(), target.voices.end(), nullptr); + if (slot == target.voices.end()) { DEBUG_ERROR(DEBUG_CATEGORY_CALL, "audio: voice route merge capacity exceeded"); return; } + *slot = voice; + } + next = m_voiceRoutes.erase(next); + DEBUG_INFO(DEBUG_CATEGORY_CALL, "audio: merged voice streams on physical route=%016llx", (unsigned long long)physical); + openVoiceRoute(target); + } + } + for (auto it = m_voiceRoutes.begin(); it != m_voiceRoutes.end();) { + auto& route = *it->second; + bool same = m_backend == ma_backend_pulseaudio ? m_output.route() && m_output.route() == route.device.route() + : route.actual == m_actualOutput; + if (!route.device.initialized() || !same) { ++it; continue; } + m_output.pause(); route.device.close(); + for (auto* voice : route.voices) if (voice) { + auto slot = std::find(m_voices.begin(), m_voices.end(), nullptr); + if (slot == m_voices.end()) { DEBUG_ERROR(DEBUG_CATEGORY_CALL, "audio: output merge capacity exceeded"); break; } + *slot = voice; voice->resetPlayback(); + } + it = m_voiceRoutes.erase(it); + DEBUG_INFO(DEBUG_CATEGORY_CALL, "audio: merged voice fallback with shared output"); + openOutput(); + } } void SoundManager::setDevices(const QString& output, const QString& input) { @@ -111,7 +305,7 @@ void SoundManager::setDevices(const QString& output, const QString& input) { bool outputChanged = output != m_playbackId; m_playbackId = output; m_captureId = input; DEBUG_INFO(DEBUG_CATEGORY_CALL, "audio: device selection changed output_default=%d input_default=%d", output.isEmpty(), input.isEmpty()); - if (outputChanged) { m_output.close(); m_output.retryNow(); } + if (outputChanged) { m_output.close(); m_actualOutput.clear(); m_output.retryNow(); } emit devicesChanged(); recoverAudio(); } @@ -137,6 +331,7 @@ bool SoundManager::createPcmNode(PcmSession& session, bool playing) { return false; } session.sound = sound; + session.detached = false; ma_sound_set_volume(sound, m_volume); if (playing && ma_sound_start(sound) != MA_SUCCESS) { DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "playback: node start failed"); @@ -150,14 +345,11 @@ void SoundManager::shutdown() { if (!m_initialized) return; m_recoveryTimer.stop(); m_initialized = false; + m_resetting = true; emit contextAboutToReset(); - m_output.close(); + m_output.close(); m_voiceRoutes.clear(); m_voices = {}; m_voiceSelections.clear(); m_pcmSessions.clear(); - for (auto it = m_sounds.begin(); it != m_sounds.end(); ++it) { - if (it->sound) { ma_sound_uninit(it->sound); delete it->sound; } - if (it->decoder) { ma_decoder_uninit(it->decoder); delete it->decoder; } - } m_sounds.clear(); DEBUG_DEBUG(DEBUG_CATEGORY_GENERAL, "SoundManager: before ma_engine_uninit engine=%p", (void*)m_engine); @@ -197,47 +389,49 @@ SoundManager::EventSettings SoundManager::eventSettings(const QString& name) con bool SoundManager::loadSound(const QString& name, const QString& filePath) { if (!m_initialized) { - qWarning("SoundManager: not initialized, cannot load %s", qPrintable(name)); + DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "SoundManager: not initialized, cannot load '%s'", qPrintable(name)); return false; } if (m_sounds.contains(name)) { - qWarning("SoundManager: sound '%s' already loaded", qPrintable(name)); + DEBUG_WARN(DEBUG_CATEGORY_MEDIA, "SoundManager: sound '%s' already loaded", qPrintable(name)); return false; } QFile f(filePath); if (!f.open(QIODevice::ReadOnly)) { - qWarning("SoundManager: cannot open %s", qPrintable(filePath)); + DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "SoundManager: cannot open '%s'", qPrintable(filePath)); return false; } - LoadedSound ls; - ls.mp3Data = f.readAll(); + QByteArray encoded = f.readAll(); f.close(); - - ls.decoder = new ma_decoder; - ma_decoder_config decConfig = ma_decoder_config_init_default(); - ma_result result = ma_decoder_init_memory(ls.mp3Data.constData(), ls.mp3Data.size(), - &decConfig, ls.decoder); - if (result != MA_SUCCESS) { - qWarning("SoundManager: ma_decoder_init_memory failed for %s (%d)", qPrintable(name), result); - delete ls.decoder; + /* Декодируем короткие сигналы при загрузке, до playback: hardware callback читает только PCM. */ + ma_decoder_config decConfig = ma_decoder_config_init(ma_format_s16, 2, 48000); + ma_uint64 frames = 0; void* decoded = nullptr; + ma_result result = ma_decode_memory(encoded.constData(), encoded.size(), &decConfig, &frames, &decoded); + if (result != MA_SUCCESS || !frames) { + ma_free(decoded, &decConfig.allocationCallbacks); + DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "SoundManager: decode '%s' failed result=%d frames=%llu", + qPrintable(name), result, (unsigned long long)frames); return false; } - - ls.sound = new ma_sound; - result = ma_sound_init_from_data_source(m_engine, (ma_data_source*)&ls.decoder->ds, - 0, NULL, ls.sound); + LoadedSound ls; + ls.pcm = std::make_shared(); + ls.pcm->frames = frames; ls.pcm->rate = 48000; + ma_audio_buffer_config bufferConfig = ma_audio_buffer_config_init(ma_format_s16, 2, frames, decoded, nullptr); + bufferConfig.sampleRate = 48000; + ls.pcm->buffer = new ma_audio_buffer{}; + result = ma_audio_buffer_init_copy(&bufferConfig, ls.pcm->buffer); + ma_free(decoded, &decConfig.allocationCallbacks); if (result != MA_SUCCESS) { - qWarning("SoundManager: sound init failed for %s (%d)", qPrintable(name), result); - ma_decoder_uninit(ls.decoder); delete ls.decoder; - delete ls.sound; + delete ls.pcm->buffer; ls.pcm->buffer = nullptr; + DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "SoundManager: signal PCM init failed result=%d", result); return false; } - + if (!createPcmNode(*ls.pcm, false)) return false; m_sounds.insert(name, ls); - qDebug("SoundManager: loaded '%s' from %s (%lld bytes)", - qPrintable(name), qPrintable(filePath), (long long)ls.mp3Data.size()); + DEBUG_INFO(DEBUG_CATEGORY_MEDIA, "SoundManager: predecoded '%s' frames=%llu rate=48000 ch=2", + qPrintable(name), (unsigned long long)frames); return true; } @@ -254,12 +448,15 @@ void SoundManager::play(const QString& name) { return; } - ma_sound* s = it->sound; + auto& pcm = *it->pcm; + if (pcm.sound) { ma_sound_uninit(pcm.sound); delete pcm.sound; pcm.sound = nullptr; } + ma_audio_buffer_seek_to_pcm_frame(pcm.buffer, 0); + if (!createPcmNode(pcm, false)) return; + ma_sound* s = pcm.sound; float effectiveVolume = m_volume * es.volume; ma_sound_set_volume(s, effectiveVolume); ma_sound_stop(s); - ma_sound_seek_to_pcm_frame(s, 0); - ma_sound_start(s); + if (ma_sound_start(s) != MA_SUCCESS) DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "SoundManager: signal start failed '%s'", qPrintable(name)); } void SoundManager::playLoop(const QString& name) { @@ -278,12 +475,15 @@ void SoundManager::playLoop(const QString& name) { return; } - ma_sound* s = it->sound; + auto& pcm = *it->pcm; + if (pcm.sound) { ma_sound_uninit(pcm.sound); delete pcm.sound; pcm.sound = nullptr; } + ma_audio_buffer_seek_to_pcm_frame(pcm.buffer, 0); + if (!createPcmNode(pcm, false)) return; + ma_sound* s = pcm.sound; ma_sound_stop(s); - ma_sound_seek_to_pcm_frame(s, 0); ma_sound_set_volume(s, m_volume * es.volume); ma_sound_set_looping(s, MA_TRUE); - ma_sound_start(s); + if (ma_sound_start(s) != MA_SUCCESS) { DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "SoundManager: loop start failed '%s'", qPrintable(name)); return; } m_loopName = name; } @@ -292,7 +492,8 @@ void SoundManager::stopLoop() { auto it = m_sounds.find(m_loopName); m_loopName.clear(); if (it == m_sounds.end()) return; - ma_sound* s = it->sound; + ma_sound* s = it->pcm->sound; + if (!s) return; ma_sound_stop(s); ma_sound_set_looping(s, MA_FALSE); ma_sound_seek_to_pcm_frame(s, 0); @@ -336,13 +537,26 @@ SoundManager::PlaybackId SoundManager::playRawPcm(const int16_t* pcm, int frameC void SoundManager::pausePcm(PlaybackId id) { auto* session = pcmSession(id); - if (session && session->sound) ma_sound_stop(session->sound); + if (!session || !session->sound) return; + ma_result result = ma_sound_stop(session->sound); + /* Stop меняет флаг; detach также ждёт уже начатое чтение и фиксирует cursor. */ + if (result == MA_SUCCESS && !session->detached) { + result = ma_node_detach_output_bus(session->sound, 0); + if (result == MA_SUCCESS) session->detached = true; + } + if (result != MA_SUCCESS) DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "playback: pause failed id=%llu result=%d", (unsigned long long)id, result); } void SoundManager::resumePcm(PlaybackId id) { auto* session = pcmSession(id); - if (session && session->sound && ma_sound_start(session->sound) != MA_SUCCESS) - DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "playback: resume failed id=%llu", (unsigned long long)id); + if (!session || !session->sound) return; + ma_result result = MA_SUCCESS; + if (session->detached) { + result = ma_node_attach_output_bus(session->sound, 0, ma_engine_get_endpoint(m_engine), 0); + if (result == MA_SUCCESS) session->detached = false; + } + if (result == MA_SUCCESS) result = ma_sound_start(session->sound); + if (result != MA_SUCCESS) DEBUG_ERROR(DEBUG_CATEGORY_MEDIA, "playback: resume failed id=%llu result=%d", (unsigned long long)id, result); } void SoundManager::stopPcm(PlaybackId id) { diff --git a/tools/chatgui/src/sound_manager.h b/tools/chatgui/src/sound_manager.h index 9536d241..6bf162ac 100644 --- a/tools/chatgui/src/sound_manager.h +++ b/tools/chatgui/src/sound_manager.h @@ -7,6 +7,7 @@ #include #include #include +#include #include #include #include "audio_device.h" @@ -15,6 +16,7 @@ struct ma_engine; struct ma_context; struct ma_decoder; struct ma_sound; +class VoiceAudioIo; class SoundManager : public QObject { Q_OBJECT @@ -48,6 +50,10 @@ public: QString playbackDevice() const { return m_playbackId; } QString captureDevice() const { return m_captureId; } + /* GUI: voice живёт до detachVoice; изменение списка выполняется при остановленном output. */ + bool attachVoice(VoiceAudioIo* voice, const QString& output); + void detachVoice(VoiceAudioIo* voice); + using PlaybackId = uint64_t; /* GUI: независимая сессия. owner удерживает PCM до detach; без owner данные копируются. */ PlaybackId playRawPcm(const int16_t* pcm, int frameCount, int sampleRate, int channels, @@ -84,6 +90,7 @@ private: uint64_t frames = 0, origin = 0; int rate = 0; bool transient = false; + bool detached = false; ~PcmSession(); }; PcmSession* pcmSession(PlaybackId id) const; @@ -92,10 +99,26 @@ private: bool openContext(); void openOutput(); + struct VoiceRoute { + AudioDevice device; + QString wanted, actual; + std::array voices{}; + }; + QString resolveOutput(const QString& requested); + bool openVoiceRoute(VoiceRoute& route); + void mergeVoiceRoutes(); + static unsigned renderOutput(ma_engine* engine, const std::array& voices, + float* out, unsigned frames, int64_t firstTimeUs); + std::atomic m_renderErrors{0}; + std::array m_voices{}; + std::map> m_voiceRoutes; + std::map m_voiceSelections; // Desired выбор сохраняется при объединении physical outputs. + QString m_actualOutput; + bool m_resetting = false; + bool m_restoreSelections = false; + struct LoadedSound { - QByteArray mp3Data; - ma_decoder* decoder = nullptr; - ma_sound* sound = nullptr; + std::shared_ptr pcm; }; ma_engine* m_engine = nullptr; diff --git a/tools/chatgui/src/voice_audio_io.cpp b/tools/chatgui/src/voice_audio_io.cpp new file mode 100644 index 00000000..49234866 --- /dev/null +++ b/tools/chatgui/src/voice_audio_io.cpp @@ -0,0 +1,331 @@ +#include "voice_audio_io.h" +extern "C" { +#include "call/call_audio.h" +#include "radio/radio_audio.h" +#include "speex_aec.h" +#include "debug_config.h" +} +#include +#include +#include +#include +#include +#include + +static int64_t voice_time_us() { + return std::chrono::duration_cast(std::chrono::steady_clock::now().time_since_epoch()).count(); +} + +VoiceAudioIo::~VoiceAudioIo() { stop(); } + +bool VoiceAudioIo::start(Kind kind, uint64_t session) { + stop(); + m_kind = kind; m_session = session; + m_captureEvents.reset(); m_references.reset(); m_playback.reset(); + m_playPosition = 960; m_playEpoch = 1; + m_captureDropped = m_referenceDropped = m_playbackUnderruns = 0; + m_running = true; + try { m_playbackThread = std::thread(&VoiceAudioIo::playbackLoop, this); } + catch (const std::system_error& error) { + m_running = false; + DEBUG_ERROR(DEBUG_CATEGORY_CALL, "voice-io: RX worker start failed reason=%s", error.what()); return false; + } + DEBUG_INFO(DEBUG_CATEGORY_CALL, "voice-io: started session=%llu kind=%s PCM reserve<=40ms", + (unsigned long long)session, kind == Kind::Call ? "call" : "radio"); + return true; +} + +void VoiceAudioIo::stop() { + stopCapture(); + m_running = false; m_wake.notify_all(); + if (m_playbackThread.joinable()) m_playbackThread.join(); +} + +bool VoiceAudioIo::startCapture(int channels, bool aecEnabled) { + stopCapture(); + if (!m_running || channels < 1 || channels > 2 || (m_kind == Kind::Call && channels != 1)) { + DEBUG_ERROR(DEBUG_CATEGORY_CALL, "voice-io: invalid capture channels=%d", channels); return false; + } + m_captureEvents.reset(); m_channels = channels; m_captureSequence = 0; + m_aecRequested = aecEnabled; m_capturing = true; + try { m_captureThread = std::thread(&VoiceAudioIo::captureLoop, this); } + catch (const std::system_error& error) { + m_capturing = false; + DEBUG_ERROR(DEBUG_CATEGORY_CALL, "voice-io: capture worker start failed reason=%s", error.what()); return false; + } + return true; +} + +void VoiceAudioIo::stopCapture() { + if (m_captureThread.joinable()) { + if (m_capturing) command(EventKind::Stop); + m_captureThread.join(); + } + m_capturing = false; +} + +/* GUI-команда не теряется при заполнении PCM: зарезервированы слоты; отказ завершает capture. */ +void VoiceAudioIo::command(EventKind kind, bool value) { + if (!m_capturing) return; + Event event{}; event.kind = kind; event.value = value; + for (int attempt = 0; attempt < 100; ++attempt) { + if (m_captureEvents.push(event, true)) { m_wake.notify_all(); return; } + m_wake.notify_all(); std::this_thread::sleep_for(std::chrono::milliseconds(1)); + } + DEBUG_ERROR(DEBUG_CATEGORY_CALL, "voice-io: control queue full session=%llu command=%d; stopping capture", + (unsigned long long)m_session, (int)kind); + m_capturing = false; m_wake.notify_all(); +} +void VoiceAudioIo::setPtt(bool pressed) { command(EventKind::Ptt, pressed); } +void VoiceAudioIo::setMuted(bool muted) { command(EventKind::Mute, muted); } +void VoiceAudioIo::setAecEnabled(bool enabled) { + if (enabled == m_aecRequested) return; + m_aecRequested = enabled; command(EventKind::Aec, enabled); +} +void VoiceAudioIo::resetPlayback() { ++m_playEpoch; m_wake.notify_all(); } + +void VoiceAudioIo::capture(const int16_t* pcm, unsigned frames, int64_t firstTimeUs, uint64_t route, bool missing) { + if (!pcm || !m_capturing) return; + if (missing) { m_captureSequence += frames; m_captureDropped.fetch_add(frames); return; } + int64_t enqueued = voice_time_us(); + if (!firstTimeUs) firstTimeUs = enqueued - (int64_t)frames * 1000000 / 48000; + unsigned offset = 0; + while (offset < frames) { + Event event{}; event.frames = std::min(frames - offset, 960u); + event.time = firstTimeUs + (int64_t)offset * 1000000 / 48000; + event.sequence = m_captureSequence; m_captureSequence += event.frames; + event.route = route; + event.enqueued = enqueued; + memcpy(event.pcm, pcm + offset * m_channels, event.frames * m_channels * sizeof(int16_t)); + if (!m_captureEvents.push(event)) m_captureDropped.fetch_add(event.frames, std::memory_order_relaxed); + offset += event.frames; + } + m_wake.notify_all(); +} + +void VoiceAudioIo::reference(const float* pcm, unsigned frames, int64_t firstTimeUs) { + if (!m_capturing) return; + if (!firstTimeUs) firstTimeUs = voice_time_us(); + for (unsigned offset = 0; offset < frames;) { + Event event{}; event.frames = std::min(frames - offset, 960u); + event.time = firstTimeUs + (int64_t)offset * 1000000 / 48000; + event.sequence = m_playEpoch.load(); + for (unsigned i = 0; i < event.frames * 2; ++i) { + float sample = pcm[offset * 2 + i]; + event.pcm[i] = (int16_t)std::clamp(sample * 32768.0f, -32768.0f, 32767.0f); + } + if (!m_references.push(event)) m_referenceDropped.fetch_add(event.frames, std::memory_order_relaxed); + offset += event.frames; + } + m_wake.notify_all(); +} + +/* Выход всегда обслуживается независимо от capture/VAD. Уже потреблённый PCM становится референсом у владельца микшера. */ +void VoiceAudioIo::mix(float* stereo, unsigned frames) { + uint64_t epoch = m_playEpoch.load(); + for (unsigned i = 0; i < frames; ++i) { + if (m_playCurrent.epoch != epoch) m_playPosition = 960; + if (m_playPosition == 960) { + bool available = false; + for (int attempt = 0; attempt < 12 && m_playback.pop(m_playCurrent); ++attempt) { + if (m_playCurrent.epoch == epoch) { available = true; m_playPosition = 0; break; } + } + if (!available) { m_playbackUnderruns.fetch_add(frames - i, std::memory_order_relaxed); break; } + } + stereo[2*i] += m_playCurrent.pcm[2*m_playPosition] / 32768.0f; + stereo[2*i+1] += m_playCurrent.pcm[2*m_playPosition+1] / 32768.0f; + ++m_playPosition; + } + m_wake.notify_all(); +} + +void VoiceAudioIo::playbackLoop() { + uint64_t epoch = m_playEpoch.load(); + while (m_running) { + uint64_t requested = m_playEpoch.load(); + if (requested != epoch) { + if (m_kind == Kind::Call) call_audio_reset_io(m_session, 1); + else radio_audio_reset_playback(m_session); + epoch = requested; + } + if (m_playback.size() < 2) { + Playback block{}; block.epoch = epoch; + if (m_kind == Kind::Call) { + int16_t mono[960]{}; + int count = std::clamp(call_audio_pull_pcm(m_session, mono, 960), 0, 960); + for (int i = 0; i < count; ++i) block.pcm[2*i] = block.pcm[2*i+1] = mono[i]; + } else radio_audio_pull_pcm(m_session, block.pcm, 1920); + if (!m_playback.push(block)) DEBUG_WARN(DEBUG_CATEGORY_CALL, "voice-io: playback queue contention"); + } else { + std::unique_lock lock(m_waitMutex); + m_wake.wait_for(lock, std::chrono::milliseconds(5)); + } + } +} + +/* Референс хранится как PCM + время, а не как фиксированный двухкадровый FIFO независимых часов. */ +void VoiceAudioIo::captureLoop() { + std::array history{}; + size_t historyHead = 0, historyCount = 0; + uint64_t referenceEpoch = m_playEpoch.load(), referenceDrops = m_referenceDropped.load(); + speex_aec_t* aec = nullptr; + auto configureAec = [&](bool enabled) { + speex_aec_destroy(aec); aec = nullptr; + if (enabled) aec = speex_aec_create_mc(48000, 960, 14400, 1, m_channels, 2); + if (enabled && !aec) DEBUG_ERROR(DEBUG_CATEGORY_AEC, "voice-io: AEC unavailable session=%llu", (unsigned long long)m_session); + DEBUG_INFO(DEBUG_CATEGORY_AEC, "voice-io: capture AEC=%d mic=%d render=2 session=%llu", + aec != nullptr, m_channels, (unsigned long long)m_session); + }; + configureAec(m_aecRequested); + struct Marker { unsigned position; EventKind kind; bool value; }; + std::array markers{}; + size_t markerCount = 0; + int16_t raw[1920]{}, cleaned[1920]{}, referencePcm[1920]{}; + uint8_t muteMask[960]{}; + unsigned filled = 0; + int64_t frameTime = 0, lastReport = voice_time_us(); + uint64_t expectedSequence = 0, missingReference = 0, reportedReferenceDrops = referenceDrops, captureRoute = 0; + int64_t expectedCaptureTime = 0; + bool referenceAvailable = false; + bool muted = false; + auto applyMarker = [&](EventKind kind, bool value) { + if (kind == EventKind::Mute) muted = value; + else if (kind == EventKind::Ptt && m_kind == Kind::Radio) { + if (value) radio_audio_talk_begin(m_session); else radio_audio_talk_end(m_session); + } + }; + auto consumeReference = [&]() { + uint64_t epoch = m_playEpoch.load(), drops = m_referenceDropped.load(); + if (epoch != referenceEpoch || drops != referenceDrops) { + DEBUG_DEBUG(DEBUG_CATEGORY_AEC, "voice-io: reference reset epoch=%llu->%llu dropped=%llu", + (unsigned long long)referenceEpoch, (unsigned long long)epoch, (unsigned long long)(drops - referenceDrops)); + historyHead = historyCount = 0; + if (aec) speex_aec_reset(aec); + referenceEpoch = epoch; referenceDrops = drops; + } + Event event{}; + for (unsigned drained = 0; drained < 32 && m_references.pop(event); ++drained) { + if (event.sequence != referenceEpoch) continue; + if (historyCount && event.time <= history[(historyHead + historyCount - 1) % history.size()].time) { + DEBUG_WARN(DEBUG_CATEGORY_AEC, "voice-io: non-monotonic render clock, resetting AEC"); + historyHead = historyCount = 0; + if (aec) speex_aec_reset(aec); + } + if (historyCount == history.size()) { historyHead = (historyHead + 1) % history.size(); --historyCount; } + history[(historyHead + historyCount++) % history.size()] = event; + } + }; + auto alignReference = [&]() { + if (!historyCount) return false; + size_t index = 0; + for (unsigned i = 0; i < 960; ++i) { + double time = frameTime + (double)i * 1000000.0 / 48000.0; + while (index + 1 < historyCount && history[(historyHead + index + 1) % history.size()].time <= time) ++index; + auto& block = history[(historyHead + index) % history.size()]; + double interval = 1000000.0 / 48000.0; + if (index + 1 < historyCount) { + auto& next = history[(historyHead + index + 1) % history.size()]; + double measured = (double)(next.time - block.time) / block.frames; + if (measured > interval * 0.95 && measured < interval * 1.05) interval = measured; + } + double position = (time - block.time) / interval; + if (position < 0 || position >= block.frames) return false; + unsigned sample = (unsigned)position; + double fraction = position - sample; + for (unsigned channel = 0; channel < 2; ++channel) { + int first = block.pcm[2*sample+channel]; + int second = sample + 1 < block.frames ? block.pcm[2*(sample+1)+channel] : first; + referencePcm[2*i+channel] = (int16_t)(first + (second - first) * fraction); + } + } + return true; + }; + auto deliver = [&](unsigned frames) { + unsigned begin = 0; + for (size_t mark = 0; mark <= markerCount; ++mark) { + unsigned end = mark < markerCount ? markers[mark].position : frames; + if (m_kind == Kind::Call) memset(muteMask + begin, muted, end - begin); + else if (end > begin) radio_audio_feed_pcm(m_session, cleaned + begin * m_channels, (end - begin) * m_channels); + if (mark < markerCount) applyMarker(markers[mark].kind, markers[mark].value); + begin = end; + } + if (m_kind == Kind::Call && frames == 960) call_audio_feed_prepared_pcm(m_session, cleaned, muteMask); + filled = 0; markerCount = 0; + }; + bool stopping = false; + while (m_capturing && !stopping) { + consumeReference(); + Event event{}; + if (!m_captureEvents.pop(event)) { + std::unique_lock lock(m_waitMutex); m_wake.wait_for(lock, std::chrono::milliseconds(5)); continue; + } + if (event.kind == EventKind::Stop) { stopping = true; break; } + if (event.kind == EventKind::Aec) { configureAec(event.value); continue; } + if (event.kind != EventKind::Pcm) { + if (!filled) applyMarker(event.kind, event.value); + else if (markerCount < markers.size()) markers[markerCount++] = {filled, event.kind, event.value}; + else { DEBUG_ERROR(DEBUG_CATEGORY_CALL, "voice-io: too many controls within one capture frame"); stopping = true; } + continue; + } + if (voice_time_us() - event.enqueued > 100000) { + m_captureDropped.fetch_add(event.frames); continue; + } + bool clockJump = expectedCaptureTime && std::abs(event.time - expectedCaptureTime) > 20000; + if (event.sequence != expectedSequence || (captureRoute && event.route != captureRoute) || clockJump) { + DEBUG_WARN(DEBUG_CATEGORY_CALL, "voice-io: capture discontinuity expected=%llu actual=%llu pending=%u route=%016llx->%016llx", + (unsigned long long)expectedSequence, (unsigned long long)event.sequence, filled, + (unsigned long long)captureRoute, (unsigned long long)event.route); + if (clockJump) DEBUG_WARN(DEBUG_CATEGORY_AEC, "voice-io: capture clock jump=%lldus", (long long)(event.time - expectedCaptureTime)); + for (size_t i = 0; i < markerCount; ++i) applyMarker(markers[i].kind, markers[i].value); + filled = 0; markerCount = 0; + if (aec) speex_aec_reset(aec); + if (m_kind == Kind::Radio) radio_audio_capture_discontinuity(m_session); + } + expectedSequence = event.sequence + event.frames; + captureRoute = event.route; + expectedCaptureTime = event.time + (int64_t)event.frames * 1000000 / 48000; + for (unsigned offset = 0; offset < event.frames;) { + if (!filled) frameTime = event.time + (int64_t)offset * 1000000 / 48000; + unsigned take = std::min(960 - filled, event.frames - offset); + memcpy(raw + filled * m_channels, event.pcm + offset * m_channels, take * m_channels * sizeof(int16_t)); + filled += take; offset += take; + if (filled != 960) continue; + memcpy(cleaned, raw, 960 * m_channels * sizeof(int16_t)); + consumeReference(); + bool available = aec && alignReference(); + if (available) { + speex_aec_feed_playback(aec, referencePcm, 1920); + speex_aec_process_capture(aec, raw, 960 * m_channels, cleaned); + } else if (aec) { + ++missingReference; + if (referenceAvailable) speex_aec_reset(aec); + } + referenceAvailable = available; + deliver(960); + } + int64_t now = voice_time_us(); + if (now - lastReport >= 1000000) { + uint64_t captureDrops = m_captureDropped.exchange(0), refDrops = m_referenceDropped.load(); + uint64_t underruns = m_playbackUnderruns.exchange(0); + if (captureDrops || refDrops != reportedReferenceDrops || missingReference || underruns) + DEBUG_WARN(DEBUG_CATEGORY_AEC, "voice-io: session=%llu capture_drop=%llu reference_drop=%llu missing_ref=%llu output_missing=%llu", + (unsigned long long)m_session, (unsigned long long)captureDrops, (unsigned long long)refDrops, + (unsigned long long)missingReference, (unsigned long long)underruns); + else DEBUG_DEBUG(DEBUG_CATEGORY_AEC, "voice-io: session=%llu aligned history=%zu capture_queue=%zu", + (unsigned long long)m_session, historyCount, m_captureEvents.size()); + missingReference = 0; reportedReferenceDrops = refDrops; lastReport = now; + } + } + if (filled) { + DEBUG_WARN(DEBUG_CATEGORY_CALL, "voice-io: capture tail=%u frames bypasses AEC; call tail discarded, radio tail preserved", filled); + memcpy(cleaned, raw, filled * m_channels * sizeof(int16_t)); + deliver(filled); + } + uint64_t lostCapture = m_captureDropped.exchange(0), lostReference = m_referenceDropped.load() - reportedReferenceDrops; + if (lostCapture || lostReference || missingReference) + DEBUG_WARN(DEBUG_CATEGORY_AEC, "voice-io: capture stopped lost_capture=%llu lost_reference=%llu missing_reference=%llu", + (unsigned long long)lostCapture, (unsigned long long)lostReference, (unsigned long long)missingReference); + + speex_aec_destroy(aec); + m_capturing = false; +} diff --git a/tools/chatgui/src/voice_audio_io.h b/tools/chatgui/src/voice_audio_io.h new file mode 100644 index 00000000..6d4654f7 --- /dev/null +++ b/tools/chatgui/src/voice_audio_io.h @@ -0,0 +1,61 @@ +#pragma once +#include "audio_event_queue.h" +#include +#include +#include +#include +#include + +/* Desktop: аппаратные callbacks только копируют PCM, render также выполняет готовое смешивание. + * Capture worker — единственный владелец AEC и порядка PCM/PTT/mute. RX worker независим от VAD. + * Управляющий поток закрывает capture и отсоединяет output до stop/destruction. + * Финальный stereo render приходит с того же выходного устройства после всего микширования. */ +class VoiceAudioIo { +public: + enum class Kind { Call, Radio }; + VoiceAudioIo() = default; + ~VoiceAudioIo(); + bool start(Kind kind, uint64_t session); + void stop(); + bool startCapture(int channels, bool aecEnabled); + void stopCapture(); + bool captureRunning() const { return m_capturing.load(); } + void setPtt(bool pressed); + void setMuted(bool muted); + void setAecEnabled(bool enabled); + void resetPlayback(); + /* firstTimeUs — монотонное время первого физического PCM-фрейма (backend либо оценка). */ + void capture(const int16_t* pcm, unsigned frames, int64_t firstTimeUs, uint64_t route = 0, bool missing = false); + void reference(const float* pcm, unsigned frames, int64_t firstTimeUs); + void mix(float* stereo, unsigned frames); +private: + enum class EventKind { Pcm, Ptt, Mute, Aec, Stop }; + struct Event { + EventKind kind = EventKind::Pcm; + unsigned frames = 0; + int64_t time = 0; + int64_t enqueued = 0; + uint64_t sequence = 0; + uint64_t route = 0; + bool value = false; + int16_t pcm[1920]{}; + }; + struct Playback { uint64_t epoch = 0; int16_t pcm[1920]{}; }; + AudioEventQueue m_captureEvents, m_references; + AudioEventQueue m_playback; + std::atomic m_running{false}, m_capturing{false}; + std::atomic m_playEpoch{0}, m_captureDropped{0}, m_referenceDropped{0}, m_playbackUnderruns{0}; + uint64_t m_captureSequence = 0; // Только capture callback; reset после его остановки. + Kind m_kind = Kind::Call; + uint64_t m_session = 0; + int m_channels = 1; + std::atomic m_aecRequested{false}; + std::thread m_captureThread, m_playbackThread; + std::mutex m_waitMutex; + std::condition_variable m_wake; + Playback m_playCurrent{}; + unsigned m_playPosition = 960; + void command(EventKind kind, bool value = false); + void captureLoop(); + void playbackLoop(); +}; diff --git a/tools/chatgui/tests/run_pulse_recovery.py b/tools/chatgui/tests/run_pulse_recovery.py index 7de808ef..4707213e 100644 --- a/tools/chatgui/tests/run_pulse_recovery.py +++ b/tools/chatgui/tests/run_pulse_recovery.py @@ -1,6 +1,7 @@ #!/usr/bin/env python3 """Изолированный PipeWire/PulseAudio с виртуальным sink; системный сервер не затрагивается.""" import os +import json from pathlib import Path import shutil import signal @@ -54,14 +55,35 @@ def main(): return subprocess.run(["pactl", "info"], env=env, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, timeout=2).returncode == 0 + def pactl(*args): + return subprocess.check_output(["pactl", *args], env=env, text=True, timeout=5) + + def create_sinks(): + for name in ("utun_test_main", "utun_test_voice"): + pactl("load-module", "module-null-sink", f"sink_name={name}") + pactl("set-default-sink", "utun_test_main") + pactl("set-default-source", "utun_test_main.monitor") + try: launch(["pipewire"], "pipewire") wait_for(lambda: (directory / "pipewire-0").exists()) launch(["dbus-run-session", "--", "wireplumber"], "wireplumber") pulse = launch(["pipewire-pulse"], "pulse-before") wait_for(pulse_ready) - # module-always-sink создаёт виртуальный dummy sink и его monitor source. + create_sinks() test = launch([str(binary), "--pulse-integration", root], "client") + wait_for(lambda: (directory / "routes_ready").exists() or test.poll() is not None) + if test.poll() is not None: + raise RuntimeError("client failed before route move") + sinks = json.loads(pactl("--format=json", "list", "sinks")) + main_sink = next(sink["index"] for sink in sinks if sink["name"] == "utun_test_main") + streams = json.loads(pactl("--format=json", "list", "sink-inputs")) + main_streams = [stream for stream in streams if stream["sink"] == main_sink] + if len(main_streams) != 1: + raise RuntimeError("expected one shared default stream") + # Явно выбранный stream имеет DONT_MOVE; default stream разрешено перемещать сервером. + pactl("move-sink-input", str(main_streams[0]["index"]), "utun_test_voice") + (directory / "moved").touch() wait_for(lambda: (directory / "ready").exists() or test.poll() is not None) if test.poll() is not None: raise RuntimeError("client failed before server restart") @@ -69,6 +91,7 @@ def main(): wait_for(lambda: (directory / "closed").exists() or test.poll() is not None) launch(["pipewire-pulse"], "pulse-after") wait_for(pulse_ready) + create_sinks() if test.wait(timeout=15) != 0: raise RuntimeError("client recovery failed") logs[3].flush(); logs[3].seek(0) diff --git a/tools/chatgui/tests/test_audio_event_queue.cpp b/tools/chatgui/tests/test_audio_event_queue.cpp new file mode 100644 index 00000000..fb3d74a5 --- /dev/null +++ b/tools/chatgui/tests/test_audio_event_queue.cpp @@ -0,0 +1,50 @@ +/* MPSC: резерв управляющих сообщений, публикация слота и порядок каждого producer. */ +#include "../src/audio_event_queue.h" +#include +#include +#include +#include +#define REQUIRE(x) do { if (!(x)) { fprintf(stderr, "queue:%d %s\n", __LINE__, #x); abort(); } } while (0) + +static std::atomic entered{false}, releaseCopy{false}; +struct Item { + int producer = 0, sequence = 0; + bool block = false; + Item& operator=(const Item& other) { + if (other.block) { entered = true; while (!releaseCopy) std::this_thread::yield(); } + producer = other.producer; sequence = other.sequence; block = other.block; return *this; + } +}; +int main() { + AudioEventQueue queue; + Item item{}; + for (int i = 0; i < 8; ++i) REQUIRE(queue.push(item)); + REQUIRE(!queue.push(item)); + for (int i = 0; i < 8; ++i) REQUIRE(queue.push(item, true)); + REQUIRE(!queue.push(item, true)); + for (int i = 0; i < 16; ++i) REQUIRE(queue.pop(item)); + REQUIRE(!queue.pop(item)); + std::thread first([&] { REQUIRE(queue.push({0, 123, true})); }); + while (!entered) std::this_thread::yield(); + REQUIRE(queue.push({1, 456, false}, true)); + REQUIRE(!queue.pop(item)); // Второй слот опубликован, первый ещё принадлежит producer. + releaseCopy = true; first.join(); + REQUIRE(queue.pop(item) && item.sequence == 123); + REQUIRE(queue.pop(item) && item.sequence == 456); + queue.reset(); + constexpr int producers = 4, count = 20000; + std::vector threads; + for (int p = 0; p < producers; ++p) threads.emplace_back([&, p] { + for (int i = 0; i < count; ++i) while (!queue.push({p, i, false}, i % 7 == 0)) std::this_thread::yield(); + }); + int expected[producers]{}; + for (int total = 0; total < producers * count;) { + if (!queue.pop(item)) { std::this_thread::yield(); continue; } + REQUIRE(queue.size() <= 16); + REQUIRE(item.producer >= 0 && item.producer < producers); + REQUIRE(item.sequence == expected[item.producer]++); ++total; + } + for (auto& thread : threads) thread.join(); + REQUIRE(queue.size() == 0); + puts("audio queue: control reserve, unpublished slot, 80000 ordered messages PASS"); +} diff --git a/tools/chatgui/tests/test_audio_recovery.cpp b/tools/chatgui/tests/test_audio_recovery.cpp index bb585640..90ffff07 100644 --- a/tools/chatgui/tests/test_audio_recovery.cpp +++ b/tools/chatgui/tests/test_audio_recovery.cpp @@ -82,11 +82,40 @@ static void pump(int milliseconds) { static void pulse_recovery(SoundManager* manager, const QString& control) { REQUIRE(manager->context()->backend == ma_backend_pulseaudio); + instance.ua = uasync_create(); REQUIRE(instance.ua); + REQUIRE(call_audio_init(&instance) == 0 && radio_audio_init(&instance) == 0); + auto* call = CallAudioEngine::instance(); auto* radio = RadioAudioEngine::instance(); + REQUIRE(call->start(42) == 0 && radio->start(7) == 0); + QString separate; + for (const auto& output : manager->enumeratePlaybackDevices()) { + ma_device_id id{}; + if (manager->deviceId(output.id, &id) && !strcmp(id.pulse, "utun_test_voice")) separate = output.id; + } + REQUIRE(!separate.isEmpty()); call->setPlaybackDevice(separate); AudioRecorder recorder; recorder.init(); recorder.startRecording(); std::vector pcm(48000 * 20, 1000); auto pcmId = manager->playRawPcm(pcm.data(), (int)pcm.size(), 48000, 1); for (int i = 0; i < 100 && recorder.durationMs() < 300; ++i) pump(50); REQUIRE(recorder.durationMs() >= 300 && manager->pcmCursor(pcmId) > 0); + QFile routesReady(control + "/routes_ready"); REQUIRE(routesReady.open(QIODevice::WriteOnly)); routesReady.close(); + for (int i = 0; i < 100 && !QFile::exists(control + "/moved"); ++i) pump(50); + REQUIRE(QFile::exists(control + "/moved")); + pump(500); + int outputs = 0; + for (auto* device : devices) if (device->type == ma_device_type_playback) ++outputs; + REQUIRE(outputs == 1); // Сервер переместил default stream на voice sink: получился один микшер. + qInfo("PulseAudio route move: call/radio/media merged into one output PASS"); + for (auto* device : devices) if (device->type == ma_device_type_playback) REQUIRE(ma_device_stop(device) == MA_SUCCESS); + pump(1000); outputs = 0; + uint64_t wantedRoute = UINT64_C(14695981039346656037); + for (const char* name = "utun_test_voice"; *name; ++name) { wantedRoute ^= (unsigned char)*name; wantedRoute *= UINT64_C(1099511628211); } + bool foundWanted = false; + for (auto* device : devices) if (device->type == ma_device_type_playback) { + ++outputs; + if (static_cast(device->pUserData)->route() == wantedRoute) foundWanted = true; + } + REQUIRE(foundWanted && outputs >= 1 && outputs <= 2); // Сервер может восстановить прежний move для default stream. + qInfo("PulseAudio output recovery: desired call route preserved outputs=%d PASS", outputs); int duration = recorder.durationMs(); auto cursor = manager->pcmCursor(pcmId); QFile ready(control + "/ready"); REQUIRE(ready.open(QIODevice::WriteOnly)); ready.close(); @@ -95,9 +124,12 @@ static void pulse_recovery(SoundManager* manager, const QString& control) { QFile closed(control + "/closed"); REQUIRE(closed.open(QIODevice::WriteOnly)); closed.close(); for (int i = 0; i < 150 && (!manager->context() || recorder.durationMs() <= duration + 200); ++i) pump(50); REQUIRE(manager->context() && recorder.durationMs() > duration + 200 && manager->pcmCursor(pcmId) > cursor); + REQUIRE(call->isActive() && call->callId() == 42 && radio->isActive() && radio->groupId() == 7); qInfo("PulseAudio progress: capture=%d -> %dms playback=%llu -> %llu frames", duration, recorder.durationMs(), (unsigned long long)cursor, (unsigned long long)manager->pcmCursor(pcmId)); - recorder.stopRecording(); manager->shutdown(); + call->stop(); radio->stop(); recorder.stopRecording(); + call_audio_destroy(&instance); radio_audio_destroy(&instance); uasync_destroy(instance.ua, 0); instance.ua = nullptr; + manager->shutdown(); REQUIRE(devices.empty()); qInfo("real PulseAudio: capture/playback progress restored after server restart PASS"); } @@ -144,6 +176,12 @@ int main(int argc, char** argv) { REQUIRE(a.pcm[0] == 1000 && b.pcm[0] == 1000); // pUserData маршрутизирует каждый callback своему recorder. a.pcm_release(a.pcm_owner); b.pcm_release(b.pcm_owner); } + { + AudioRecorder meter; meter.init(); meter.startRecording(true); pump(60); meter.stopRecording(); + attachment_send_req request{}; + REQUIRE(meter.durationMs() > 0 && !meter.takeRecording(&request)); + REQUIRE(!request.pcm && !request.pcm_owner && !request.compressor); + } instance.ua = uasync_create(); REQUIRE(instance.ua); REQUIRE(call_audio_init(&instance) == 0 && radio_audio_init(&instance) == 0); auto* call = CallAudioEngine::instance(); @@ -151,7 +189,7 @@ int main(int argc, char** argv) { pump(60); ma_device* playback = nullptr; for (auto* device : devices) - if (device->type == ma_device_type_playback && device->playback.format == ma_format_s16 && device->playback.channels == 1) + if (device->type == ma_device_type_playback && device->playback.format == ma_format_f32 && device->playback.channels == 2) playback = device; REQUIRE(playback); int previousCapture = captureStarts, previousPlayback = playbackStarts; @@ -203,10 +241,22 @@ int main(int argc, char** argv) { manager->pausePcm(player24); cursor24 = manager->pcmCursor(player24); pump(80); - REQUIRE(manager->pcmCursor(player24) == cursor24 && manager->pcmCursor(player48) > cursor48); + uint64_t pausedAfter = manager->pcmCursor(player24), neighbourAfter = manager->pcmCursor(player48); + qInfo("PCM pause: paused=%llu->%llu neighbour=%llu->%llu", (unsigned long long)cursor24, (unsigned long long)pausedAfter, + (unsigned long long)cursor48, (unsigned long long)neighbourAfter); + REQUIRE(pausedAfter == cursor24 && neighbourAfter > cursor48); REQUIRE(manager->seekPcm(player24, 12000)); REQUIRE(manager->pcmCursor(player24) == 12000 && !manager->isPcmPlaying(player24)); manager->resumePcm(player24); + // Пауза синхронизируется с уже начатым чтением графа, сосед продолжает играть. + for (int i = 0; i < 50; ++i) { + pump(1); + manager->pausePcm(player24); + uint64_t paused = manager->pcmCursor(player24); + pump(1); + REQUIRE(manager->pcmCursor(player24) == paused); + manager->resumePcm(player24); + } manager->stopPcm(player48); pump(60); REQUIRE(manager->pcmCursor(player24) > 12000 && manager->pcmCursor(player48) == 0); diff --git a/tools/chatgui/tests/test_audio_timing.c b/tools/chatgui/tests/test_audio_timing.c new file mode 100644 index 00000000..f6adf0e5 --- /dev/null +++ b/tools/chatgui/tests/test_audio_timing.c @@ -0,0 +1,57 @@ +/* PA snapshot без сервера: несколько converter callbacks принадлежат одному physical PCM-блоку. */ +#define MA_DATA_CONVERTER_STACK_BUFFER_SIZE 128 +#include "../transport/miniaudio_impl.c" +#include +#include +static int64_t times[64]; +static unsigned counts[64], seen; +static int capture_missing; +static int fake_latency(const void* stream, uint64_t* latency, int* negative) { + (void)stream; *latency = 100000; *negative = 0; return 0; +} +static void observe(ma_device* device, void* output, const void* input, ma_uint32 frames) { + (void)output; (void)input; + int measured = 0; + assert(seen < 64); + times[seen] = audio_stream_frame_time(device, frames, &measured); + counts[seen++] = frames; + assert(measured && seen < 64); + assert(audio_stream_capture_missing(device) == capture_missing); +} +static int64_t time_us(void) { + struct timespec now; clock_gettime(CLOCK_MONOTONIC, &now); + return now.tv_sec * INT64_C(1000000) + now.tv_nsec / 1000; +} +int main(void) { + /* Только dispatch/converter: backend union не подменяется у живого null-device. */ + ma_context context = {0}; ma_device device = {0}; + device.pContext = &context; device.type = ma_device_type_capture; + device.sampleRate = 48000; device.noFixedSizedCallback = MA_TRUE; device.onData = observe; + device.capture.format = ma_format_s16; device.capture.channels = 1; + device.capture.internalFormat = ma_format_f32; device.capture.internalChannels = 2; + assert(ma_device_set_master_volume(&device, 1) == MA_SUCCESS); + ma_data_converter_config converter = ma_data_converter_config_init(ma_format_f32, ma_format_s16, 2, 1, 48000, 48000); + assert(ma_data_converter_init(&converter, NULL, &device.capture.converter) == MA_SUCCESS); + device.capture.internalSampleRate = 48000; + context.backend = ma_backend_pulseaudio; + context.pulse.pa_stream_get_latency = (ma_proc)fake_latency; + device.pulse.pStreamCapture = (void*)1; + float input[180 * 2] = {0}; + int64_t now = time_us(); + assert(ma_device_handle_backend_data_callback(&device, NULL, input, 180) == MA_SUCCESS); + int64_t after = time_us(); + assert(seen == 3 && counts[0] == 64 && counts[1] == 64 && counts[2] == 52); + assert(times[0] >= now - 100000 && times[0] <= after - 100000); + for (unsigned i = 1; i < seen; ++i) { + int64_t difference = times[i] - times[0] - (int64_t)i * 64 * 1000000 / 48000; + assert(difference >= -1 && difference <= 1); + } + capture_missing = 1; seen = 0; audio_capture_hole(120); + now = time_us(); + assert(ma_device_handle_backend_data_callback(&device, NULL, input, 64) == MA_SUCCESS); + after = time_us(); + assert(seen == 1 && times[0] >= now - 100000 + 2500 && times[0] <= after - 100000 + 2500); + assert(!audio_stream_capture_missing(&device)); // Hint действует только внутри dispatch. + ma_data_converter_uninit(&device.capture.converter, NULL); + puts("audio timing: capture read index, converter offsets, explicit hole PASS"); +} diff --git a/tools/chatgui/tests/test_voice_audio_io.cpp b/tools/chatgui/tests/test_voice_audio_io.cpp new file mode 100644 index 00000000..cad83e53 --- /dev/null +++ b/tools/chatgui/tests/test_voice_audio_io.cpp @@ -0,0 +1,106 @@ +/* Настоящие workers/AEC; граница сервисного PCM заменена наблюдателями без сети/устройства. */ +#include "../src/voice_audio_io.h" +extern "C" { +#include "call/call_audio.h" +#include "radio/radio_audio.h" +#include "debug_config.h" +} +#include +#include +#include +#include +#include +#include +#include +#include +#include +#define REQUIRE(x) do { if (!(x)) { fprintf(stderr, "voice-io:%d %s\n", __LINE__, #x); abort(); } } while (0) +static std::mutex mutex; +static std::vector capture, transmitted; +static std::vector masks; +static bool talking; +static std::atomic pulls{0}, blocks{0}; +static std::atomic stall{false}, stalled{false}; +extern "C" int call_audio_pull_pcm(uint64_t, int16_t* out, int count) { ++pulls; std::fill_n(out, count, 1000); return count; } +extern "C" int radio_audio_pull_pcm(uint64_t, int16_t* out, int count) { + ++pulls; for (int i = 0; i < count; ++i) out[i] = i % 2 ? 2000 : 1000; return count; +} +extern "C" void call_audio_reset_io(uint64_t, int) {} +extern "C" void radio_audio_reset_playback(uint64_t) {} +extern "C" void radio_audio_capture_discontinuity(uint64_t) {} +extern "C" void radio_audio_talk_begin(uint64_t) { std::lock_guard lock(mutex); talking = true; } +extern "C" void radio_audio_talk_end(uint64_t) { std::lock_guard lock(mutex); talking = false; } +extern "C" int radio_audio_feed_pcm(uint64_t, const int16_t* pcm, int count) { + if (stall) { stalled = true; while (stall) std::this_thread::sleep_for(std::chrono::milliseconds(1)); } + std::lock_guard lock(mutex); + capture.insert(capture.end(), pcm, pcm + count); + if (talking) transmitted.insert(transmitted.end(), pcm, pcm + count); + ++blocks; return 0; +} +extern "C" int call_audio_feed_prepared_pcm(uint64_t, const int16_t* pcm, const uint8_t* mask) { + std::lock_guard lock(mutex); + capture.insert(capture.end(), pcm, pcm + 960); masks.insert(masks.end(), mask, mask + 960); + ++blocks; return 0; +} +static int64_t now_us() { + return std::chrono::duration_cast(std::chrono::steady_clock::now().time_since_epoch()).count(); +} +template static void await(F ready) { + for (int i = 0; i < 2000 && !ready(); ++i) std::this_thread::sleep_for(std::chrono::milliseconds(1)); + REQUIRE(ready()); +} +static void reset() { std::lock_guard lock(mutex); capture.clear(); transmitted.clear(); masks.clear(); talking = false; blocks = 0; } +int main() { + debug_config_init(); debug_set_level(DEBUG_LEVEL_WARN); + VoiceAudioIo voice; + std::array pcm{}; pcm.fill(1234); + voice.start(VoiceAudioIo::Kind::Radio, 7); REQUIRE(voice.startCapture(1, false)); + voice.capture(pcm.data(), 240, 0); voice.setPtt(true); + voice.capture(pcm.data(), 360, 0); voice.setPtt(false); voice.capture(pcm.data(), 360, 0); + await([] { return blocks.load() >= 3; }); voice.stopCapture(); + REQUIRE(capture.size() == 960 && transmitted.size() == 360 && !talking); + reset(); REQUIRE(voice.startCapture(1, false)); + voice.setPtt(true); voice.capture(pcm.data(), 120, 0); voice.setPtt(false); voice.stopCapture(); + REQUIRE(capture.size() == 120 && transmitted.size() == 120 && !talking); + reset(); REQUIRE(voice.startCapture(1, false)); stall = true; + voice.capture(pcm.data(), 960, 0); await([] { return stalled.load(); }); + int before = pulls; + for (int i = 0; i < 10; ++i) { + float output[1920]{}; voice.mix(output, 960); + std::this_thread::sleep_for(std::chrono::milliseconds(3)); + } + REQUIRE(pulls > before + 3); stall = false; voice.stopCapture(); voice.stop(); + reset(); voice.start(VoiceAudioIo::Kind::Call, 42); REQUIRE(voice.startCapture(1, false)); + voice.capture(pcm.data(), 480, 0); voice.setMuted(true); voice.capture(pcm.data(), 480, 0); + await([] { return blocks.load() == 1; }); voice.stopCapture(); + REQUIRE(masks.size() == 960); + for (int i = 0; i < 960; ++i) REQUIRE(masks[i] == (i >= 480) && capture[i] == 1234); + voice.stop(); reset(); + voice.start(VoiceAudioIo::Kind::Call, 42); REQUIRE(voice.startCapture(1, false)); + voice.capture(pcm.data(), 960, now_us() - 300000); // Backend latency не является возрастом worker-очереди. + await([] { return blocks.load() == 1; }); voice.stopCapture(); voice.stop(); + REQUIRE(capture.size() == 960); reset(); + // Референс противофазный: mono downmix дал бы ноль и не смог подавить эхо. + voice.start(VoiceAudioIo::Kind::Call, 43); REQUIRE(voice.startCapture(1, true)); + uint32_t rng = 1234567; + double echoPower = 0, outputPower = 0; + int64_t time = now_us(); + for (int frame = 0; frame < 240; ++frame) { + float reference[1920]; + for (int i = 0; i < 960; ++i) { + rng = rng * 1664525u + 1013904223u; + int16_t sample = (int16_t)((rng >> 18) - 8192); + reference[2*i] = sample / 32768.0f; reference[2*i+1] = -reference[2*i]; + pcm[i] = sample / 2; + if (frame >= 180) echoPower += (double)pcm[i] * pcm[i]; + } + voice.reference(reference, 960, time + frame * 20000); + voice.capture(pcm.data(), 960, time + frame * 20000); + await([&] { return blocks.load() == frame + 1; }); + } + voice.stopCapture(); voice.stop(); + for (size_t i = 180 * 960; i < capture.size(); ++i) outputPower += (double)capture[i] * capture[i]; + printf("stereo reference residual ratio=%.6f\n", outputPower / echoPower); + REQUIRE(outputPower < echoPower * 0.1); + puts("voice workers: exact PTT/mute boundaries, short tail, independent RX, stereo AEC PASS"); +} diff --git a/tools/chatgui/tests/test_voice_tx.cpp b/tools/chatgui/tests/test_voice_tx.cpp new file mode 100644 index 00000000..ae689e87 --- /dev/null +++ b/tools/chatgui/tests/test_voice_tx.cpp @@ -0,0 +1,143 @@ +/* Реальные codec/AGC/автоматы/очередь uasync; наблюдаем сетевую границу вместо отправки пакетов. */ +extern "C" { +#include "utun_instance.h" +#include "radio/radio_audio.h" +#include "call/call_audio.h" +#include "opus_codec.h" +#include "debug_config.h" +#include "u_async.h" +#include "silero_vad.h" +} +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#define REQUIRE(x) do { if (!(x)) { fprintf(stderr, "voice-tx:%d %s\n", __LINE__, #x); abort(); } } while (0) +static std::vector events; +static std::vector> packets; +static float probability; +static int vad_windows; +static std::atomic stall_vad{false}, vad_stalled{false}; +extern "C" silero_vad_t* __wrap_silero_vad_create_default() { return reinterpret_cast(1); } +extern "C" void __wrap_silero_vad_destroy(silero_vad_t*) {} +extern "C" void __wrap_silero_vad_reset(silero_vad_t*) {} +extern "C" int __wrap_silero_vad_process(silero_vad_t*, const float*, float* result) { + if (stall_vad) { vad_stalled = true; while (stall_vad) std::this_thread::sleep_for(std::chrono::milliseconds(1)); } + ++vad_windows; *result = probability; return 0; +} +extern "C" int __wrap_radio_talk_begin(UTUN_INSTANCE*, uint64_t) { events.push_back('B'); return 0; } +extern "C" int __wrap_radio_talk_end(UTUN_INSTANCE*, uint64_t) { events.push_back('F'); return 0; } +extern "C" int __wrap_radio_talk_send(UTUN_INSTANCE*, uint64_t, const uint8_t* data, int len) { + events.push_back('M'); packets.emplace_back(data, data + len); return 0; +} +extern "C" int __wrap_call_send_media(UTUN_INSTANCE*, uint64_t, const uint8_t* data, int len) { + packets.emplace_back(data, data + len); return 0; +} +extern "C" int __real_chat_setting_get_int(UTUN_INSTANCE*, const char*, int); +extern "C" int __wrap_chat_setting_get_int(UTUN_INSTANCE* inst, const char* key, int fallback) { + if (!strcmp(key, "radio_compressor_enabled") || !strcmp(key, "call_compressor_enabled")) return 0; + if (!strcmp(key, "radio_vad_hangover_ms")) return 0; + return __real_chat_setting_get_int(inst, key, fallback); +} +static void check_packet(opus_codec_encoder_t* encoder, const int16_t* pcm, size_t index) { + uint8_t data[1024]; int size = opus_codec_encode(encoder, pcm, 960, data, sizeof(data)); + REQUIRE(size > 0 && index < packets.size()); + REQUIRE(packets[index].size() == (size_t)size && !memcmp(packets[index].data(), data, size)); +} +int main() { + debug_config_init(); debug_set_level(DEBUG_LEVEL_WARN); + UTUN_INSTANCE instance{}; instance.ua = uasync_create(); REQUIRE(instance.ua); + REQUIRE(radio_audio_init(&instance) == 0 && call_audio_init(&instance) == 0); + REQUIRE(radio_audio_start(7, 0) == 0 && radio_audio_capture_start(7, 1, 0) == 0); + std::array input{}; + for (int i = 0; i < 1920; ++i) input[i] = (int16_t)(12000 * sin(i * 0.1)); + radio_audio_talk_begin(7); radio_audio_talk_begin(7); + REQUIRE(radio_audio_feed_pcm(7, input.data(), 1200) == 0); radio_audio_talk_end(7); + radio_audio_talk_begin(7); REQUIRE(radio_audio_feed_pcm(7, input.data() + 1200, 120) == 0); radio_audio_talk_end(7); + uasync_poll(instance.ua, 0); + REQUIRE(events == std::vector({'B','M','M','F','B','M','F'})); + auto* encoder = opus_codec_encoder_create(48000, 1); REQUIRE(encoder); + check_packet(encoder, input.data(), 0); + std::array tail{}; memcpy(tail.data(), input.data() + 960, 240 * sizeof(int16_t)); + check_packet(encoder, tail.data(), 1); + tail.fill(0); memcpy(tail.data(), input.data() + 1200, 120 * sizeof(int16_t)); check_packet(encoder, tail.data(), 2); + // Разрыв не склеивает отсчёты до и после потери внутри одного Opus-кадра. + packets.clear(); events.clear(); + radio_audio_talk_begin(7); radio_audio_feed_pcm(7, input.data(), 100); + radio_audio_capture_discontinuity(7); radio_audio_feed_pcm(7, input.data() + 100, 100); radio_audio_talk_end(7); + uasync_poll(instance.ua, 0); + REQUIRE(events == std::vector({'B','M','M','F'})); + tail.fill(0); memcpy(tail.data(), input.data(), 100 * sizeof(int16_t)); check_packet(encoder, tail.data(), 0); + tail.fill(0); memcpy(tail.data(), input.data() + 100, 100 * sizeof(int16_t)); check_packet(encoder, tail.data(), 1); + opus_codec_encoder_destroy(encoder); + // Capture teardown завершает принятую серию до FIN; следующее capture не получает старый хвост. + packets.clear(); events.clear(); + radio_audio_talk_begin(7); radio_audio_feed_pcm(7, input.data(), 120); radio_audio_capture_stop(); + REQUIRE(radio_audio_capture_start(7, 1, 0) == 0); + radio_audio_talk_begin(7); radio_audio_feed_pcm(7, input.data() + 120, 120); radio_audio_talk_end(7); + uasync_poll(instance.ua, 0); + REQUIRE(events == std::vector({'B','M','F','B','M','F'})); + // Старое прослушивание того же group ID отменено: старые BEGIN/media не попадают в новую сессию. + packets.clear(); events.clear(); + radio_audio_talk_begin(7); radio_audio_feed_pcm(7, input.data(), 120); radio_audio_stop(); + REQUIRE(radio_audio_start(7, 1) == 0); + uasync_poll(instance.ua, 0); + REQUIRE(events.empty() && packets.empty()); + // У остановленного core ограничены не только media, но и быстрые PTT-команды. + for (int i = 0; i < 1000; ++i) { + radio_audio_talk_begin(7); radio_audio_feed_pcm(7, input.data(), 120); radio_audio_talk_end(7); + } + int pending = 0; + for (auto* task = instance.ua->posted_tasks_head; task; task = task->next) ++pending; + REQUIRE(pending <= 24); // 16 control + 8 media, FIN каждой принятой серии уже зарезервирован. + uasync_poll(instance.ua, 0); + packets.clear(); events.clear(); radio_audio_capture_stop(); + REQUIRE(radio_audio_capture_start(7, 1, 1) == 0 && radio_audio_vad_mode()); + auto feed = [&](int frames) { + while (frames) { int count = std::min(frames, 1920); REQUIRE(radio_audio_feed_pcm(7, input.data(), count) == 0); frames -= count; } + }; + probability = 0.9f; radio_audio_talk_begin(7); feed(3072); // Два окна VAD даже во время manual PTT. + REQUIRE(vad_windows == 2); radio_audio_talk_end(7); REQUIRE(radio_audio_transmitting()); + probability = 0; feed(3072); REQUIRE(!radio_audio_transmitting()); // Первое окно начинает silence, второе завершает hangover. + uasync_poll(instance.ua, 0); + REQUIRE(std::count(events.begin(), events.end(), 'B') == 1 && std::count(events.begin(), events.end(), 'F') == 1); + packets.clear(); events.clear(); + probability = 0.9f; feed(3072); REQUIRE(radio_audio_transmitting()); + radio_audio_talk_begin(7); probability = 0; feed(3072); REQUIRE(radio_audio_transmitting()); + radio_audio_talk_end(7); uasync_poll(instance.ua, 0); + REQUIRE(std::count(events.begin(), events.end(), 'B') == 1 && std::count(events.begin(), events.end(), 'F') == 1); + stall_vad = true; + std::thread tx([&] { feed(1536); }); + for (int i = 0; i < 200 && !vad_stalled; ++i) std::this_thread::sleep_for(std::chrono::milliseconds(1)); + REQUIRE(vad_stalled); + std::atomic rx_finished{false}; + std::thread rx([&] { int16_t out[1920]; REQUIRE(radio_audio_pull_pcm(7, out, 1920) == 1920); rx_finished = true; }); + for (int i = 0; i < 200 && !rx_finished; ++i) std::this_thread::sleep_for(std::chrono::milliseconds(1)); + REQUIRE(rx_finished); stall_vad = false; tx.join(); rx.join(); + radio_audio_stop(); uasync_poll(instance.ua, 0); + // Старый FIN не обращается к новому ядру, даже когда allocator повторно использовал тот же адрес. + REQUIRE(radio_audio_start(7, 1) == 0); + radio_audio_talk_begin(7); uasync_poll(instance.ua, 0); + radio_audio_talk_end(7); radio_audio_stop(); radio_audio_destroy(&instance); + REQUIRE(radio_audio_init(&instance) == 0); + events.clear(); uasync_poll(instance.ua, 0); REQUIRE(events.empty()); + // Prepared policy исключает повторный core AEC; mute применяется только к нужной части кадра. + packets.clear(); REQUIRE(call_audio_start_prepared(42) == 0); + REQUIRE(call_audio_start(42) == -1); + REQUIRE(call_audio_feed_pcm(42, input.data(), 960) == -1); + std::array mask{}; + std::fill(mask.begin() + 300, mask.end(), 1); + REQUIRE(call_audio_feed_prepared_pcm(42, input.data(), mask.data()) == 0); + uasync_poll(instance.ua, 0); + encoder = opus_codec_encoder_create(48000, 1); REQUIRE(encoder); + tail.fill(0); memcpy(tail.data(), input.data(), 300 * sizeof(int16_t)); check_packet(encoder, tail.data(), 0); + call_audio_stop(); opus_codec_encoder_destroy(encoder); + call_audio_destroy(&instance); radio_audio_destroy(&instance); uasync_destroy(instance.ua, 0); + puts("voice TX: exact Opus tail, rapid bursts, discontinuity, capture policy, final mute PASS"); +} diff --git a/tools/chatgui/transport/miniaudio_impl.c b/tools/chatgui/transport/miniaudio_impl.c index 3829a742..6450e834 100644 --- a/tools/chatgui/transport/miniaudio_impl.c +++ b/tools/chatgui/transport/miniaudio_impl.c @@ -1,7 +1,14 @@ #include "audio_diagnostics.h" #ifdef __linux__ #include +#include static void audio_diag_pulse_init(void* device); +static void audio_timing_begin(void* device, unsigned int frames, int capture); +static void audio_timing_end(void); +static void audio_capture_hole(uint64_t offset); +#define MA_AUDIO_TIMING_BEGIN(device, frames, capture) audio_timing_begin(device, frames, capture) +#define MA_AUDIO_TIMING_END() audio_timing_end() +#define MA_AUDIO_CAPTURE_HOLE(offset) audio_capture_hole(offset) #define MA_DIAGNOSTIC_EVENT(device, kind, stage, value, detail) audio_diag_event(device, kind, stage, value, detail) #define MA_DIAGNOSTIC_LOG(log, level, message) audio_diag_log(log, level, message) #define MA_DIAGNOSTIC_DEVICE_INIT(device) audio_diag_pulse_init(device) @@ -69,3 +76,115 @@ int audio_context_failed(ma_context* context) { #endif return 0; } + +/* Только device callback того же PA mainloop: запрос использует готовый timing snapshot без roundtrip. */ +int64_t audio_stream_latency_us(ma_device* device, int capture, int* valid) { + *valid = 0; +#ifdef MA_SUPPORT_PULSEAUDIO + if (device && device->pContext->backend == ma_backend_pulseaudio && device->pContext->pulse.pa_stream_get_latency) { + ma_pa_stream* stream = capture ? device->pulse.pStreamCapture : device->pulse.pStreamPlayback; + ma_uint64 latency = 0; int negative = 0; + if (stream && ((ma_pa_stream_get_latency_proc)device->pContext->pulse.pa_stream_get_latency)(stream, &latency, &negative) == 0) { + *valid = 1; + return negative ? -(int64_t)latency : (int64_t)latency; + } + } +#endif + return 0; +} + +/* PA device name читается в его mainloop; GUI получает только непрозрачный идентификатор маршрута. */ +uint64_t audio_stream_route(ma_device* device, int capture) { +#ifdef MA_SUPPORT_PULSEAUDIO + if (device && device->pContext->backend == ma_backend_pulseaudio) { + ma_pa_stream* stream = capture ? device->pulse.pStreamCapture : device->pulse.pStreamPlayback; + const char* name = stream ? ((ma_pa_stream_get_device_name_proc)device->pContext->pulse.pa_stream_get_device_name)(stream) : NULL; + if (name && *name) { + uint64_t hash = UINT64_C(14695981039346656037); + for (; *name; ++name) { hash ^= (unsigned char)*name; hash *= UINT64_C(1099511628211); } + return hash; + } + } +#else + (void)device; (void)capture; +#endif + return 0; +} + +/* Reporter копирует сообщение в bounded очередь; DEBUG/file I/O в аппаратном потоке не выполняются. */ +void audio_stream_log_route(ma_device* device, int capture) { +#ifdef MA_SUPPORT_PULSEAUDIO + if (device && device->pContext->backend == ma_backend_pulseaudio) { + ma_pa_stream* stream = capture ? device->pulse.pStreamCapture : device->pulse.pStreamPlayback; + const char* name = stream ? ((ma_pa_stream_get_device_name_proc)device->pContext->pulse.pa_stream_get_device_name)(stream) : NULL; + if (name) { + char message[320]; + snprintf(message, sizeof(message), "PulseAudio physical %s route: %.255s", capture ? "capture" : "playback", name); + audio_diag_log(ma_device_get_log(device), MA_LOG_LEVEL_INFO, message); + } + } +#else + (void)device; (void)capture; +#endif +} + +#ifdef __linux__ +/* Один backend dispatch может породить несколько client callbacks при конвертации. + * Начало snapshot фиксируется один раз; offset растёт по фактически выданному client PCM. */ +static _Thread_local struct { + ma_device* device; + int valid; + int64_t first_us; + uint64_t frames; + uint64_t hole_offset; + int missing, hole_pending; +} audio_timing; + +static void audio_timing_begin(void* opaque, unsigned int frames, int capture) { + ma_device* device = opaque; + struct timespec now; + clock_gettime(CLOCK_MONOTONIC, &now); + int valid = 0; + int64_t latency = audio_stream_latency_us(device, capture, &valid); + ma_uint32 rate = capture ? device->capture.internalSampleRate : device->playback.internalSampleRate; + ma_uint64 converter_delay = capture ? ma_data_converter_get_output_latency(&device->capture.converter) + : ma_data_converter_get_input_latency(&device->playback.converter); + audio_timing.device = device; + audio_timing.missing = audio_timing.hole_pending; + audio_timing.valid = valid && device->noFixedSizedCallback; + int64_t time = now.tv_sec * INT64_C(1000000) + now.tv_nsec / 1000; + if (capture) { + time -= valid ? latency : (int64_t)frames * 1000000 / rate; + time -= (int64_t)converter_delay * 1000000 / device->sampleRate; + if (audio_timing.missing) time += (int64_t)audio_timing.hole_offset * 1000000 / rate; + } else { + time += latency + (int64_t)(converter_delay + device->playback.inputCacheRemaining) * 1000000 / device->sampleRate; + } + audio_timing.first_us = time; audio_timing.frames = 0; + audio_timing.hole_pending = 0; +} +static void audio_timing_end(void) { audio_timing.device = NULL; } +static void audio_capture_hole(uint64_t offset) { audio_timing.hole_pending = 1; audio_timing.hole_offset = offset; } +#endif + +int64_t audio_stream_frame_time(ma_device* device, unsigned int frames, int* valid) { + *valid = 0; +#ifdef __linux__ + if (audio_timing.device == device) { + int64_t time = audio_timing.first_us + (int64_t)audio_timing.frames * 1000000 / device->sampleRate; + audio_timing.frames += frames; *valid = audio_timing.valid; + return time; + } +#else + (void)device; (void)frames; +#endif + return 0; +} + +int audio_stream_capture_missing(ma_device* device) { +#ifdef __linux__ + return audio_timing.device == device && audio_timing.missing; +#else + (void)device; return 0; +#endif +}