#include "voice_recorder.h" #include "audio_compressor.h" #include "instance_lite.h" #include "../../../lib/opus_codec.h" #include "../../../lib/debug_config.h" #include "../../../lib/mem.h" #include "../../../lib/platform_compat.h" #include "../../../lib/u_async.h" #include "../../../src/chat/chat_core.h" #include #include #include #include #include #include #include #include #include #define OPUS_MAGIC 0x5355504F #define FRAME_MS 20 #define MAX_PACKET 4000 static const int s_bitrates[] = {16000, 32000, 64000}; static const int s_complexities[] = {2, 5, 10}; struct voice_recorder { pthread_mutex_t mtx; int active; char channel_id[64]; int sample_rate; int channels; int frame_samples; int16_t* pcm_buffer; size_t pcm_count; size_t pcm_cap; struct audio_compressor* compressor; int compressor_enabled; char db_path[512]; int preset; float peak_level; float waveform_levels[100]; int waveform_count; int64_t next_wf_time_ms; int64_t start_time_ms; }; static struct voice_recorder* g_rec = NULL; static pthread_mutex_t g_init_mtx = PTHREAD_MUTEX_INITIALIZER; static int64_t now_ms(void) { struct timespec ts; clock_gettime(CLOCK_MONOTONIC, &ts); return ts.tv_sec * 1000LL + ts.tv_nsec / 1000000LL; } int voice_recorder_init(const char* db_path) { pthread_mutex_lock(&g_init_mtx); if (g_rec) { pthread_mutex_unlock(&g_init_mtx); return 0; } g_rec = u_calloc(1, sizeof(*g_rec)); if (!g_rec) { DEBUG_ERROR(DEBUG_CATEGORY_GENERAL, "voice_recorder_init: OOM"); pthread_mutex_unlock(&g_init_mtx); return -1; } pthread_mutex_init(&g_rec->mtx, NULL); snprintf(g_rec->db_path, sizeof(g_rec->db_path), "%s", db_path ? db_path : ""); g_rec->preset = 1; g_rec->compressor_enabled = 1; g_rec->compressor = audio_compressor_create(); if (g_rec->compressor) { audio_compressor_config_t cfg = {0}; cfg.sample_rate = 48000; cfg.channels = 1; cfg.block_duration_ms = 20; cfg.lookback_ms = 200; cfg.lookahead_ms = 100; cfg.max_gain_db = 25.0f; cfg.rise_rate_per_sec = 10.0f; cfg.target_level = 1.0f; audio_compressor_configure(g_rec->compressor, &cfg); } DEBUG_INFO(DEBUG_CATEGORY_GENERAL, "voice_recorder_init: db=%s preset=%d compressor=%d", g_rec->db_path, g_rec->preset, g_rec->compressor_enabled); debug_set_category_level(DEBUG_CATEGORY_GENERAL, DEBUG_LEVEL_INFO); pthread_mutex_unlock(&g_init_mtx); return 0; } void voice_recorder_deinit(void) { pthread_mutex_lock(&g_init_mtx); if (!g_rec) { pthread_mutex_unlock(&g_init_mtx); return; } if (g_rec->active) { u_free(g_rec->pcm_buffer); g_rec->pcm_buffer = NULL; g_rec->pcm_count = 0; g_rec->pcm_cap = 0; g_rec->active = 0; } audio_compressor_destroy(g_rec->compressor); pthread_mutex_destroy(&g_rec->mtx); u_free(g_rec); g_rec = NULL; DEBUG_INFO(DEBUG_CATEGORY_GENERAL, "voice_recorder_deinit"); pthread_mutex_unlock(&g_init_mtx); } int voice_recorder_start(const char* channel_id, int sample_rate, int channels) { pthread_mutex_lock(&g_init_mtx); if (!g_rec) { DEBUG_ERROR(DEBUG_CATEGORY_GENERAL, "voice_recorder_start: not initialized (g_rec=NULL)"); pthread_mutex_unlock(&g_init_mtx); return -1; } pthread_mutex_lock(&g_rec->mtx); if (g_rec->active) { DEBUG_WARN(DEBUG_CATEGORY_GENERAL, "voice_recorder_start: already active"); pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); return -1; } if (sample_rate <= 0) sample_rate = 48000; if (channels <= 0) channels = 1; g_rec->sample_rate = sample_rate; g_rec->channels = channels; g_rec->frame_samples = sample_rate * FRAME_MS / 1000; snprintf(g_rec->channel_id, sizeof(g_rec->channel_id), "%s", channel_id ? channel_id : ""); g_rec->pcm_count = 0; g_rec->active = 1; g_rec->waveform_count = 0; g_rec->next_wf_time_ms = 0; struct timespec ts; clock_gettime(CLOCK_MONOTONIC, &ts); g_rec->start_time_ms = ts.tv_sec * 1000LL + ts.tv_nsec / 1000000LL; if (g_rec->compressor && g_rec->compressor_enabled) { audio_compressor_reset(g_rec->compressor); audio_compressor_set_enabled(g_rec->compressor, 1); } DEBUG_INFO(DEBUG_CATEGORY_GENERAL, "voice_recorder_start: ch=%s rate=%d chans=%d frame=%d", g_rec->channel_id, g_rec->sample_rate, g_rec->channels, g_rec->frame_samples); pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); return 0; } int voice_recorder_feed(const int16_t* samples, int count) { if (!samples || count <= 0) return -1; pthread_mutex_lock(&g_init_mtx); if (!g_rec) { pthread_mutex_unlock(&g_init_mtx); return -1; } pthread_mutex_lock(&g_rec->mtx); if (!g_rec->active) { pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); return -1; } /* compute peak level for UI */ { float sum_sq = 0.0f; float peak = 0.0f; for (int i = 0; i < count; i++) { float v = (float)samples[i] / 32768.0f; sum_sq += v * v; float av = fabsf(v); if (av > peak) peak = av; } g_rec->peak_level = (sqrtf(sum_sq / (float)count) * 2.0f + peak) / 2.0f; /* accumulate waveform level every ~100ms */ { int64_t now = now_ms(); if (g_rec->waveform_count < 100 && now >= g_rec->next_wf_time_ms) { g_rec->waveform_levels[g_rec->waveform_count++] = g_rec->peak_level; g_rec->next_wf_time_ms = now + 100; } } } if (g_rec->compressor && g_rec->compressor_enabled) { audio_compressor_push(g_rec->compressor, samples, (size_t)count); } else { size_t new_cnt = g_rec->pcm_count + (size_t)count; if (new_cnt > g_rec->pcm_cap) { size_t new_cap = g_rec->pcm_cap ? g_rec->pcm_cap * 2 : (size_t)count * 2; while (new_cap < new_cnt) new_cap *= 2; int16_t* tmp = u_realloc(g_rec->pcm_buffer, new_cap * sizeof(int16_t)); if (!tmp) { pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); return -1; } g_rec->pcm_buffer = tmp; g_rec->pcm_cap = new_cap; } memcpy(g_rec->pcm_buffer + g_rec->pcm_count, samples, (size_t)count * sizeof(int16_t)); g_rec->pcm_count = new_cnt; } pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); return 0; } static void voice_cleanup_locked(struct voice_recorder* rec) { u_free(rec->pcm_buffer); rec->pcm_buffer = NULL; rec->pcm_count = 0; rec->pcm_cap = 0; rec->active = 0; } int voice_recorder_stop(int* out_duration_ms) { pthread_mutex_lock(&g_init_mtx); if (!g_rec) { pthread_mutex_unlock(&g_init_mtx); return -1; } pthread_mutex_lock(&g_rec->mtx); if (!g_rec->active) { DEBUG_WARN(DEBUG_CATEGORY_GENERAL, "voice_recorder_stop: not active"); pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); return -1; } int64_t elapsed_ms = now_ms() - g_rec->start_time_ms; if (out_duration_ms) *out_duration_ms = (int)elapsed_ms; struct UASYNC* ua = instance_lite_get_uasync(); if (!ua) { DEBUG_ERROR(DEBUG_CATEGORY_GENERAL, "voice_recorder_stop: no uasync, discarding"); voice_cleanup_locked(g_rec); pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); return -1; } if (elapsed_ms < 1000) { DEBUG_INFO(DEBUG_CATEGORY_GENERAL, "voice_recorder_stop: too short %lldms, discarding", (long long)elapsed_ms); voice_cleanup_locked(g_rec); pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); return 0; } int16_t* final_pcm = NULL; size_t final_pcm_count = 0; if (g_rec->compressor && g_rec->compressor_enabled) { audio_compressor_flush(g_rec->compressor); final_pcm = (int16_t*)audio_compressor_output(g_rec->compressor); final_pcm_count = audio_compressor_output_size(g_rec->compressor); } else { final_pcm = g_rec->pcm_buffer; final_pcm_count = g_rec->pcm_count; } if (!final_pcm || final_pcm_count == 0) { DEBUG_WARN(DEBUG_CATEGORY_GENERAL, "voice_recorder_stop: no PCM data"); voice_cleanup_locked(g_rec); pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); return -1; } /* generate file name parts */ time_t t = time(NULL); struct tm tm_buf; localtime_r(&t, &tm_buf); char dt[32], basename[256], suffix[8]; snprintf(dt, sizeof(dt), "%04d%02d%02d-%02d%02d%02d", tm_buf.tm_year + 1900, tm_buf.tm_mon + 1, tm_buf.tm_mday, tm_buf.tm_hour, tm_buf.tm_min, tm_buf.tm_sec); int rnd = rand() & 0xFFFF; snprintf(basename, sizeof(basename), "voice_%s_%04x", dt, rnd); snprintf(suffix, sizeof(suffix), "%04x", rnd); /* ensure media dir exists */ char media_dir[1024]; snprintf(media_dir, sizeof(media_dir), "%s/media/%s", g_rec->db_path, g_rec->channel_id); DEBUG_INFO(DEBUG_CATEGORY_GENERAL, "voice_recorder_stop: db_path=%s ch=%s media_dir=%s", g_rec->db_path, g_rec->channel_id, media_dir); { char tmp[1024]; size_t off = 0; for (size_t i = 0; media_dir[i] && off < sizeof(tmp) - 1; i++) { tmp[off++] = media_dir[i]; if (media_dir[i] == '/' && off > 1) { tmp[off - 1] = '\0'; if (mkdir(tmp, 0755) != 0 && errno != EEXIST) DEBUG_WARN(DEBUG_CATEGORY_GENERAL, "voice_recorder_stop: mkdir(%s) errno=%d", tmp, errno); tmp[off - 1] = '/'; } } if (mkdir(tmp, 0755) != 0 && errno != EEXIST) DEBUG_WARN(DEBUG_CATEGORY_GENERAL, "voice_recorder_stop: mkdir(%s) errno=%d", tmp, errno); } /* encode to Opus and write blocks */ opus_codec_encoder_t* enc = opus_codec_encoder_create(g_rec->sample_rate, g_rec->channels); if (!enc) { DEBUG_ERROR(DEBUG_CATEGORY_GENERAL, "voice_recorder_stop: encoder create failed"); voice_cleanup_locked(g_rec); pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); return -1; } opus_codec_encoder_bitrate_set(enc, s_bitrates[g_rec->preset]); opus_codec_encoder_complexity_set(enc, s_complexities[g_rec->preset]); char temp_path[1280]; snprintf(temp_path, sizeof(temp_path), "%s/%s_%s_%s.opus", media_dir, dt, basename, suffix); FILE* of = fopen(temp_path, "wb"); if (!of) { DEBUG_ERROR(DEBUG_CATEGORY_GENERAL, "voice_recorder_stop: cannot create %s (errno=%d)", temp_path, errno); opus_codec_encoder_destroy(enc); voice_cleanup_locked(g_rec); pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); return -1; } /* Opus header */ uint32_t magic = OPUS_MAGIC; uint32_t sr = (uint32_t)g_rec->sample_rate; uint16_t ch = (uint16_t)g_rec->channels; uint16_t fs = (uint16_t)g_rec->frame_samples; fwrite(&magic, 4, 1, of); fwrite(&sr, 4, 1, of); fwrite(&ch, 2, 1, of); fwrite(&fs, 2, 1, of); uint8_t packet[MAX_PACKET]; int frame_count = 0; size_t pcm_off = 0; size_t total_samples = g_rec->channels == 1 ? final_pcm_count : final_pcm_count / g_rec->channels; while (pcm_off + (size_t)g_rec->frame_samples * g_rec->channels <= final_pcm_count) { int len = opus_codec_encode(enc, final_pcm + pcm_off, g_rec->frame_samples, packet, MAX_PACKET); pcm_off += (size_t)g_rec->frame_samples * g_rec->channels; if (len > 0) { uint16_t plen = (uint16_t)len; fwrite(&plen, 2, 1, of); fwrite(packet, 1, (size_t)len, of); frame_count++; } else if (len < 0) { DEBUG_WARN(DEBUG_CATEGORY_GENERAL, "voice_recorder_stop: encode error frame %d: %d", frame_count, len); } } uint16_t zero = 0; fwrite(&zero, 2, 1, of); fclose(of); opus_codec_encoder_destroy(enc); float duration_sec = (float)frame_count * FRAME_MS / 1000.0f; DEBUG_INFO(DEBUG_CATEGORY_GENERAL, "voice_recorder_stop: wrote %d frames (%.1fs) dur=%lldms compressor=%d path=%s", frame_count, duration_sec, (long long)elapsed_ms, g_rec->compressor_enabled, temp_path); if (frame_count <= 0) { unlink(temp_path); voice_cleanup_locked(g_rec); pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); return -1; } /* file is already at temp_path in media dir – no splitting needed */ /* media_index_register_async will handle block size calculation, signing, and DB */ /* build compact waveform: 100 bytes base64 (A-Za-z0-9+/), log scale */ static const char b64_table[] = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; char wf_data[256]; snprintf(wf_data, sizeof(wf_data), "wf="); int wf_off = 3; { int wfN = g_rec->waveform_count > 0 ? g_rec->waveform_count : 1; for (int i = 0; i < 100; i++) { int s = i * wfN / 100, e = (i + 1) * wfN / 100; if (s >= wfN) s = wfN - 1; if (e <= s) e = s + 1; if (e > wfN) e = wfN; float sum = 0; int cnt = 0; for (int j = s; j < e; j++) { sum += g_rec->waveform_levels[j]; cnt++; } float v = cnt > 0 ? sum / cnt : 0; float db = v > 0.00001f ? 20.0f * log10f(v) : -100.0f; int idx = (int)((db + 60.0f) / 60.0f * 64.0f); if (idx < 0) idx = 0; if (idx > 63) idx = 63; wf_data[wf_off++] = b64_table[idx]; } } wf_off += snprintf(wf_data + wf_off, sizeof(wf_data) - wf_off, ";dur=%lld;", (long long)elapsed_ms); size_t total_data_len = (size_t)(wf_off); struct chat_msg_submit* req = u_calloc(1, sizeof(struct chat_msg_submit) + total_data_len + 1); if (!req) { voice_cleanup_locked(g_rec); pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); return -1; } req->inst = instance_lite_get_instance(); snprintf(req->channel_id, sizeof(req->channel_id), "%s", g_rec->channel_id); snprintf(req->content_type, sizeof(req->content_type), "audio/opus"); snprintf(req->media_src, sizeof(req->media_src), "%s", temp_path); snprintf(req->media_dest, sizeof(req->media_dest), "%s", temp_path); req->media_copy = 0; req->data = (uint8_t*)(req + 1); req->data_len = (uint32_t)total_data_len; memcpy(req->data, wf_data, total_data_len); req->timestamp = 0; DEBUG_INFO(DEBUG_CATEGORY_GENERAL, "voice_recorder_stop: posting media ch=%s path=%s", req->channel_id, temp_path); uasync_post(ua, chat_core_submit_trampoline, req); voice_cleanup_locked(g_rec); pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); return 0; } void voice_recorder_cancel(void) { pthread_mutex_lock(&g_init_mtx); if (!g_rec) { pthread_mutex_unlock(&g_init_mtx); return; } pthread_mutex_lock(&g_rec->mtx); if (!g_rec->active) { pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); return; } DEBUG_INFO(DEBUG_CATEGORY_GENERAL, "voice_recorder_cancel"); voice_cleanup_locked(g_rec); pthread_mutex_unlock(&g_rec->mtx); pthread_mutex_unlock(&g_init_mtx); } int voice_recorder_is_active(void) { int active = 0; pthread_mutex_lock(&g_init_mtx); if (g_rec) { pthread_mutex_lock(&g_rec->mtx); active = g_rec->active; pthread_mutex_unlock(&g_rec->mtx); } pthread_mutex_unlock(&g_init_mtx); return active; } void voice_recorder_set_preset(int preset) { if (preset < 0) preset = 0; if (preset > 2) preset = 2; pthread_mutex_lock(&g_init_mtx); if (g_rec) { pthread_mutex_lock(&g_rec->mtx); g_rec->preset = preset; DEBUG_INFO(DEBUG_CATEGORY_GENERAL, "voice_recorder_set_preset: %d (%d kbps)", preset, s_bitrates[preset] / 1000); pthread_mutex_unlock(&g_rec->mtx); } pthread_mutex_unlock(&g_init_mtx); } int voice_recorder_get_preset(void) { int p = 1; pthread_mutex_lock(&g_init_mtx); if (g_rec) { pthread_mutex_lock(&g_rec->mtx); p = g_rec->preset; pthread_mutex_unlock(&g_rec->mtx); } pthread_mutex_unlock(&g_init_mtx); return p; } void voice_recorder_set_compressor_enabled(int enabled) { pthread_mutex_lock(&g_init_mtx); if (g_rec) { pthread_mutex_lock(&g_rec->mtx); g_rec->compressor_enabled = enabled ? 1 : 0; DEBUG_INFO(DEBUG_CATEGORY_GENERAL, "voice_recorder_set_compressor: %s", g_rec->compressor_enabled ? "on" : "off"); pthread_mutex_unlock(&g_rec->mtx); } pthread_mutex_unlock(&g_init_mtx); } int voice_recorder_is_compressor_enabled(void) { int e = 1; pthread_mutex_lock(&g_init_mtx); if (g_rec) { pthread_mutex_lock(&g_rec->mtx); e = g_rec->compressor_enabled; pthread_mutex_unlock(&g_rec->mtx); } pthread_mutex_unlock(&g_init_mtx); return e; } void voice_recorder_set_compressor_config(int max_gain_db, int lookback_ms, int lookahead_ms, float rise_rate_per_sec, float target_level_db) { pthread_mutex_lock(&g_init_mtx); if (g_rec && g_rec->compressor) { pthread_mutex_lock(&g_rec->mtx); audio_compressor_config_t cfg = {0}; cfg.sample_rate = 48000; cfg.channels = 1; cfg.block_duration_ms = 20; cfg.lookback_ms = lookback_ms; cfg.lookahead_ms = lookahead_ms; cfg.max_gain_db = (float)max_gain_db; cfg.rise_rate_per_sec = rise_rate_per_sec; cfg.target_level = powf(10.0f, target_level_db / 20.0f); audio_compressor_configure(g_rec->compressor, &cfg); DEBUG_INFO(DEBUG_CATEGORY_GENERAL, "voice_recorder compressor config: gain=%ddB lookback=%d lookahead=%d rise=%.1f target=%.0fdBFS", max_gain_db, lookback_ms, lookahead_ms, rise_rate_per_sec, target_level_db); pthread_mutex_unlock(&g_rec->mtx); } pthread_mutex_unlock(&g_init_mtx); } float voice_recorder_get_peak_level(void) { float v = 0.0f; pthread_mutex_lock(&g_init_mtx); if (g_rec) { pthread_mutex_lock(&g_rec->mtx); v = g_rec->peak_level; pthread_mutex_unlock(&g_rec->mtx); } pthread_mutex_unlock(&g_init_mtx); return v; }