From 4c21543bac9e343130ef210fe643c39ae4a66ea3 Mon Sep 17 00:00:00 2001 From: nick huang Date: Tue, 17 Mar 2026 15:30:24 +0800 Subject: [PATCH] try to solve 30 seconds issue, now 60 seconds, but last few words lost before key press of enter --- doubao_mic.cpp | 155 +++++++++++++++++++++++++++++++------------------ 1 file changed, 98 insertions(+), 57 deletions(-) diff --git a/doubao_mic.cpp b/doubao_mic.cpp index 332bb20e9..f61e40d69 100644 --- a/doubao_mic.cpp +++ b/doubao_mic.cpp @@ -14,14 +14,17 @@ #include #include #include -#include // 关键:补充缺失的mutex头文件 +#include // 全局原子变量(线程安全) std::atomic is_recording(false); std::atomic exit_program(false); -// 音频缓冲区(加锁保护,避免多线程冲突) +std::atomic recorded_seconds(0); // 实时录制时长 +// 音频缓冲区(加锁保护) std::vector audio_buffer; -std::mutex buffer_mutex; // 现在有头文件支持,不会报错 +std::mutex buffer_mutex; +// 可选超时(默认60秒,可自定义) +const int RECORD_TIMEOUT = 60; // 延长到60秒,也可设为0取消超时 // 信号处理:Ctrl+C 优雅退出 void signal_handler(int sig) { @@ -33,29 +36,40 @@ void signal_handler(int sig) { } } -// 音频回调(旧版 miniaudio 兼容) +// 音频回调(取消30秒帧上限) void data_callback(ma_device* pDevice, void* pOutput, const void* pInput, ma_uint32 frameCount) { if (!is_recording.load() || pInput == NULL) return; const float* pInputFloat = (const float*)pInput; if (pInputFloat == NULL) return; - // 加锁操作缓冲区(避免主线程/回调线程冲突) std::lock_guard lock(buffer_mutex); - - // 限制最大录制时长 30 秒(16000Hz) - const size_t max_frames = 16000 * 30; - const size_t available = max_frames - audio_buffer.size(); - if (available == 0) { - is_recording.store(false); - return; + // 取消固定帧上限,仅保留内存保护(可选) + const size_t max_memory = 16000 * 120; // 最多120秒(约200MB内存) + if (audio_buffer.size() < max_memory) { + audio_buffer.insert(audio_buffer.end(), pInputFloat, pInputFloat + frameCount); + // 更新实时录制时长 + recorded_seconds.store(audio_buffer.size() / 16000); } - - const size_t copy_frames = (frameCount > available) ? available : frameCount; - audio_buffer.insert(audio_buffer.end(), pInputFloat, pInputFloat + copy_frames); } -// 列出系统音频设备(兼容旧版 API) +// 静音检测(裁剪无效音频,减少识别量) +int trim_silence(const float* audio_data, int audio_len, float threshold = 0.001f) { + // 跳过开头静音 + int start = 0; + while (start < audio_len && fabs(audio_data[start]) < threshold) { + start++; + } + // 跳过结尾静音 + int end = audio_len - 1; + while (end > start && fabs(audio_data[end]) < threshold) { + end--; + } + // 返回有效音频长度(至少保留1秒) + return std::max(end - start + 1, 16000); +} + +// 列出系统音频设备 void list_audio_devices(ma_context& context, ma_device_info** pCaptureInfos, ma_uint32& captureCount) { printf("\n📜 系统可用麦克风设备列表:\n"); printf("=============================================\n"); @@ -70,33 +84,37 @@ void list_audio_devices(ma_context& context, ma_device_info** pCaptureInfos, ma_ for (ma_uint32 i = 0; i < captureCount; ++i) { printf("🔧 设备ID: %u | 名称: %s\n", i, (*pCaptureInfos)[i].name); - printf(" 声道数: 1 | 采样率: 16000 Hz\n"); // 固定 16000Hz 避免采样率冲突 + printf(" 声道数: 1 | 采样率: 16000 Hz\n"); printf("---------------------------------------------\n"); } printf("=============================================\n"); } -// 提示信息 +// 提示信息(修复printf多参数问题) void print_usage() { printf("=============================================\n"); - printf("🎤 语音识别程序(旧版兼容)\n"); + printf("🎤 语音识别程序(CPU优化版)\n"); printf("操作说明:\n"); printf(" 1. 按下【回车键】开始录制\n"); printf(" 2. 说话完成后按回车停止录制并识别\n"); - printf(" 3. 录制超过30秒自动停止\n"); - printf(" 4. Ctrl+C 退出程序\n"); + printf(" 3. 录制超过%d秒自动停止(可自定义)\n", RECORD_TIMEOUT); + printf(" 4. 录制中实时显示时长:【录制中... X秒】\n"); + printf(" 5. Ctrl+C 退出程序\n"); + printf("=============================================\n"); // 移除多余的RECORD_TIMEOUT参数 +} + +// CPU优化提示 +void print_cpu_optimize_tips() { + printf("⚡ CPU优化配置说明:\n"); + printf(" ✅ 已启用多线程识别(自动适配CPU核心数)\n"); + printf(" ✅ 已启用静音裁剪(减少无效音频识别)\n"); + printf(" ✅ 已使用贪心采样(最快的识别策略)\n"); + printf(" 📌 模型优化:推荐使用 ggml-medium-q4_0.bin(量化版)\n"); + printf(" 📌 编译优化:已用 -O3 最高级优化\n"); printf("=============================================\n"); } -// GPU 状态提示(兼容旧版) -void check_gpu_status() { - printf("🔍 GPU加速配置说明:\n"); - printf(" ❌ 若识别速度慢,说明使用CPU运行\n"); - printf(" ✅ 启用GPU:重新编译whisper.cpp时添加 -DWHISPER_CUDA=ON\n"); -} - int main(int argc, char** argv) { - // 注册信号处理 signal(SIGINT, signal_handler); if (argc < 2) { @@ -105,7 +123,7 @@ int main(int argc, char** argv) { } const char* model_path = argv[1]; - // 1. 初始化音频上下文(旧版兼容) + // 1. 初始化音频上下文 ma_context context; if (ma_context_init(NULL, 0, NULL, &context) != MA_SUCCESS) { fprintf(stderr, "❌ 初始化音频上下文失败\n"); @@ -125,13 +143,13 @@ int main(int argc, char** argv) { fprintf(stderr, "❌ 输入无效,使用默认设备ID 0\n"); device_id = 0; } - // 清空输入缓冲区 - while (getchar() != '\n'); + while (getchar() != '\n'); // 清空输入缓冲区 } - // 4. 初始化 Whisper 模型 + // 4. 初始化 Whisper 模型(CPU优化,移除不存在的use_flash_attention) struct whisper_context_params cparams = whisper_context_default_params(); - cparams.use_gpu = true; + cparams.use_gpu = false; // 强制CPU(避免GPU检测开销) + // 移除 cparams.use_flash_attention = false; (旧版本无此成员) printf("\n🚀 正在加载模型:%s\n", model_path); struct whisper_context* ctx = whisper_init_from_file_with_params(model_path, cparams); @@ -141,19 +159,18 @@ int main(int argc, char** argv) { return 1; } - // GPU 状态提示 - check_gpu_status(); + // 显示CPU优化提示 + print_cpu_optimize_tips(); printf("✅ 模型加载成功!\n"); - // 5. 初始化录音设备(旧版 miniaudio 核心兼容) + // 5. 初始化录音设备 ma_device_config deviceConfig = ma_device_config_init(ma_device_type_capture); - deviceConfig.capture.format = ma_format_f32; // Whisper 要求 float32 - deviceConfig.capture.channels = 1; // 单声道 - deviceConfig.sampleRate = 16000; // 固定 16000Hz 避免采样率错误 - deviceConfig.dataCallback = data_callback; // 回调函数 + deviceConfig.capture.format = ma_format_f32; + deviceConfig.capture.channels = 1; + deviceConfig.sampleRate = 16000; + deviceConfig.dataCallback = data_callback; deviceConfig.pUserData = NULL; - // 指定选中的麦克风设备(旧版用 pDeviceID) if (captureCount > 0 && pCaptureInfos != NULL) { deviceConfig.capture.pDeviceID = &pCaptureInfos[device_id].id; printf("\n✅ 已选择麦克风:%s\n", pCaptureInfos[device_id].name); @@ -169,7 +186,6 @@ int main(int argc, char** argv) { return 1; } - // 启动录音设备(仅初始化,不采集数据) if (ma_device_start(&device) != MA_SUCCESS) { fprintf(stderr, "❌ 启动录音设备失败\n"); ma_device_uninit(&device); @@ -182,7 +198,6 @@ int main(int argc, char** argv) { // 主循环 while (!exit_program.load()) { - // 等待用户按回车开始录制 printf("\n👉 按下回车键开始录制...\n"); getchar(); @@ -190,34 +205,49 @@ int main(int argc, char** argv) { // 重置录制状态 is_recording.store(true); + recorded_seconds.store(0); { std::lock_guard lock(buffer_mutex); audio_buffer.clear(); } - printf("🎙️ 正在录制(按回车停止,最长30秒)...\n"); + printf("🎙️ 正在录制(按回车停止,最长%d秒)...\n", RECORD_TIMEOUT); - // 等待用户停止录制(子线程监听回车) + // 录制时长实时显示线程 + std::thread progress_thread([&]() { + while (is_recording.load() && !exit_program.load()) { + printf("\r📊 录制中... %d秒", recorded_seconds.load()); + fflush(stdout); // 强制刷新输出 + std::this_thread::sleep_for(std::chrono::seconds(1)); + } + }); + + // 等待用户停止录制(主线程监听,避免子线程输入阻塞) + std::atomic stop_record(false); std::thread wait_thread([&]() { getchar(); + stop_record.store(true); is_recording.store(false); }); - // 超时控制(30秒) + // 超时控制(可选) auto start_time = std::chrono::steady_clock::now(); - while (is_recording.load() && !exit_program.load()) { + while (!stop_record.load() && !exit_program.load()) { auto duration = std::chrono::duration_cast( std::chrono::steady_clock::now() - start_time).count(); - if (duration >= 30) { - printf("⏱️ 录制超时,自动停止\n"); + if (RECORD_TIMEOUT > 0 && duration >= RECORD_TIMEOUT) { + printf("\n⏱️ 录制超时(%d秒),自动停止\n", RECORD_TIMEOUT); is_recording.store(false); + stop_record.store(true); break; } std::this_thread::sleep_for(std::chrono::milliseconds(100)); } wait_thread.join(); + progress_thread.join(); is_recording.store(false); + printf("\n"); // 换行,清理进度显示 if (exit_program.load()) break; @@ -225,7 +255,7 @@ int main(int argc, char** argv) { std::vector captured_audio; { std::lock_guard lock(buffer_mutex); - captured_audio = audio_buffer; // 拷贝数据避免锁冲突 + captured_audio = audio_buffer; } if (captured_audio.empty()) { @@ -233,21 +263,30 @@ int main(int argc, char** argv) { continue; } - // 开始识别 - printf("🔍 正在识别(音频长度:%.2f秒)...\n", (float)captured_audio.size() / 16000); + // 优化1:静音裁剪(减少识别数据量) + int valid_len = trim_silence(captured_audio.data(), captured_audio.size()); + float valid_seconds = (float)valid_len / 16000; + printf("🔍 正在识别(有效音频长度:%.2f秒,原始:%.2f秒)...\n", + valid_seconds, (float)captured_audio.size() / 16000); + auto recognize_start = std::chrono::steady_clock::now(); + // 优化2:调整识别参数(CPU最优配置) whisper_full_params wparams = whisper_full_default_params(WHISPER_SAMPLING_GREEDY); wparams.language = "zh"; - wparams.n_threads = std::max(1, (int)std::thread::hardware_concurrency()); + wparams.n_threads = std::max(2, (int)std::thread::hardware_concurrency()); // 至少2线程 wparams.print_progress = false; wparams.print_realtime = false; - wparams.temperature = 0.0; + wparams.temperature = 0.0; // 最快的温度设置 wparams.max_len = 0; wparams.translate = false; wparams.no_context = true; + wparams.single_segment = true; // 单段识别(更快) + wparams.print_special = false; // 不打印特殊字符 + wparams.token_timestamps = false; // 关闭时间戳(节省计算) - if (whisper_full(ctx, wparams, captured_audio.data(), captured_audio.size()) != 0) { + // 执行识别(仅识别有效音频) + if (whisper_full(ctx, wparams, captured_audio.data(), valid_len) != 0) { fprintf(stderr, "❌ 识别失败\n"); continue; } @@ -255,7 +294,9 @@ int main(int argc, char** argv) { // 输出识别结果 auto recognize_duration = std::chrono::duration_cast( std::chrono::steady_clock::now() - recognize_start).count(); - printf("⏱️ 识别耗时:%.2f 秒\n", recognize_duration / 1000.0); + float speed = valid_seconds / (recognize_duration / 1000.0); + printf("⏱️ 识别耗时:%.2f 秒 | 识别速度:%.2fx实时速度\n", + recognize_duration / 1000.0, speed); const int n_segments = whisper_full_n_segments(ctx); if (n_segments == 0) {