From 8146d4f030f6610c6693fa83263929029268c7a3 Mon Sep 17 00:00:00 2001 From: nick huang Date: Tue, 17 Mar 2026 15:07:43 +0800 Subject: [PATCH] compile and download run with microphone, not solve usb apply headset crash issue yet --- compile.txt | 5 + doubao.cpp | 157 +++++++++++++++++++++++++ doubao_gpu.cpp | 195 +++++++++++++++++++++++++++++++ doubao_mic.cpp | 299 ++++++++++++++++++++++++++++++++++++++++++++++++ download.txt | 2 + minimal_mic.cpp | 82 +++++++++++++ run.txt | 2 + 7 files changed, 742 insertions(+) create mode 100644 compile.txt create mode 100644 doubao.cpp create mode 100644 doubao_gpu.cpp create mode 100644 doubao_mic.cpp create mode 100644 download.txt create mode 100644 minimal_mic.cpp create mode 100644 run.txt diff --git a/compile.txt b/compile.txt new file mode 100644 index 000000000..007e59390 --- /dev/null +++ b/compile.txt @@ -0,0 +1,5 @@ +g++ -O3 minimal_mic.cpp \ + -I. -I./include -I./ggml/include -I./examples \ + ./build/src/libwhisper.so \ + -L/usr/local/cuda/lib64 -lcudart -lcublas \ + -lpthread -ldl -lm -lrt -o minimal_mic diff --git a/doubao.cpp b/doubao.cpp new file mode 100644 index 000000000..49217508a --- /dev/null +++ b/doubao.cpp @@ -0,0 +1,157 @@ +#include "whisper.h" +#include "common.h" + +#define MINIAUDIO_IMPLEMENTATION +#include "miniaudio.h" + +#include +#include +#include +#include +#include +#include + +// 全局原子变量控制录制状态(线程安全) +std::atomic is_recording(false); +// 音频缓冲区 +std::vector audio_buffer; + +// 音频回调:仅在录制状态时才采集数据 +void data_callback(ma_device* pDevice, void* pOutput, const void* pInput, ma_uint32 frameCount) { + if (!is_recording.load()) return; // 非录制状态直接返回,不采集数据 + const float* pInputFloat = (const float*)pInput; + if (pInputFloat == NULL) return; + + // 采集数据到缓冲区(限制最大录制时长为30秒,防止溢出) + const size_t max_frames = 16000 * 30; // 30秒 @ 16kHz + const size_t available = max_frames - audio_buffer.size(); + if (available == 0) return; // 缓冲区已满,停止采集 + + const size_t copy_frames = (frameCount > available) ? available : frameCount; + audio_buffer.insert(audio_buffer.end(), pInputFloat, pInputFloat + copy_frames); +} + +// 提示信息函数 +void print_usage() { + printf("=============================================\n"); + printf("🎤 语音识别程序(精准录制版)\n"); + printf("操作说明:\n"); + printf(" 1. 按下【回车键】开始录制\n"); + printf(" 2. 说话完成后,再次按下【回车键】停止录制并识别\n"); + printf(" 3. 录制超过30秒会自动停止\n"); + printf(" 4. Ctrl+C 退出程序\n"); + printf("=============================================\n"); +} + +int main(int argc, char** argv) { + if (argc < 2) { + fprintf(stderr, "Usage: %s \n", argv[0]); + return 1; + } + + const char* model_path = argv[1]; + + // 1. 初始化 Whisper + struct whisper_context_params cparams = whisper_context_default_params(); + cparams.use_gpu = true; // 4050 显卡 + struct whisper_context* ctx = whisper_init_from_file_with_params(model_path, cparams); + if (!ctx) { + fprintf(stderr, "❌ 初始化Whisper模型失败\n"); + return 1; + } + + // 2. 初始化 Miniaudio(仅初始化设备,不立即采集) + ma_device_config deviceConfig = ma_device_config_init(ma_device_type_capture); + deviceConfig.capture.format = ma_format_f32; // Whisper 需要 float32 + deviceConfig.capture.channels = 1; // 单声道 + deviceConfig.sampleRate = 16000; // Whisper 硬指标 16kHz + deviceConfig.dataCallback = data_callback; + deviceConfig.pUserData = nullptr; // 不再传buffer,用全局变量 + + ma_device device; + if (ma_device_init(NULL, &deviceConfig, &device) != MA_SUCCESS) { + fprintf(stderr, "❌ 打开录音设备失败\n"); + whisper_free(ctx); + return -2; + } + + // 启动设备(但此时is_recording=false,不会采集数据) + if (ma_device_start(&device) != MA_SUCCESS) { + fprintf(stderr, "❌ 启动录音设备失败\n"); + ma_device_uninit(&device); + whisper_free(ctx); + return -3; + } + + print_usage(); + + while (true) { + // 第一步:等待用户按回车开始录制 + printf("\n👉 按下回车键开始录制...\n"); + getchar(); // 等待回车 + + // 开始录制 + is_recording.store(true); + audio_buffer.clear(); // 清空旧数据 + printf("🎙️ 正在录制(说话完成后按回车键停止,最长录制30秒)...\n"); + + // 等待用户停止录制(按回车)或超时30秒 + std::thread wait_thread([&]() { + getchar(); // 等待用户按回车停止 + is_recording.store(false); + }); + + // 超时控制(30秒) + auto start_time = std::chrono::steady_clock::now(); + while (is_recording.load()) { + auto now = std::chrono::steady_clock::now(); + auto duration = std::chrono::duration_cast(now - start_time).count(); + if (duration >= 30) { + printf("⏱️ 录制超时(30秒),自动停止\n"); + is_recording.store(false); + break; + } + std::this_thread::sleep_for(std::chrono::milliseconds(100)); // 避免CPU空转 + } + + wait_thread.join(); // 等待停止线程结束 + is_recording.store(false); // 确保录制停止 + + // 检查录制的数据量 + if (audio_buffer.empty()) { + printf("⚠️ 未采集到任何音频数据,请重新录制\n"); + continue; + } + + // 第二步:开始识别 + printf("🔍 正在识别...\n"); + whisper_full_params wparams = whisper_full_default_params(WHISPER_SAMPLING_GREEDY); + wparams.language = "zh"; + wparams.n_threads = 12; + wparams.print_progress = false; + wparams.print_realtime = false; + + if (whisper_full(ctx, wparams, audio_buffer.data(), audio_buffer.size()) != 0) { + fprintf(stderr, "❌ 识别失败\n"); + continue; + } + + // 输出识别结果 + const int n_segments = whisper_full_n_segments(ctx); + if (n_segments == 0) { + printf("📝: 未识别到有效内容\n"); + } else { + printf("📝 识别结果:\n"); + for (int i = 0; i < n_segments; ++i) { + const char* text = whisper_full_get_segment_text(ctx, i); + printf(" %s\n", text); + } + } + } + + // 清理资源(实际中Ctrl+C会中断,这里是兜底) + ma_device_uninit(&device); + whisper_free(ctx); + return 0; +} + diff --git a/doubao_gpu.cpp b/doubao_gpu.cpp new file mode 100644 index 000000000..d045ad499 --- /dev/null +++ b/doubao_gpu.cpp @@ -0,0 +1,195 @@ +#include "whisper.h" +#include "common.h" + +#define MINIAUDIO_IMPLEMENTATION +#include "miniaudio.h" + +#include +#include +#include +#include +#include +#include + +// 全局原子变量控制录制状态(线程安全) +std::atomic is_recording(false); +// 音频缓冲区 +std::vector audio_buffer; + +// 音频回调:仅在录制状态时才采集数据 +void data_callback(ma_device* pDevice, void* pOutput, const void* pInput, ma_uint32 frameCount) { + if (!is_recording.load()) return; // 非录制状态直接返回,不采集数据 + const float* pInputFloat = (const float*)pInput; + if (pInputFloat == NULL) return; + + // 采集数据到缓冲区(限制最大录制时长为30秒,防止溢出) + const size_t max_frames = 16000 * 30; // 30秒 @ 16kHz + const size_t available = max_frames - audio_buffer.size(); + if (available == 0) return; // 缓冲区已满,停止采集 + + const size_t copy_frames = (frameCount > available) ? available : frameCount; + audio_buffer.insert(audio_buffer.end(), pInputFloat, pInputFloat + copy_frames); +} + +// 提示信息函数 +void print_usage() { + printf("=============================================\n"); + printf("🎤 语音识别程序(精准录制版)\n"); + printf("操作说明:\n"); + printf(" 1. 按下【回车键】开始录制\n"); + printf(" 2. 说话完成后,再次按下【回车键】停止录制并识别\n"); + printf(" 3. 录制超过30秒会自动停止\n"); + printf(" 4. Ctrl+C 退出程序\n"); + printf("=============================================\n"); +} + +// 适配旧版本的GPU状态提示(不依赖新函数) +void check_gpu_status() { + printf("🔍 GPU加速配置说明...\n"); + printf(" 当前已启用GPU加速(use_gpu = true)\n"); + printf(" ✅ 如果编译时链接了CUDA库,模型会自动使用GPU\n"); + printf(" ❌ 如果识别速度很慢,说明实际使用CPU运行\n"); + printf(" 验证方法:观察识别耗时,GPU版本比CPU快5-10倍\n"); +} + +int main(int argc, char** argv) { + if (argc < 2) { + fprintf(stderr, "Usage: %s \n", argv[0]); + return 1; + } + + const char* model_path = argv[1]; + + // GPU状态提示(适配旧版本) + check_gpu_status(); + + // 1. 初始化 Whisper(仅保留旧版本支持的参数) + struct whisper_context_params cparams = whisper_context_default_params(); + cparams.use_gpu = true; // 启用GPU(旧版本核心参数) + // 移除use_gpu_fp16和gpu_device(旧版本没有这些字段) + + printf("\n🚀 正在加载模型:%s\n", model_path); + struct whisper_context* ctx = whisper_init_from_file_with_params(model_path, cparams); + if (!ctx) { + fprintf(stderr, "❌ 初始化Whisper模型失败\n"); + return 1; + } + + // 旧版本没有whisper_is_using_gpu,改用间接提示 + printf("✅ 模型加载成功!\n"); + printf(" 📌 若识别速度快(几秒内完成)= GPU运行\n"); + printf(" 📌 若识别速度慢(十几秒/分钟)= CPU运行\n"); + + // 2. 初始化 Miniaudio(仅初始化设备,不立即采集) + ma_device_config deviceConfig = ma_device_config_init(ma_device_type_capture); + deviceConfig.capture.format = ma_format_f32; // Whisper 需要 float32 + deviceConfig.capture.channels = 1; // 单声道 + deviceConfig.sampleRate = 16000; // Whisper 硬指标 16kHz + deviceConfig.dataCallback = data_callback; + deviceConfig.pUserData = nullptr; + + ma_device device; + if (ma_device_init(NULL, &deviceConfig, &device) != MA_SUCCESS) { + fprintf(stderr, "❌ 打开录音设备失败\n"); + whisper_free(ctx); + return -2; + } + + // 启动设备(但此时is_recording=false,不会采集数据) + if (ma_device_start(&device) != MA_SUCCESS) { + fprintf(stderr, "❌ 启动录音设备失败\n"); + ma_device_uninit(&device); + whisper_free(ctx); + return -3; + } + + print_usage(); + + while (true) { + // 第一步:等待用户按回车开始录制 + printf("\n👉 按下回车键开始录制...\n"); + getchar(); // 等待回车 + + // 开始录制 + is_recording.store(true); + audio_buffer.clear(); // 清空旧数据 + printf("🎙️ 正在录制(说话完成后按回车键停止,最长录制30秒)...\n"); + + // 等待用户停止录制(按回车)或超时30秒 + std::thread wait_thread([&]() { + getchar(); // 等待用户按回车停止 + is_recording.store(false); + }); + + // 超时控制(30秒) + auto start_time = std::chrono::steady_clock::now(); + while (is_recording.load()) { + auto now = std::chrono::steady_clock::now(); + auto duration = std::chrono::duration_cast(now - start_time).count(); + if (duration >= 30) { + printf("⏱️ 录制超时(30秒),自动停止\n"); + is_recording.store(false); + break; + } + std::this_thread::sleep_for(std::chrono::milliseconds(100)); // 避免CPU空转 + } + + wait_thread.join(); // 等待停止线程结束 + is_recording.store(false); // 确保录制停止 + + // 检查录制的数据量 + if (audio_buffer.empty()) { + printf("⚠️ 未采集到任何音频数据,请重新录制\n"); + continue; + } + + // 第二步:开始识别(优化识别参数提升精度) + printf("🔍 正在识别...\n"); + // 记录识别开始时间(用于判断GPU/CPU) + auto recognize_start = std::chrono::steady_clock::now(); + + whisper_full_params wparams = whisper_full_default_params(WHISPER_SAMPLING_GREEDY); + wparams.language = "zh"; + wparams.n_threads = 12; // 根据CPU核心数调整 + wparams.print_progress = false; + wparams.print_realtime = false; + + // 精度优化参数(旧版本也支持) + wparams.temperature = 0.0; // 降低随机性,提升稳定性 + wparams.max_len = 0; // 不限制输出长度 + wparams.translate = false; // 不翻译,直接识别 + wparams.no_context = true; // 不使用上下文,避免干扰 + + if (whisper_full(ctx, wparams, audio_buffer.data(), audio_buffer.size()) != 0) { + fprintf(stderr, "❌ 识别失败\n"); + continue; + } + + // 计算识别耗时(判断GPU/CPU) + auto recognize_end = std::chrono::steady_clock::now(); + auto recognize_duration = std::chrono::duration_cast(recognize_end - recognize_start).count(); + printf("⏱️ 识别耗时:%.2f 秒\n", recognize_duration / 1000.0); + if (recognize_duration < 5000) { + printf(" 🎯 识别速度快,应该是GPU在运行!\n"); + } else { + printf(" ⚠️ 识别速度慢,可能是CPU在运行!\n"); + } + + // 输出识别结果 + const int n_segments = whisper_full_n_segments(ctx); + if (n_segments == 0) { + printf("📝: 未识别到有效内容\n"); + } else { + printf("📝 识别结果:\n"); + for (int i = 0; i < n_segments; ++i) { + const char* text = whisper_full_get_segment_text(ctx, i); + printf(" %s\n", text); + } + } + } + + // 清理资源 + ma_device_uninit(&device); + whisper_free(ctx); + return 0; +} diff --git a/doubao_mic.cpp b/doubao_mic.cpp new file mode 100644 index 000000000..1c1946d94 --- /dev/null +++ b/doubao_mic.cpp @@ -0,0 +1,299 @@ +#include "whisper.h" +#include +#include +#include +#include +#include +#include +#include +#include + +// ====================== 1. 枚举并选择麦克风设备(纯PortAudio原生实现) ====================== +void enumerate_audio_devices() { + PaError err = Pa_Initialize(); + if (err != paNoError) { + fprintf(stderr, "❌ PortAudio初始化失败: %s\n", Pa_GetErrorText(err)); + return; + } + + int numDevices = Pa_GetDeviceCount(); + printf("\n📜 系统可用麦克风设备列表:\n"); + printf("=============================================\n"); + for (int i = 0; i < numDevices; i++) { + const PaDeviceInfo* pInfo = Pa_GetDeviceInfo(i); + // 只显示输入设备(麦克风,至少1个输入声道) + if (pInfo->maxInputChannels > 0) { + printf("🔧 设备ID: %d | 名称: %s\n", i, pInfo->name); + printf(" 最大输入声道: %d | 默认采样率: %.1f Hz\n", + pInfo->maxInputChannels, pInfo->defaultSampleRate); + printf("---------------------------------------------\n"); + } + } + printf("=============================================\n\n"); + + Pa_Terminate(); +} + +int select_mic_device() { + int selected_id = -1; + printf("👉 请输入你要使用的麦克风设备ID(比如苹果耳机对应的ID):"); + std::cin >> selected_id; + + // 验证设备ID有效性 + PaError err = Pa_Initialize(); + if (err != paNoError) { + fprintf(stderr, "❌ PortAudio初始化失败: %s\n", Pa_GetErrorText(err)); + return -1; + } + + int numDevices = Pa_GetDeviceCount(); + if (selected_id < 0 || selected_id >= numDevices) { + fprintf(stderr, "❌ 设备ID无效!请输入列表中的有效ID\n"); + Pa_Terminate(); + return -1; + } + + const PaDeviceInfo* pInfo = Pa_GetDeviceInfo(selected_id); + if (pInfo->maxInputChannels == 0) { + fprintf(stderr, "❌ 选择的设备不是麦克风(无输入声道)!\n"); + Pa_Terminate(); + return -1; + } + + printf("\n✅ 已选择麦克风:\n"); + printf(" ID: %d | 名称: %s\n", selected_id, pInfo->name); + printf(" 采样率: %.1f Hz | 声道数: %d\n\n", + pInfo->defaultSampleRate, pInfo->maxInputChannels); + + Pa_Terminate(); + return selected_id; +} + +// ====================== 2. 音频采集函数(纯PortAudio原生实现) ====================== +int audio_record(short* buffer, int buffer_size, int sample_rate, int channels, int max_seconds, int device_id) { + PaError err; + PaStream* stream; + PaStreamParameters input_params; + + // 初始化PortAudio + err = Pa_Initialize(); + if (err != paNoError) { + fprintf(stderr, "❌ PortAudio初始化失败: %s\n", Pa_GetErrorText(err)); + return -1; + } + + // 配置输入参数(指定麦克风设备ID) + input_params.device = device_id; + input_params.channelCount = channels; + input_params.sampleFormat = paInt16; // 16位深(Whisper要求) + input_params.suggestedLatency = Pa_GetDeviceInfo(device_id)->defaultLowInputLatency; + input_params.hostApiSpecificStreamInfo = NULL; + + // 打开音频流 + err = Pa_OpenStream( + &stream, + &input_params, + NULL, // 无输出 + sample_rate, + 1024, // 缓冲区大小 + paClipOff, // 关闭裁剪 + NULL, // 无回调 + NULL + ); + + if (err != paNoError) { + fprintf(stderr, "❌ 打开音频流失败: %s\n", Pa_GetErrorText(err)); + Pa_Terminate(); + return -1; + } + + // 开始录制 + err = Pa_StartStream(stream); + if (err != paNoError) { + fprintf(stderr, "❌ 开始录制失败: %s\n", Pa_GetErrorText(err)); + Pa_CloseStream(stream); + Pa_Terminate(); + return -1; + } + + printf("🎙️ 录制中(按回车键停止,最长%d秒)...\n", max_seconds); + int total_samples = 0; + time_t start_time = time(NULL); + + // 录制逻辑:要么按回车停止,要么超时停止 + while (1) { + // 读取音频数据 + int samples_to_read = buffer_size - total_samples; + if (samples_to_read <= 0) break; + + err = Pa_ReadStream(stream, buffer + total_samples, 1024); + if (err != paNoError) { + fprintf(stderr, "❌ 读取音频失败: %s\n", Pa_GetErrorText(err)); + break; + } + + total_samples += 1024; + + // 超时检查(max_seconds秒) + if (difftime(time(NULL), start_time) >= max_seconds) { + printf("\n⏰ 录制超时(%d秒),自动停止\n", max_seconds); + break; + } + + // 检查是否按了回车 + if (std::cin.rdbuf()->in_avail() > 0) { + getchar(); + printf("\n🛑 用户停止录制\n"); + break; + } + } + + // 停止录制 + Pa_StopStream(stream); + Pa_CloseStream(stream); + Pa_Terminate(); + + return total_samples; +} + +// ====================== 3. 新增:short转float(Whisper要求) ====================== +void convert_short_to_float(const short* src, float* dst, int count) { + // 16位short的范围是[-32768, 32767],归一化到float的[-1.0, 1.0] + for (int i = 0; i < count; i++) { + dst[i] = static_cast(src[i]) / 32768.0f; + } +} + +// ====================== 4. 主函数(修正数据类型转换) ====================== +int main(int argc, char **argv) { + // 检查参数 + if (argc < 2) { + fprintf(stderr, "用法: %s 模型文件路径(如 ./models/ggml-medium.bin)\n", argv[0]); + return 1; + } + const char* model_path = argv[1]; + + // 步骤1:枚举并选择麦克风 + enumerate_audio_devices(); + int mic_device_id = select_mic_device(); + if (mic_device_id < 0) { + fprintf(stderr, "❌ 麦克风选择失败,程序退出\n"); + return 1; + } + + // 步骤2:GPU加速配置说明 + printf("\n🔍 GPU加速配置说明...\n"); + printf(" 当前已启用GPU加速(use_gpu = true)\n"); + printf(" ✅ 如果编译时链接了CUDA库,模型会自动使用GPU\n"); + printf(" ❌ 如果识别速度很慢,说明实际使用CPU运行\n"); + printf(" 验证方法:观察识别耗时,GPU版本比CPU快5-10倍\n\n"); + + // 步骤3:加载Whisper模型(启用GPU) + printf("🚀 正在加载模型:%s\n", model_path); + struct whisper_context_params cparams = whisper_context_default_params(); + cparams.use_gpu = true; + cparams.gpu_device = 0; + + struct whisper_context* ctx = whisper_init_from_file_with_params(model_path, cparams); + if (!ctx) { + fprintf(stderr, "❌ 加载模型失败: %s\n", model_path); + return 1; + } + + // 打印模型信息 + whisper_print_system_info(); + printf("✅ 模型加载成功!\n"); + printf(" 📌 若识别速度快(几秒内完成)= GPU运行\n"); + printf(" 📌 若识别速度慢(十几秒/分钟)= CPU运行\n"); + printf("=============================================\n"); + printf("🎤 语音识别程序(指定麦克风版)\n"); + printf("操作说明:\n"); + printf(" 1. 按下【回车键】开始录制\n"); + printf(" 2. 说话完成后,再次按下【回车键】停止录制并识别\n"); + printf(" 3. 录制超过30秒会自动停止\n"); + printf(" 4. Ctrl+C 退出程序\n"); + printf("=============================================\n\n"); + + // 步骤4:准备音频缓冲区 + const int sample_rate = 16000; // Whisper标准采样率 + const int channels = 1; // 单声道 + const int max_seconds = 30; // 最长录制30秒 + const int buffer_size = sample_rate * channels * max_seconds; + + // 原始音频缓冲区(short类型) + short* buffer_short = (short*)malloc(buffer_size * sizeof(short)); + // Whisper输入缓冲区(float类型) + float* buffer_float = (float*)malloc(buffer_size * sizeof(float)); + + if (!buffer_short || !buffer_float) { + fprintf(stderr, "❌ 分配音频缓冲区失败\n"); + free(buffer_short); + free(buffer_float); + whisper_free(ctx); + return 1; + } + + // 步骤5:等待用户开始录制 + printf("👉 按下回车键开始录制...\n"); + getchar(); + + // 步骤6:录制音频(指定选择的麦克风) + int samples_read = audio_record(buffer_short, buffer_size, sample_rate, channels, max_seconds, mic_device_id); + if (samples_read <= 0) { + fprintf(stderr, "❌ 录制音频失败\n"); + free(buffer_short); + free(buffer_float); + whisper_free(ctx); + return 1; + } + + // 步骤7:关键修正:short转float(Whisper要求) + convert_short_to_float(buffer_short, buffer_float, samples_read); + + // 步骤8:语音识别(传入float缓冲区) + printf("\n🔍 正在识别...\n"); + clock_t start = clock(); + + struct whisper_full_params wparams = whisper_full_default_params(WHISPER_SAMPLING_GREEDY); + wparams.language = "zh"; // 中文识别 + wparams.translate = false; + wparams.print_special = false; + wparams.print_progress = false; + wparams.print_realtime = false; + wparams.print_timestamps = false; + + // 传入float类型的buffer_float,而非short类型的buffer_short + if (whisper_full(ctx, wparams, buffer_float, samples_read) != 0) { + fprintf(stderr, "❌ 识别音频失败\n"); + free(buffer_short); + free(buffer_float); + whisper_free(ctx); + return 1; + } + + // 步骤9:输出结果 + clock_t end = clock(); + double elapsed = (double)(end - start) / CLOCKS_PER_SEC; + printf("⏱️ 识别耗时:%.2f 秒\n", elapsed); + + if (elapsed < 5.0) { + printf(" 🎯 识别速度快,应该是GPU在运行!\n"); + } else { + printf(" ⚠️ 识别速度慢,当前使用CPU运行(需编译CUDA版本)\n"); + } + + printf("📝 识别结果:\n "); + const int n_segments = whisper_full_n_segments(ctx); + for (int i = 0; i < n_segments; i++) { + const char* text = whisper_full_get_segment_text(ctx, i); + printf("%s\n ", text); + } + printf("\n"); + + // 步骤10:清理资源 + free(buffer_short); + free(buffer_float); + whisper_free(ctx); + + return 0; +} diff --git a/download.txt b/download.txt new file mode 100644 index 000000000..38cd46030 --- /dev/null +++ b/download.txt @@ -0,0 +1,2 @@ +export HF_ENDPOINT=https://hf-mirror.com +hf download ggerganov/whisper.cpp ggml-medium.bin --local-dir ./models diff --git a/minimal_mic.cpp b/minimal_mic.cpp new file mode 100644 index 000000000..de03ee67f --- /dev/null +++ b/minimal_mic.cpp @@ -0,0 +1,82 @@ +#include "whisper.h" +#include "common.h" + +#define MINIAUDIO_IMPLEMENTATION +#include "miniaudio.h" + +#include +#include +#include + +// 音频回调:将采集到的数据存入 buffer +void data_callback(ma_device* pDevice, void* pOutput, const void* pInput, ma_uint32 frameCount) { + std::vector* pBuffer = (std::vector*)pDevice->pUserData; + const float* pInputFloat = (const float*)pInput; + if (pInputFloat == NULL) return; + + pBuffer->insert(pBuffer->end(), pInputFloat, pInputFloat + frameCount); + // 保持 buffer 在最近 10 秒以内,防止内存溢出 + if (pBuffer->size() > 16000 * 10) { + pBuffer->erase(pBuffer->begin(), pBuffer->begin() + (pBuffer->size() - 16000 * 10)); + } +} + +int main(int argc, char** argv) { + if (argc < 2) { + fprintf(stderr, "Usage: %s \n", argv[0]); + return 1; + } + + const char* model_path = argv[1]; + + // 1. 初始化 Whisper + struct whisper_context_params cparams = whisper_context_default_params(); + cparams.use_gpu = true; // 你的 4050 显卡 + struct whisper_context* ctx = whisper_init_from_file_with_params(model_path, cparams); + if (!ctx) return 1; + + // 2. 初始化 Miniaudio + std::vector audio_buffer; + ma_device_config deviceConfig = ma_device_config_init(ma_device_type_capture); + deviceConfig.capture.format = ma_format_f32; // Whisper 需要 float32 + deviceConfig.capture.channels = 1; // 单声道 + deviceConfig.sampleRate = 16000; // Whisper 硬指标 16kHz + deviceConfig.dataCallback = data_callback; + deviceConfig.pUserData = &audio_buffer; + + ma_device device; + if (ma_device_init(NULL, &deviceConfig, &device) != MA_SUCCESS) { + fprintf(stderr, "Failed to open capture device.\n"); + return -2; + } + + ma_device_start(&device); + printf("🎤 录音中... 请说话 (按回车键进行单次识别,Ctrl+C 退出)\n"); + + while (true) { + getchar(); // 等待用户敲回车触发识别 + + printf("正在识别...\n"); + + whisper_full_params wparams = whisper_full_default_params(WHISPER_SAMPLING_GREEDY); + wparams.language = "zh"; + wparams.n_threads = 12; + wparams.print_progress = false; + + if (whisper_full(ctx, wparams, audio_buffer.data(), audio_buffer.size()) != 0) { + fprintf(stderr, "识别失败\n"); + continue; + } + + const int n_segments = whisper_full_n_segments(ctx); + for (int i = 0; i < n_segments; ++i) { + const char* text = whisper_full_get_segment_text(ctx, i); + printf("📝: %s\n", text); + } + audio_buffer.clear(); // 清空,准备下一轮 + } + + ma_device_uninit(&device); + whisper_free(ctx); + return 0; +} diff --git a/run.txt b/run.txt new file mode 100644 index 000000000..43c761ff1 --- /dev/null +++ b/run.txt @@ -0,0 +1,2 @@ +export LD_LIBRARY_PATH=./build/src +./minimal_mic ./models/ggml-small.bin