gemini try to solve traditional chinese issue

This commit is contained in:
nick huang 2026-03-17 21:56:29 +08:00
parent 8eb7806471
commit 26ad600783
1 changed files with 68 additions and 52 deletions

View File

@ -13,28 +13,24 @@
#include <sys/select.h>
// =============================================
// 1. 工程常量定义 (拒绝 Magic Numbers)
// 工程常量:全局统一管理,杜绝 Magic Numbers
// =============================================
struct RecordingConfig {
static constexpr int SAMPLE_RATE = 16000;
static constexpr int PROGRESS_REFRESH_MS = 100; // 进度条刷新频率
static constexpr int UI_WAIT_MS = 10; // 循环等待步长
static constexpr int INPUT_CHECK_MS = 20; // 输入检测超时
static constexpr int POST_STOP_BUFFER_MS = 1500; // 停止后的平滑采样缓冲
static constexpr int CLOCK_TOLERANCE_MS = 200; // 系统时钟容差(确保跳到目标秒数)
static constexpr int PROGRESS_MS = 100; // 进度条刷新间隔
static constexpr int UI_LOOP_MS = 10; // UI循环步长
static constexpr int SELECT_TIMEOUT_MS = 20; // select 超时
static constexpr int SMOOTH_FINISH_MS = 1500; // 平滑收尾时长 (1.5秒)
static constexpr int CLOCK_TOLERANCE_MS = 300; // 边界容差,确保显示完整
};
// 全局状态管理
std::atomic<bool> is_recording(false);
std::atomic<bool> exit_program(false);
std::atomic<int> recorded_seconds(0);
std::vector<float> audio_buffer;
std::mutex buffer_mutex;
int g_timeout_setting = 30; // 用户设定的超时秒数
int g_timeout_limit = 30; // 默认 30s
// =============================================
// 系统辅助函数
// =============================================
void signal_handler(int sig) {
if (sig == SIGINT) {
exit_program.store(true);
@ -43,15 +39,18 @@ void signal_handler(int sig) {
}
}
bool check_input_non_blocking(int timeout_ms = RecordingConfig::INPUT_CHECK_MS) {
// 非阻塞输入检测
bool check_stdin_ready(int timeout_ms = RecordingConfig::SELECT_TIMEOUT_MS) {
fd_set fds; FD_ZERO(&fds); FD_SET(STDIN_FILENO, &fds);
struct timeval tv = {0, timeout_ms * 1000};
return select(STDIN_FILENO + 1, &fds, NULL, NULL, &tv) > 0;
}
// =============================================
// 音频回调
// =============================================
void clear_stdin() {
while (check_stdin_ready(0)) getchar();
}
// 音频采集回调
void data_callback(ma_device* pDevice, void* pOutput, const void* pInput, ma_uint32 frameCount) {
if (!is_recording.load() || pInput == NULL) return;
std::lock_guard<std::mutex> lock(buffer_mutex);
@ -59,96 +58,113 @@ void data_callback(ma_device* pDevice, void* pOutput, const void* pInput, ma_uin
recorded_seconds.store(static_cast<int>(audio_buffer.size() / (float)RecordingConfig::SAMPLE_RATE));
}
// =============================================
// 主逻辑
// =============================================
int main(int argc, char** argv) {
signal(SIGINT, signal_handler);
if (argc < 2) return 1;
if (argc >= 3) g_timeout_setting = atoi(argv[2]);
if (argc < 2) {
printf("用法: %s <模型路径> [超时秒数]\n", argv[0]);
return 1;
}
if (argc >= 3) g_timeout_limit = atoi(argv[2]);
// 初始化 Whisper
struct whisper_context_params cparams = whisper_context_default_params();
cparams.use_gpu = true;
struct whisper_context* ctx = whisper_init_from_file_with_params(argv[1], cparams);
// 麦克风设备枚举与选择
// 设备枚举与选择
ma_context context; ma_context_init(NULL, 0, NULL, &context);
ma_device_info* pCapInfos; ma_uint32 capCount;
ma_context_get_devices(&context, NULL, NULL, &pCapInfos, &capCount);
for (ma_uint32 i = 0; i < capCount; ++i) printf("[%u] %s\n", i, pCapInfos[i].name);
printf("👉 请输入设备 ID: ");
ma_uint32 dev_id; scanf("%u", &dev_id);
while (getchar() != '\n');
printf("\n📜 可用麦克风列表:\n");
for (ma_uint32 i = 0; i < capCount; ++i) printf(" [%u] %s\n", i, pCapInfos[i].name);
printf("👉 请输入设备 ID (默认5): ");
ma_uint32 dev_id = 5;
if(scanf("%u", &dev_id) != 1) dev_id = 5;
clear_stdin();
ma_device_config devCfg = ma_device_config_init(ma_device_type_capture);
devCfg.capture.format = ma_format_f32;
devCfg.capture.channels = 1;
devCfg.sampleRate = RecordingConfig::SAMPLE_RATE;
devCfg.dataCallback = data_callback;
devCfg.capture.format = ma_format_f32; devCfg.capture.channels = 1;
devCfg.sampleRate = RecordingConfig::SAMPLE_RATE; devCfg.dataCallback = data_callback;
if (dev_id < capCount) devCfg.capture.pDeviceID = &pCapInfos[dev_id].id;
ma_device device; ma_device_init(&context, &devCfg, &device);
ma_device_start(&device);
while (!exit_program.load()) {
printf("\n[回车] 录制 | [回车] 停止\n👉 等待指令...");
while (!check_input_non_blocking(50) && !exit_program.load());
printf("\n=============================================\n");
printf("🎙️ 操作提示 (自动断开设置: %d 秒):\n", g_timeout_limit);
printf(" ▶ [回车键] : 开始录制\n");
printf(" ■ [回车键] : 停止录制 (含 %.1f 秒补录)\n", (float)RecordingConfig::SMOOTH_FINISH_MS/1000.0f);
printf("=============================================\n");
printf("👉 等待指令...");
fflush(stdout);
while (!check_stdin_ready(100) && !exit_program.load());
if (exit_program.load()) break;
while (check_input_non_blocking(0)) getchar();
clear_stdin();
{ std::lock_guard<std::mutex> lock(buffer_mutex); audio_buffer.clear(); }
recorded_seconds.store(0);
is_recording.store(true);
auto start_time = std::chrono::steady_clock::now();
// 进度显示线程
printf("\n🎙️ 录制中...\n");
std::thread progress_thread([&]() {
while (is_recording.load()) {
printf("\r📊 进度: %d 秒 ", recorded_seconds.load());
fflush(stdout);
std::this_thread::sleep_for(std::chrono::milliseconds(RecordingConfig::PROGRESS_REFRESH_MS));
std::this_thread::sleep_for(std::chrono::milliseconds(RecordingConfig::PROGRESS_MS));
}
});
bool stop_triggered = false;
while (!stop_triggered && !exit_program.load()) {
bool trigger_stop = false;
while (!trigger_stop && !exit_program.load()) {
auto now = std::chrono::steady_clock::now();
auto elapsed = std::chrono::duration_cast<std::chrono::milliseconds>(now - start_time).count();
if (check_input_non_blocking(RecordingConfig::UI_WAIT_MS)) {
if (getchar() == '\n') {
printf("\n🛑 手动停止 (进入平滑刷新模式)...");
stop_triggered = true;
if (check_stdin_ready(RecordingConfig::UI_LOOP_MS)) {
if (getchar() == '\n') {
printf("\n🛑 手动停止,正在收尾以确保不丢字...");
trigger_stop = true;
}
}
// 修正边界:加上 CLOCK_TOLERANCE_MS 确保进度条在视觉上能显示到设定的秒数
else if (elapsed >= (g_timeout_setting * 1000 + RecordingConfig::CLOCK_TOLERANCE_MS)) {
printf("\r📊 进度: %d 秒 ", g_timeout_setting); // 强制补完最后一秒显示
printf("\n⏱️ 超时停止 (%d秒进入平滑刷新模式)...", g_timeout_setting);
stop_triggered = true;
// 关键n + 容差,确保进度条能显示出最后那一秒
else if (elapsed >= (g_timeout_limit * 1000 + RecordingConfig::CLOCK_TOLERANCE_MS)) {
printf("\r📊 进度: %d 秒 ", g_timeout_limit);
printf("\n⏱️ 时间已到 (%d秒),正在自动收尾...", g_timeout_limit);
trigger_stop = true;
}
}
// 平滑刷新:等待硬件缓冲区数据入库,防止丢字
std::this_thread::sleep_for(std::chrono::milliseconds(RecordingConfig::POST_STOP_BUFFER_MS));
// 执行平滑刷新
std::this_thread::sleep_for(std::chrono::milliseconds(RecordingConfig::SMOOTH_FINISH_MS));
is_recording.store(false);
if (progress_thread.joinable()) progress_thread.join();
// 识别逻辑
std::vector<float> captured;
{ std::lock_guard<std::mutex> lock(buffer_mutex); captured = audio_buffer; }
printf("\n🔍 识别 (音频长: %.2fs)...", (float)captured.size()/RecordingConfig::SAMPLE_RATE);
printf("\n🔍 正在识别 (音频长: %.2fs)...", (float)captured.size()/RecordingConfig::SAMPLE_RATE);
whisper_full_params wparams = whisper_full_default_params(WHISPER_SAMPLING_GREEDY);
wparams.language = "zh";
// 2. 【核心修正】注入简体中文引导词,强制模型输出简体
// "以下是普通话的句子。" 作为一个初始提示Prompt
wparams.initial_prompt = "以下是普通话的句子,使用简体中文。";
// 3. 翻译控制(确保不开启翻译模式)
wparams.translate = false;
wparams.n_threads = 4;
whisper_full(ctx, wparams, captured.data(), captured.size());
int n_segments = whisper_full_n_segments(ctx);
for (int i = 0; i < n_segments; ++i) printf("\n📝 %s", whisper_full_get_segment_text(ctx, i));
printf("\n📝 识别结果:");
for (int i = 0; i < n_segments; ++i) {
printf("\n %s", whisper_full_get_segment_text(ctx, i));
}
printf("\n");
}
ma_device_uninit(&device);
ma_context_uninit(&context);
whisper_free(ctx);
return 0;
}