From 32f90908ddf4503c3f272a50650e2290cd895fdd Mon Sep 17 00:00:00 2001 From: xiaosheng <73678111+xiaoshengbao@users.noreply.github.com> Date: Mon, 21 Sep 2026 22:58:19 +0800 Subject: [PATCH 1/4] =?UTF-8?q?feat(stt):=20=E6=8E=A5=E5=85=A5=E5=88=86?= =?UTF-8?q?=E6=A1=A3=E8=AF=AD=E9=9F=B3=E6=A8=A1=E5=9E=8B=E5=B9=B6=E5=AE=8C?= =?UTF-8?q?=E6=88=90=E6=9C=AC=E6=9C=BA=E5=8A=9F=E8=83=BD=E9=AA=8C=E6=94=B6?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .gitignore | 4 + desktop/package.json | 2 + desktop/scripts/build-backend.cjs | 11 + desktop/src/main.cjs | 4 +- desktop/tests/native-core-runtime.test.cjs | 3 +- docs/stt-benchmark-2026-09-21.md | 83 +++ docs/stt-integration-2026-09-21.md | 79 +++ docs/stt-model-upgrade-plan-2026-09-21.md | 116 ++++ frontend/components/SettingsDialog.vue | 44 +- frontend/components/chat/MessageContent.vue | 2 +- .../chat/VoiceTranscriptionSidebar.vue | 18 +- frontend/composables/chat/useChatMessages.js | 8 +- frontend/tests/voice-model-settings.test.mjs | 2 +- .../tests/voice-transcription-sidebar.test.js | 24 +- pyproject.toml | 15 + src/wechat_decrypt_tool/asr_backends.py | 178 ++++++ src/wechat_decrypt_tool/asr_models.py | 77 +++ src/wechat_decrypt_tool/asr_worker.py | 162 ++++++ .../resources/voice_models.json | 146 +++++ src/wechat_decrypt_tool/routers/chat_media.py | 14 +- .../voice_transcription.py | 196 ++++++- tests/test_asr_upgrade.py | 196 +++++++ tests/test_voice_transcription_manager.py | 5 +- tools/benchmark_stt_local.py | 230 ++++++++ uv.lock | 541 +++++++++++++++++- 25 files changed, 2083 insertions(+), 77 deletions(-) create mode 100644 docs/stt-benchmark-2026-09-21.md create mode 100644 docs/stt-integration-2026-09-21.md create mode 100644 docs/stt-model-upgrade-plan-2026-09-21.md create mode 100644 src/wechat_decrypt_tool/asr_backends.py create mode 100644 src/wechat_decrypt_tool/asr_models.py create mode 100644 src/wechat_decrypt_tool/asr_worker.py create mode 100644 src/wechat_decrypt_tool/resources/voice_models.json create mode 100644 tests/test_asr_upgrade.py create mode 100644 tools/benchmark_stt_local.py diff --git a/.gitignore b/.gitignore index 13098690..160c0887 100644 --- a/.gitignore +++ b/.gitignore @@ -121,3 +121,7 @@ pro-shots/ # 隔离验收数据库、真实聊天资料和本地状态快照不得进入版本库。 /tmp/deepagents-migration/ + +# 语音验收包含真实聊天音频、转写、模型权重及本地运行环境。 +/tmp/stt-benchmark-20260921/ +/tmp/stt-integration-20260921/ diff --git a/desktop/package.json b/desktop/package.json index fe18c30d..1d92093c 100644 --- a/desktop/package.json +++ b/desktop/package.json @@ -5,9 +5,11 @@ "main": "src/main.cjs", "scripts": { "dev": "node scripts/dev.cjs", + "dev:gpu": "cross-env WECHAT_TOOL_QWEN_GPU=1 node scripts/dev.cjs", "dev:static": "npm --prefix ../frontend run generate && cross-env WECHAT_TOOL_STATIC_UI=1 electron .", "build:ui": "npm --prefix ../frontend run generate && node scripts/copy-ui.cjs", "build:backend": "uv sync --no-editable --extra build --extra voice-transcription && node scripts/build-backend.cjs", + "build:backend:gpu": "uv sync --no-editable --extra build --extra voice-transcription --extra voice-transcription-gpu && node scripts/build-backend.cjs --qwen-gpu", "build:icon": "node scripts/build-icon.cjs", "build:mac:image-helper": "node scripts/build-macos-image-helper.cjs", "verify:mac:native": "node scripts/verify-macos-native.cjs --arch arm64 --require-host-arch", diff --git a/desktop/scripts/build-backend.cjs b/desktop/scripts/build-backend.cjs index de26e1eb..a0bb2797 100644 --- a/desktop/scripts/build-backend.cjs +++ b/desktop/scripts/build-backend.cjs @@ -652,11 +652,22 @@ function main() { "--collect-all", "opencc", "--collect-all", + "sherpa_onnx", + "--add-data", + pyInstallerAddData(path.join(repoRoot, "src/wechat_decrypt_tool/resources/voice_models.json"), "wechat_decrypt_tool/resources"), + "--collect-all", "watchfiles", ...aiPackagingArgs(repoRoot), entry, ]; + // CUDA/PyTorch 体积较大,仅在明确构建 Qwen GPU 版本时收集。 + if (process.argv.includes("--qwen-gpu")) { + args.splice(args.length - 1, 0, "--collect-all", "torch", "--collect-all", "transformers"); + } else { + args.splice(args.length - 1, 0, "--exclude-module", "torch", "--exclude-module", "transformers"); + } + if (process.platform === "win32") { args.splice( args.length - 1, diff --git a/desktop/src/main.cjs b/desktop/src/main.cjs index 810b3f57..9f6983f2 100644 --- a/desktop/src/main.cjs +++ b/desktop/src/main.cjs @@ -2174,7 +2174,9 @@ function startBackend() { // The desktop backend only needs runtime dependencies. Letting `uv run` // include the default dev group can block Electron startup on an unrelated // pytest/Pygments download before Python is even launched. - backendProc = spawn("uv", ["run", "--no-dev", "main.py"], { + const voiceExtras = ["--extra", "voice-transcription"]; + if (env.WECHAT_TOOL_QWEN_GPU === "1") voiceExtras.push("--extra", "voice-transcription-gpu"); + backendProc = spawn("uv", ["run", "--no-dev", ...voiceExtras, "main.py"], { cwd: repoRoot(), env, stdio: "inherit", diff --git a/desktop/tests/native-core-runtime.test.cjs b/desktop/tests/native-core-runtime.test.cjs index 0c0a0284..4d75f9c2 100644 --- a/desktop/tests/native-core-runtime.test.cjs +++ b/desktop/tests/native-core-runtime.test.cjs @@ -544,7 +544,8 @@ test("desktop startBackend clears legacy WCDB state and never starts the sidecar const startBackend = mainSource.match(/function startBackend\(\) \{([\s\S]*?)\n\}/)?.[1] || ""; assert.match(startBackend, /configureNativeCoreRuntime\(env\)/); assert.match(startBackend, /clearLegacyWcdbEnvironment\(env\)/); - assert.match(startBackend, /spawn\("uv", \["run", "--no-dev", "main\.py"\]/); + assert.match(startBackend, /spawn\("uv", \["run", "--no-dev", \.\.\.voiceExtras, "main\.py"\]/); + assert.match(startBackend, /voiceExtras = \["--extra", "voice-transcription"\]/); assert.match(startBackend, /PYTHONIOENCODING:\s*"utf-8"/); assert.match(startBackend, /WECHAT_TOOL_NODE_EXECUTABLE:\s*process\.execPath/); assert.match(startBackend, /WECHAT_TOOL_NODE_MODE:\s*"electron-run-as-node"/); diff --git a/docs/stt-benchmark-2026-09-21.md b/docs/stt-benchmark-2026-09-21.md new file mode 100644 index 00000000..273e8ec6 --- /dev/null +++ b/docs/stt-benchmark-2026-09-21.md @@ -0,0 +1,83 @@ +# 本机 STT 模型对照测试(2026-09-21) + +已完成 12 个运行配置。样本为项目已有的 20 条真实微信语音,共 202.54 秒、5 个会话。 + +## 对升级方案的修订 + +- **低配:优先 Zipformer CTC INT8。** 本批语音约 2.74 秒,Tiny 约 22.17 秒,快约 8.1 倍;与微信转写的差异率从 26.44% 降至 10.22%。CTC 进程峰值约 390 MiB,比 Tiny 的 279 MiB 高,不能宣称内存也更省。 +- **主流 CPU:Qwen3-ASR 0.6B ONNX INT4 值得接入。** 约 61.59 秒,相比 Medium CPU 的 163.63 秒快约 2.66 倍,差异率从 8.11% 降至 4.35%;代价是进程峰值从约 1.63 GiB 增至 3.65 GiB。建议先面向 16 GB 内存设备,4 GB 老电脑不以它作为默认。 +- **GPU:保留 Turbo 极速选项,增加 Qwen 质量优先选项。** Qwen 0.6B 为 38.57 秒、差异率 3.64%;1.7B 为 44.16 秒、2.94%;Turbo 为 7.87 秒、6.82%;原 Large v3 为 20.51 秒、6.93%。标准 Transformers 路径没有带来 GPU 提速,1.7B 是本批样本与参考最接近的配置,但耗时约为 Large v3 的 2.15 倍。 +- **不单列 Transducer 均衡档。** 本批结果 3.61 秒、差异率 11.16%,没有体现相对 CTC 的价值。20 条样本不足以判定它在其他数据上一定较差。 +- 这些结果足以调整工程接入顺序,不能证明真实准确率已经提升。下一步需要听原音校对参考,以及在真实低配设备上验收;当前只有本机 CPU 路径测试。 + +## 方法与适用边界 + +- 本机:Windows、Ryzen 5 5600X(6 核 12 线程)、32 GB 内存、RTX 4070 SUPER 12 GB,驱动 596.49。 +- 从 828 条已有微信转写的候选中,以固定种子 20260921 抽取。按 0.8–3、3–8、8–20、20–60 秒各选五条;实际最长时长见本地清单。仅有转写的语音会入选,存在选择偏差。 +- 统一解码为 16 kHz 单声道,所有模型使用完全相同的 WAV;音频和文本未发送到外部 ASR 服务。 +- 每模型独立进程、CPU 4 线程、单条串行,预热一条后随机顺序运行两轮。耗时是两轮热运行均值,包含特征提取和识别,不含 SILK 解码、下载、模型加载及应用界面开销。 +- 指定推理引擎使用 4 个 CPU 线程;这不等同于模拟低端 CPU,也不保证第三方库与整台电脑总共只有 4 个线程。部分测试期间后台仍在下载权重,速度作为本机初筛结果。 +- 原 Whisper 参数沿用项目:中文、beam_size=5、vad_filter=True、condition_on_previous_text=False。Qwen 指定中文、贪心解码;各模型按对应部署路径运行,并非相同架构/解码算法的微基准。 +- 差异率采用字符编辑距离 / 参考字符数;统一简繁、全半角、大小写,去空白和标点。参考是未人工校对的微信机器转写,因此该数值不能当作真实错误率,更不能用 100% 减去它声称准确率。 +- 不提供低配 CPU 的外推保证:本机仅在 CPU 路径上测试,并非 N100 或 4 GB 老电脑实测。样本量不足以证明所有方言、噪声和人名场景的整体提升。 +- 进程内存是 50 ms 采样的峰值工作集,包含推理库。Qwen 显存为 PyTorch 分配峰值;该值不等于整张显卡占用,不能与未测得显存的数据直接比较。 +- 测试用 faster-whisper 1.2.1 与项目一致;测试 CTranslate2 为 4.8.2,项目原环境为 4.8.1。其他依赖和模型 revision 已留档;因此这是沿用项目参数的独立环境对照,并非原应用端到端计时。 + +## 实测汇总 + +| 配置 | 203 秒音频耗时 | RTF↓ | 单条 P95 | 与微信转写差异率↓ | 峰值进程内存 | Qwen 分配显存峰值 | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| whisper-tiny-cpu | 22.17 s | 0.109 | 5.11 s | 26.44% | 279 MiB | 未测 | +| whisper-base-cpu | 21.95 s | 0.108 | 1.83 s | 17.04% | 407 MiB | 未测 | +| whisper-small-cpu | 57.80 s | 0.285 | 5.29 s | 10.58% | 593 MiB | 未测 | +| whisper-medium-cpu | 163.63 s | 0.808 | 13.75 s | 8.11% | 1674 MiB | 未测 | +| zipformer-ctc | 2.74 s | 0.014 | 0.34 s | 10.22% | 390 MiB | 未测 | +| zipformer-rnnt | 3.61 s | 0.018 | 0.43 s | 11.16% | 420 MiB | 未测 | +| qwen-onnx-cpu | 61.59 s | 0.304 | 7.87 s | 4.35% | 3737 MiB | 未测 | +| whisper-medium-cuda | 13.66 s | 0.067 | 1.52 s | 10.22% | 1607 MiB | 未测 | +| whisper-turbo-cuda | 7.87 s | 0.039 | 0.80 s | 6.82% | 1676 MiB | 未测 | +| whisper-large-v3-cuda | 20.51 s | 0.101 | 2.42 s | 6.93% | 3082 MiB | 未测 | +| qwen-06-cuda | 38.57 s | 0.190 | 4.61 s | 3.64% | 2401 MiB | 1.67 GiB | +| qwen-17-cuda | 44.16 s | 0.218 | 5.16 s | 2.94% | 4800 MiB | 4.01 GiB | + +RTF = 识别耗时 / 音频时长,越低越快。不同模型资源档位不代表质量必然单调上升。 + +每轮使用相同的随机顺序策略。差异率按第一轮输出计算;Tiny、Base 在两轮中分别有 2 条、1 条输出变化,其余配置是否变化可查原始记录。 + +## 部署中发现的问题 + +- Zipformer 的原始 ONNX 文件直接处理约 22 秒样本时,CTC 和 Transducer 均出现 Reshape 维度错误;两份原始失败日志保留在测试目录。表中结果是增加最多 15 秒、末段低能量切分后的配置,不能把它描述成无改动即可替换。 +- Qwen ONNX 上游示例硬编码的 system/user token ID 与下载模型的分词器不一致。测试入口改为用实际分词器编码提示模板;参考文本没有进入提示词。 +- ONNX CPU 候选是社区导出,官方 HF GPU 候选是另一套转换/运行路径,量化和预处理差异需要随发布版本固定。 +- Qwen CPU 特征提取与上游 PyTorch 实现做了同一条音频的数值核对:最大绝对差约 4.86e-5、平均绝对差约 3.28e-7;这验证了实现一致性,不等于量化质量验证。 + +## 冷启动参考 + +| 配置 | 加载阶段(含库导入) | 首条推理 | +| --- | ---: | ---: | +| whisper-tiny-cpu | 2.37 s | 1.43 s | +| whisper-base-cpu | 0.51 s | 6.21 s | +| whisper-small-cpu | 1.22 s | 1.83 s | +| whisper-medium-cpu | 3.13 s | 5.44 s | +| zipformer-ctc | 0.99 s | 0.02 s | +| zipformer-rnnt | 0.98 s | 0.05 s | +| qwen-onnx-cpu | 8.07 s | 2.49 s | +| whisper-medium-cuda | 2.11 s | 0.77 s | +| whisper-turbo-cuda | 2.16 s | 0.55 s | +| whisper-large-v3-cuda | 3.42 s | 0.71 s | +| qwen-06-cuda | 9.76 s | 1.24 s | +| qwen-17-cuda | 13.30 s | 1.07 s | + +首条采用同一短语音;未清空 Windows 文件缓存,因此不能视作重启电脑后的磁盘冷启动。 + +## 复现与证据 + +- 入口:`tools/benchmark_stt_local.py`。测试资产、环境版本、模型 revision、逐条结果位于 `tmp/stt-benchmark-20260921/`。 +- `summary.json` 不含聊天原文;`manifest.private.json`、`results/*.private.json` 和 `review.private.html` 含本地语音或转写,不纳入公开报告。 +- 抽样清单记录 WAV 的 SHA-256;全部 20 条音频、10 组模型资产的大小及 LFS 哈希已重新核对。12 个运行配置共 480 条推理记录,通过独立动态规划编辑距离复算;记录见 `verification.json`。 +- `model-lock.json` 记录实际下载的仓库、revision 和文件清单;`requirements-lock.txt` 留存测试环境依赖。项目 Turbo 的原仓库地址当前重定向至 `dropbox-dash/faster-whisper-large-v3-turbo`,测试使用该重定向目标。 +- 当前应用模型设置和项目主虚拟环境未更改;测试依赖安装在独立虚拟环境。 + +```powershell +tmp/stt-benchmark-20260921/venv/Scripts/python.exe tools/benchmark_stt_local.py --root tmp/stt-benchmark-20260921 --model qwen-06-cuda +``` diff --git a/docs/stt-integration-2026-09-21.md b/docs/stt-integration-2026-09-21.md new file mode 100644 index 00000000..80f21f5f --- /dev/null +++ b/docs/stt-integration-2026-09-21.md @@ -0,0 +1,79 @@ +# 语音识别升级接入说明 + +2026-09-21:源码已接入新模型,保留原 Whisper 模型 ID、用户选择及旧缓存。此次未生成或替换正式安装包。 + +## 软件中的入口 + +在设置的“语音识别模型”中下载、选择模型;聊天页的语音转写侧栏使用同一组选项。旧模型通过“兼容模型”展开,当前已选的旧模型始终可见。未安装的运行组件会显示原因,不能误选成可用模型。 + +| 档位 | 选项 | 运行设备 | +| --- | --- | --- | +| 低配极速 | Zipformer CTC INT8 | CPU;中英文、无标点 | +| 中配质量优先 | Qwen3-ASR 0.6B ONNX INT4 | CPU;建议 16 GB 内存 | +| GPU 速度优先 | 原 Whisper Turbo | NVIDIA GPU,保留原 CPU 回退逻辑 | +| GPU 质量优先 | Qwen3-ASR 0.6B / 1.7B | NVIDIA GPU,需单独的 Qwen GPU 运行组件 | + +CPU/GPU 版本是独立选项。选中新模型会设置匹配的设备;环境变量锁定设备时不会覆盖。Qwen GPU 失败会给出错误,不会悄悄切换另一模型。新后端单进程串行复用,避免批量并发创建多份模型;取消时终止工作进程,下一条任务可以重新加载。空闲 120 秒后进程自动释放。 + +本机四个新模型和 Turbo 的文件已安装到 `%APPDATA%/wechat-data-analysis-desktop/voice_models/`,新模型复制前已校验固定版本的 SHA-256。首次接入保留原选择;后续应用户要求在桌面应用验收,最终启用 Qwen3-ASR 1.7B GPU。 + +## 启动及构建 + +项目普通开发启动会安装 CPU 语音依赖。前端需要 Node 20.19+ 或 22.12+;本机系统 Node 18 太旧,本次测试和构建使用已存在的 Codex Node 运行时。使用符合版本要求的 Node 后,要启用本机已验证的 Qwen GPU,在仓库根目录运行: + +```powershell +npm --prefix desktop run dev:gpu +``` + +本机不更改系统 Node 也可以这样启动: + +```powershell +$env:WECHAT_TOOL_QWEN_GPU = '1' +& "$env:USERPROFILE/.cache/codex-runtimes/codex-primary-runtime/dependencies/node/bin/node.exe" desktop/scripts/dev.cjs +``` + +也可以手动安装: + +```powershell +uv sync --extra voice-transcription +# 需要 Qwen GPU 时加上该扩展;Windows 锁定官方 PyTorch CUDA 12.8 索引。 +uv sync --extra voice-transcription --extra voice-transcription-gpu +``` + +`desktop` 的 `build:backend` 构建 CPU 版本,包含 sherpa-onnx 和新模型资产清单,排除 PyTorch/Transformers;`build:backend:gpu` 额外收集 Qwen GPU 组件。默认 `dist:win` 仍走 CPU 后端构建,不应据此宣称普通安装包已包含 Qwen GPU。完整 GPU 安装包需要沿用项目正式签名和原生核心构建流程,使用 GPU 后端构建产物。 + +## 文件及缓存 + +- `resources/voice_models.json` 固定 Hugging Face 仓库、revision、必要文件、大小和 SHA-256;下载后先校验,再原子发布。 +- `asr_models.py` 定义模型目录及能力;`asr_backends.py` 提供三种后端;`asr_worker.py` 隔离模型、内存、取消和 CUDA 探测。 +- 新模型缓存键包含模型 ID、revision、后端和缓存版本;旧 Whisper 缓存键保持原样,切换不会误用另一模型的文本。 +- 推理只读取本地权重和音频,工作进程启用 Hugging Face 离线模式。Zipformer 最长 15 秒、Qwen 最长 25 秒分段,优先在低能量位置切分。 +- Qwen ONNX 按实际 tokenizer 编码角色提示,避免社区示例的固定 token ID 不匹配;CPU 特征提取不依赖 PyTorch。 +- Windows Whisper CUDA 可以复用已安装 PyTorch 中的 CUDA 12 DLL,解决只有系统 CUDA 13 时的依赖缺失。 + +## 本机验收 + +通过项目正式 `VoiceTranscriptionService.transcribe_voice` 读取并解码 20 条真实 SILK,四模型共 80 次成功;写缓存和批量缓存查询均验证通过。缓存写入测试目录,没有改写原会话的转写缓存。音频总长 202.54 秒。 + +| 模型 | 20 条总耗时,含加载、解码和缓存 | 与微信机器参考的字符差异率 | +| --- | ---: | ---: | +| Zipformer CTC | 6.202 秒 | 10.58% | +| Qwen 0.6B CPU | 65.125 秒 | 4.47% | +| Qwen 0.6B GPU | 51.954 秒 | 3.53% | +| Qwen 1.7B GPU | 47.384 秒 | 2.82% | + +本表是单轮完整链路验收,与之前仅计推理的双轮基准计时不同,不能混算加速倍数。微信机器转写未经人工校对,差异率不是人工标注准确率;低配最低硬件、方言、噪声和长录音仍需更大语料验证。 + +回归结果:后端 132 通过、1 跳过;前端语音组件 63 通过;设置契约与桌面启动契约 28 通过;前端 Nuxt 生产静态构建成功。真实工作进程取消后约 81 毫秒完成回收,再次识别成功;Turbo 实际 CUDA 推理成功。NumPy 声学特征与上游 PyTorch 版本比较,最大绝对误差小于 0.0001。 + +界面机械检查仅发现原有进度条的 width 动画告警,没有新增告警。后续启动真实 Electron 开发应用完成桌面验收:设置中选择 Zipformer,聊天消息实际转写并显示结果;批量扫描可以启动和取消;Qwen 1.7B GPU 同样在聊天页转写成功,来源提示显示正确模型。修正了聊天来源提示残留的 Whisper 专属文案。 + +运行中的应用 HTTP API 对四个新模型各处理 3 条真实语音,并逐条验证缓存命中、模型 ID 与 CPU/GPU 设备。另将原有 20 条 SILK 复制到隔离测试账号,使用正式批量管理器完整转写:20/20 成功、失败 0,配置并发 4 时实际限制为 1;再次扫描 20/20 命中缓存。真实账号的全库扫描仅验证启动和取消,没有等待全部历史语音完成。正式安装包及其冻结工作进程尚未完成验收。 + +私人音频、参考文本、逐条输出和测试数据库保存在被 Git 忽略的 `tmp/stt-benchmark-20260921/` 与 `tmp/stt-integration-20260921/`;本文不包含会话内容。 + +## 独立 PR 基线复验 + +PR 基于上游 `main` 的 `2646cfdf`,只移入本次语音升级。上游尚无开发分支上的导出语音选项,因此没有带入相关导出界面和导出状态改动。 + +独立工作区后端相关用例为 131 通过、1 跳过、1 失败;唯一失败是 `test_export_option_is_wired_from_dialog_to_backend`,其要求的 `exportTranscribeVoice` 控件在上游不存在。已从未修改的 `origin/main` 提取该测试及其全部输入文件,独立复现相同断言失败;本 PR 不修改这项测试。语音组件 63 项、设置与桌面启动契约 28 项均通过。使用符合前端版本要求的 Node 重新安装锁定依赖后,Nuxt 生产静态构建成功,预渲染 34 个路由。 diff --git a/docs/stt-model-upgrade-plan-2026-09-21.md b/docs/stt-model-upgrade-plan-2026-09-21.md new file mode 100644 index 00000000..e93b168f --- /dev/null +++ b/docs/stt-model-upgrade-plan-2026-09-21.md @@ -0,0 +1,116 @@ +# 本地语音转文字模型升级方案 + +日期:2026-09-21。状态:已接入四个新模型及模型管理界面,完成本机真实 SILK 的正式服务验收;用户原有模型选择未自动更改。落地方式与验收范围见 [接入说明](stt-integration-2026-09-21.md),选型实测证据见 [本机测试报告](stt-benchmark-2026-09-21.md)。20 条机器参考样本不能代替发布前的人工准确率验收。下文的硬件自动推荐和更广语料验收属于后续目标,首版提供手动选型。 + +## 目标与结论 + +面向微信短语音,提供低配 CPU、主流 CPU、NVIDIA GPU 三档能力。新增 Zipformer 和 Qwen3-ASR,同时保留实测表现突出的 Whisper Turbo;用户已选模型不自动更换。 + +初版“所有档位换新”的方案经实测后调整:低配主推 Zipformer CTC,主流 CPU 增加 Qwen ONNX,GPU 将速度优先和质量优先分开。Transducer 在本次样本上比 CTC 更慢、与微信转写的差异更多,暂不单列为更高档。Qwen 0.6B 的 CPU/GPU 版本属于不同运行配置。 + +已确认的本机取舍:CTC 约 2.74 秒处理 202.54 秒语音;Qwen CPU 约 61.59 秒,比 Medium CPU 快约 2.66 倍,但峰值进程内存约 3.65 GiB;GPU Turbo 约 7.87 秒,Qwen 0.6B 约 38.57 秒,因此 Qwen 不能直接接替 Turbo 的极速定位。差异率只代表与微信机器转写的一致程度。 + +所有硬件建议都是待验收目标,不是已验证的最低配置。权重文件大小不等于运行内存,量化位数也不代表整条推理链都采用该精度。 + +## 模型与原有档位对应 + +| 新选项 | 对应旧选项 | 模型来源与版本 | 运行方式 | 已核实的主要文件大小 | 验收目标设备 | +| --- | --- | --- | --- | --- | --- | +| 低配极速 | Tiny、Base 的优先新候选;Small 保留兼容 | pkufool/zipformer-small,CTC INT8 | CPU,sherpa-onnx,必须增加分段 | ctc.int8.onnx 约 28.7 MB,另加词表 | 4–8 GB 内存、低功耗/老款 CPU 仍需另测 | +| CPU 质量优先 | Medium 的新候选 | andrewleech/qwen3-asr-0.6b-onnx,INT4 变体 | CPU,ONNX Runtime | 指定推理文件合计约 2.03 GB | 优先 16 GB 内存;本机进程峰值约 3.65 GiB | +| GPU 极速 | Turbo 保留 | faster-whisper-large-v3-turbo | CUDA,CTranslate2 FP16 | model.bin 约 1.62 GB,另加配置和词表 | 本机 12 GB 显存已测;更低显存仍需测 | +| GPU 质量优先 | 新增选项 | Qwen/Qwen3-ASR-0.6B-hf | CUDA,PyTorch / Transformers BF16 | model.safetensors 约 1.56 GB,另加配置和词表 | 本机分配显存峰值约 1.67 GiB;不是系统最低配置保证 | +| GPU 质量优先大模型 | Large v3 的新候选 | Qwen/Qwen3-ASR-1.7B-hf | CUDA,PyTorch / Transformers BF16 | model.safetensors 约 4.08 GB,另加配置和词表 | 本机分配显存峰值约 4.01 GiB,预留显存另算 | + +大小按十进制 MB/GB 表示,只包含所列权重及文件;不包含推理运行库。CPU 两档不应依赖 PyTorch 或 CUDA。Qwen GPU 档精度依据设备能力验证 BF16/FP16,不能沿用 CTranslate2 的 compute_type 逻辑。Zipformer Transducer 保留研究记录,不作为首批必上的独立档位。 + +### 选型依据和边界 + +- [Zipformer Small 模型卡](https://huggingface.co/pkufool/zipformer-small)提供中英文 CTC 和 Transducer 两种解码头;[文件列表](https://huggingface.co/pkufool/zipformer-small/tree/main)包含 INT8 文件。模型卡上 Transducer 在列出的测试集上比 CTC 更准确,但没有证明其在本项目中一定快于 Whisper。CTC 作为最低资源档、Transducer 作为均衡档,是待实测的工程选型。 +- 该 Zipformer 仓库建立于 2026-06-25,属于近期发布的模型资产;Zipformer 架构本身来自 2023 年,不能宣称是 2026 年新发明的架构。供应方[部署说明](https://pkufool.github.io/zipformer/en/deployment/)推荐 sherpa-onnx,但必须验证这里的具体导出文件和所锁定版本相容。 +- [Qwen 0.6B ONNX](https://huggingface.co/andrewleech/qwen3-asr-0.6b-onnx)是社区导出;INT4 主要用于解码器,编码器仍为 FP32。选用它是为了评估 CPU 上的资源与精度折中,不能将其他导出版本的性能数据直接套用。 +- [Qwen 0.6B HF](https://huggingface.co/Qwen/Qwen3-ASR-0.6B-hf)和[1.7B HF](https://huggingface.co/Qwen/Qwen3-ASR-1.7B-hf)是官方 Transformers 原生版本;Qwen3-ASR 属于 2026 年模型系列,HF 原生仓库于 2026 年 6 月建立。模型卡要求 Transformers >= 5.13.0。 +- Zipformer 低配档先按中英文能力展示;不能承诺 Qwen 同等的方言、多语言和热词能力。低配档标点需要单独评估,首版允许提供无标点文本,不能为了补标点默认加载大型语言模型。 +- 不将 SenseVoiceSmall 或 BELLE 作为本次“新模型”主线:它们仍可用作评测对照,但原始模型属于 2024 年。Fun-ASR-Nano、FireRedASR2 作为后续中文专项候选,首版避免引入更多推理框架。 + +## 默认推荐规则 + +1. 新安装首次打开模型设置时,根据可用内存、CPU、GPU 能力推荐一个档位,用户点击下载后才下载资产。机器总内存不能单独决定推荐结果。 +2. 低配优先验证“低配极速”CTC。首版不把 Qwen 作为所有机器统一默认,也不按模型文件体积推算运行内存。 +3. 16 GB CPU 机器提供“CPU 质量优先”,展示本机测试环境、耗时和内存;不能把 5600X 的速度直接套用到低功耗 CPU。 +4. GPU 可用时保留 Turbo 极速路径;Qwen 作为质量优先的可选路径,说明其标准 Transformers 运行方式在本机更慢。显存不足时减小并发、分段或提示切换已安装的较小模型。 +5. Qwen 0.6B 的 CPU/GPU 版本在界面中说明为同系列不同运行方式。模型卡不使用“最高准确率”等未经项目评测支持的绝对标签。 + +## 项目改造范围 + +当前 voice_transcription.py 将目录校验、WhisperModel 加载、transcribe 参数和 CPU 回退都绑定到 faster-whisper;模型列表有 Tiny、Base、Small、Medium、Large v3、Turbo 六项。pyproject.toml 已包含 onnxruntime 和 tokenizers,但没有 Qwen 所需的 PyTorch / Transformers。 + +### 一、推理接口 + +提取独立 ASR 后端接口:load、transcribe、unload、capabilities。统一输出文本、时长、检测语言(允许未知)、实际模型标识、后端版本、实际设备和耗时。 + +- 保留 WhisperBackend 处理旧模型。 +- 增加 ZipformerBackend:CTC 和 Transducer 共用模型管理,分别使用正确的特征提取及解码器。 +- 增加 QwenOnnxBackend:处理音频特征、提示词、KV cache、逐步解码和取消;不能只调用 ONNX 文件一次就当作完整识别。 +- 增加 QwenTransformersBackend:按官方 HF 接口加载 0.6B / 1.7B,关闭训练行为,规范输出中的语言标记和文本。 + +微信 SILK 解码、任务排队、进度、转写缓存及前端结果展示继续复用。音频统一到后端要求的单声道 16 kHz 格式,各后端的特征提取不可混用。 + +### 二、模型资产与配置 + +将固定 Whisper 文件白名单改为逐模型清单,字段至少包括 modelId、backend、repoId、revision、files、文件哈希、量化方式、支持设备和语言。 + +建议新 ID:zipformer-small-ctc-int8、zipformer-small-rnnt-int8、qwen3-asr-06b-onnx-int4、qwen3-asr-06b-hf、qwen3-asr-17b-hf。 + +- 精确下载指定变体所需文件,禁止整仓下载不同精度和训练检查点。 +- 下载完成校验后再原子发布目录,继续支持取消、重试、删除和占用保护。 +- 新增通用 ASR 配置;兼容旧 WECHAT_TOOL_WHISPER_* 环境变量,旧变量仍指向旧后端,不暗中重解释。 +- 模型 ID、版本和量化变体参与缓存身份;不能把原来的 medium ID 指向 Qwen 后继续读取 Whisper 结果。 +- 旧缓存和模型不删除。已安装旧模型可在“旧版模型”区域查看、使用和主动移除。 + +### 三、低配运行与 GPU 依赖 + +- CPU 默认一次只识别一条,线程数从 2 开始按设备调整,避免与任务并发相乘造成过载。 +- 采用独立推理进程,任务间复用模型;空闲后卸载,取消或异常时可终止工作进程释放内存。 +- 音频分段并限制输出长度,测试静音、尾音、重复输出和跨段文字拼接。 +- GPU 推理依赖按需安装或作为独立运行包发布;普通 CPU 安装包不强制包含 PyTorch/CUDA。 +- 当前 CTranslate2 的 CUDA 探测不能证明 PyTorch CUDA 可用;各后端分别探测。 +- GPU 故障不能直接沿用当前“同模型 CPU int8”回退逻辑。只在事先允许且模型已安装时切换明确的 CPU 档,并显示实际使用模型;否则提供可操作错误,不自动下载另一个大模型。 +- Windows 为第一验收平台;macOS/Linux 的轮子、算子兼容和打包分别验证。项目 macOS ONNX Runtime 版本不同,不能按 Windows 测试结果宣称全平台可用。 + +## 实施顺序与验收 + +### 第一步:最小评测,确定名单 + +独立评测入口已实现为 `tools/benchmark_stt_local.py`,在固定版本上进行 20 条真实语音初筛。具体结果及限制见测试报告;以下更大规模人工评测仍是发布前的下一阶段。 + +测试集至少覆盖 100 条、3–60 秒的短语音:普通话、带口音普通话、中英混合、方言、嘈杂、近静音、人名数字和语速较快的内容。公共样本可先跑;微信样本在明确选定范围后本地处理。 + +记录中文 CER、英文 WER、人名/数字错误、无语音误识别、冷启动、热启动 p50/p95 延迟、峰值进程树内存、显存、取消耗时和长批次内存增长。标点单独评价。资源分档不等于质量严格单调,跨模型质量必须由同一套数据确认。 + +建议发布门槛:低配档在目标机与现有相应档比较不明显降低准确率,同时降低占用或耗时;Qwen 中高档在主要中文场景体现可量化收益。达不到条件的候选不作为新默认。4 GB 极低配必须单独测试,不能以 8 GB 结果代替。 + +### 第二步:上线低配 CPU 档 + +完成后端接口、Zipformer CTC、模型下载管理和原有设置兼容。修复较长输入的分段问题,再覆盖低配基本需求;Transducer 暂不作为首版必需项。 + +### 第三步:上线 Qwen CPU/GPU + +验证 Qwen ONNX INT4 中文量化退化、Windows 算子兼容和实际内存;再加入官方 HF 0.6B/1.7B GPU 档及可选运行包。若社区 ONNX 版本不通过,保留 Zipformer 默认,不发布未经验证的 CPU Qwen。 + +### 第四步:迁移展示与发布 + +设置页按实测价值展示配置、实际模型名、下载大小、语言和建议设备。Turbo 保留在主要选项中;其他原模型保留兼容入口。已选模型继续生效,升级仅提示可选的新档位,避免为了凑齐档位强行增加模型。 + +必要回归:完全离线、损坏/中断下载、模型删除与正在推理冲突、取消、切换模型、缓存隔离、GPU 不可用/显存不足、旧配置启动、桌面打包启动。验收完成后才把“候选”改为“正式推荐”。 + +## 调研版本记录 + +以下为本次核查到的 HF revision,仅供实施评测锁定与复现;发布前需要按验收版本生成完整文件清单与哈希。 + +| 仓库 | revision | +| --- | --- | +| pkufool/zipformer-small | e1764e4e54504721900d1e6b99c746e7331980af | +| andrewleech/qwen3-asr-0.6b-onnx | 4fc24a1402e74db89c4d2ef256875e71680128c4 | +| Qwen/Qwen3-ASR-0.6B-hf | 7f1569a48a89f3e3f4dc3a5c9d28bddd903bc76c | +| Qwen/Qwen3-ASR-1.7B-hf | bcd2b5b7f32b480ab5790554cfa8347f246a14f3 | diff --git a/frontend/components/SettingsDialog.vue b/frontend/components/SettingsDialog.vue index 98f0e6b7..023e5f2a 100644 --- a/frontend/components/SettingsDialog.vue +++ b/frontend/components/SettingsDialog.vue @@ -263,7 +263,7 @@
CUDA 失败自动回退 CPU
{{ voiceModelSerial ? '设备由所选模型决定' : 'CUDA 失败自动回退 CPU' }}
{{ voiceBatchConcurrencyError }} @@ -246,6 +246,8 @@ export default defineComponent({ || '' ).trim()) const voiceModels = computed(() => Array.isArray(voiceStatus.value.models) ? voiceStatus.value.models : []) + const voiceModelSerial = computed(() => !!voiceStatus.value.backend && voiceStatus.value.backend !== 'whisper') + const voiceSupportedDevices = computed(() => voiceStatus.value.supportedDevices || ['cpu', 'cuda']) const voiceSelectableModels = computed(() => voiceModels.value.filter((model) => model?.downloaded === true)) const voiceCurrentModel = computed(() => String(voiceStatus.value.model || '').trim()) const voiceCurrentModelInfo = computed(() => { @@ -363,6 +365,8 @@ export default defineComponent({ voiceNativeAvailable, voiceNativeReason, voiceModels, + voiceModelSerial, + voiceSupportedDevices, voiceSelectableModels, voiceCurrentModel, voiceCurrentModelInfo, diff --git a/frontend/composables/chat/useChatMessages.js b/frontend/composables/chat/useChatMessages.js index f83fca79..96d18ad9 100644 --- a/frontend/composables/chat/useChatMessages.js +++ b/frontend/composables/chat/useChatMessages.js @@ -169,7 +169,7 @@ export const useChatMessages = ({ const voiceTranscriptionStatusKnown = computed(() => !!voiceTranscriptionStatus.value) const voiceTranscriptionAvailable = computed(() => voiceTranscriptionStatus.value?.available === true) const voiceTranscriptionUnavailableReason = computed(() => String( - voiceTranscriptionStatus.value?.reason || '本地 Whisper 模型尚未准备好。' + voiceTranscriptionStatus.value?.reason || '本地语音模型尚未准备好。' ).trim()) const refreshVoiceTranscriptionStatus = async ({ force = false } = {}) => { @@ -1693,7 +1693,7 @@ export const useChatMessages = ({ } } - // 本地 Whisper 是用户显式选择的备用路径;原生转写失败时不要静默切换来源。 + // 本地模型是用户显式选择的备用路径;原生转写失败时不要静默切换来源。 const transcribeVoiceLocally = async (message, { force = false } = {}) => { const transcriptRevision = projectTranscriptRevision const accountAtStart = String(selectedAccount.value || '').trim() @@ -1749,8 +1749,8 @@ export const useChatMessages = ({ if (!requestIsCurrent()) return if (!capability?.available) { setVoiceError( - { message: String(capability?.reason || '本地 Whisper 模型尚未准备好。').trim() }, - '本地 Whisper 模型尚未准备好。' + { message: String(capability?.reason || '本地语音模型尚未准备好。').trim() }, + '本地语音模型尚未准备好。' ) return } diff --git a/frontend/tests/voice-model-settings.test.mjs b/frontend/tests/voice-model-settings.test.mjs index 54a1c249..35579335 100644 --- a/frontend/tests/voice-model-settings.test.mjs +++ b/frontend/tests/voice-model-settings.test.mjs @@ -74,7 +74,7 @@ test('download-time deletion stays available and supersedes stale download work' test('settings shows inference device before the model catalog', () => { assert.ok(voiceSectionStart >= 0 && voiceSectionEnd > voiceSectionStart) - assert.ok(voiceSectionSource.indexOf('>推理设备<') < voiceSectionSource.indexOf('>Whisper 模型<')) + assert.ok(voiceSectionSource.indexOf('>推理设备<') < voiceSectionSource.indexOf('>语音识别模型<')) }) test('settings keeps device and model selection feedback compact', () => { diff --git a/frontend/tests/voice-transcription-sidebar.test.js b/frontend/tests/voice-transcription-sidebar.test.js index 3a7ac7c2..a3c6b358 100644 --- a/frontend/tests/voice-transcription-sidebar.test.js +++ b/frontend/tests/voice-transcription-sidebar.test.js @@ -106,6 +106,28 @@ afterEach(() => { }) describe('聊天页语音转文字侧栏', () => { + it('新后端限制并发与设备,缺少 GPU 组件时不能选择该模型', () => { + const state = makeState({ + voiceBatchJob: ref({ status: 'idle', percent: 0 }), + voiceTranscriptionStatus: ref({ + available: true, backend: 'zipformer', model: 'zipformer-small-ctc-int8', + modelReady: true, requestedDevice: 'cpu', supportedDevices: ['cpu'], + cuda: { available: true }, + models: [ + { id: 'zipformer-small-ctc-int8', name: 'Zipformer CTC', downloaded: true, runtimeAvailable: true }, + { id: 'qwen3-asr-06b-hf', name: 'Qwen GPU', downloaded: true, runtimeAvailable: false }, + ], + }), + }) + const wrapper = mount(VoiceTranscriptionSidebar, { props: { state } }) + expect(wrapper.get('.voice-concurrency-select').element.disabled).toBe(true) + expect(wrapper.get('.voice-device-cuda').element.disabled).toBe(true) + expect(wrapper.get('.voice-device-cpu').element.disabled).toBe(false) + expect(wrapper.get('option[value="qwen3-asr-06b-hf"]').element.disabled).toBe(true) + expect(wrapper.text()).toContain('缺少运行组件') + expect(wrapper.text()).not.toContain('CUDA 失败自动回退 CPU') + }) + it('快速切换多个账号时保留最后一次选择', () => { const accountChangeSource = chatPageSource.slice( chatPageSource.indexOf('const onAccountChange = async () =>'), @@ -373,7 +395,7 @@ describe('聊天页语音转文字侧栏', () => { expect(settingsDialogSource).toContain('删除全部本项目转写结果') expect(settingsDialogSource).toContain(':disabled="voiceTranscriptDeleteBusy"') expect(settingsDialogSource).toContain('此操作不可撤销') - expect(settingsDialogSource).toContain('所有账号中由本项目 Whisper 生成的全部转写文字') + expect(settingsDialogSource).toContain('所有账号中由本项目本地模型生成的全部转写文字') expect(settingsDialogSource).toContain('微信原生转写、原始语音和已下载模型都会保留') expect(settingsDialogSource).toContain('await api.deleteAllVoiceTranscriptionCache()') expect(settingsDialogSource).toContain('notifyProjectVoiceTranscriptsInvalidated(result)') diff --git a/pyproject.toml b/pyproject.toml index c0ac2368..fa062a9e 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -54,6 +54,13 @@ build = [ voice-transcription = [ "faster-whisper>=1.1.0", "opencc-python-reimplemented>=0.1.7", + "sherpa-onnx==1.13.8", + "sherpa-onnx-core==1.13.8; sys_platform == 'win32'", +] +# GPU 运行组件单独安装,CPU 安装包不引入 PyTorch。 +voice-transcription-gpu = [ + "torch==2.9.1", + "transformers==5.17.0", ] [dependency-groups] @@ -89,6 +96,14 @@ include = [ [tool.uv] find-links = ["./tools/key_wheels/"] +[[tool.uv.index]] +name = "pytorch-cu128" +url = "https://download.pytorch.org/whl/cu128" +explicit = true + +[tool.uv.sources] +torch = [{ index = "pytorch-cu128", marker = "sys_platform == 'win32'" }] + [tool.pytest.ini_options] testpaths = ["tests"] # 让 pytest 命令与 python -m pytest 都能导入仓库工具和源码。 diff --git a/src/wechat_decrypt_tool/asr_backends.py b/src/wechat_decrypt_tool/asr_backends.py new file mode 100644 index 00000000..aa41d6f4 --- /dev/null +++ b/src/wechat_decrypt_tool/asr_backends.py @@ -0,0 +1,178 @@ +"""在语音工作进程内加载新后端;不在应用启动时导入大型推理库。""" +from __future__ import annotations + +import json +from pathlib import Path + + +LANGUAGES = {"zh": "Chinese", "en": "English", "yue": "Cantonese", "ja": "Japanese", + "ko": "Korean", "fr": "French", "de": "German", "es": "Spanish", + "ru": "Russian", "pt": "Portuguese", "ar": "Arabic", "it": "Italian"} + + +def language_name(language: str) -> str | None: + if language in ("", "auto"): + return None + if language in LANGUAGES: + return LANGUAGES[language] + if language in LANGUAGES.values(): + return language + raise ValueError("当前 Qwen 接口不支持该语言设置,请使用 zh、en 或 auto。") + + +def read_audio(path: str): + import av + import numpy as np + parts = [] + with av.open(path) as container: + resampler = av.AudioResampler(format="flt", layout="mono", rate=16000) + for frame in container.decode(audio=0): + for output in resampler.resample(frame): + parts.append(output.to_ndarray().reshape(-1)) + for output in resampler.resample(None): + parts.append(output.to_ndarray().reshape(-1)) + return np.concatenate(parts).astype(np.float32) if parts else np.zeros(0, dtype=np.float32) + + +def audio_chunks(audio, maximum_seconds: float): + """在窗口末尾的低能量处切分,避免导出模型长输入错误和截断。""" + import numpy as np + maximum = int(maximum_seconds * 16000) + remaining = audio + while len(remaining) > maximum: + candidates = range(int(maximum * 0.7), maximum - 320, 320) + cut = min(candidates, key=lambda p: float(np.mean(remaining[p:p + 320] ** 2))) + yield remaining[:cut] + remaining = remaining[cut:] + if len(remaining): + yield remaining + + +def mel_filters(): + """Slaney 归一化三角滤波器,与 Qwen 导出时的 128 维特征一致。""" + import numpy as np + log_step = np.log(6.4) / 27.0 + maximum = 15.0 + np.log(8000.0 / 1000.0) / log_step + mels = np.linspace(0.0, maximum, 130) + hz = np.where(mels < 15, mels * (200.0 / 3), 1000 * np.exp(log_step * (mels - 15))) + fft_hz = np.linspace(0, 8000, 201) + lower = (fft_hz[None, :] - hz[:-2, None]) / np.diff(hz)[:-1, None] + upper = (hz[2:, None] - fft_hz[None, :]) / np.diff(hz)[1:, None] + filters = np.maximum(0, np.minimum(lower, upper)) * (2 / (hz[2:] - hz[:-2]))[:, None] + return filters.astype(np.float32) + + +def log_mel(audio, filters): + import numpy as np + padded = np.pad(audio, (200, 200), mode="reflect") + frames = np.lib.stride_tricks.sliding_window_view(padded, 400)[::160] + window = np.hanning(401)[:-1].astype(np.float32) + powers = (np.abs(np.fft.rfft(frames * window, axis=-1)) ** 2).astype(np.float32).T + mel = np.log10(np.maximum(filters @ powers, 1e-10)) + return ((np.maximum(mel, mel.max() - 8) + 4) / 4)[None, :, :-1].astype(np.float32) + + +class ZipformerBackend: + precision = "int8" + chunk_seconds = 15 + + def __init__(self, folder: Path, threads: int): + import sherpa_onnx + self.model = sherpa_onnx.OfflineRecognizer.from_zipformer_ctc( + model=str(folder / "ctc.int8.onnx"), tokens=str(folder / "data/tokens.txt"), + num_threads=threads, sample_rate=16000, feature_dim=80, provider="cpu") + + def transcribe(self, audio, language): + if language not in ("zh", "en", "auto", ""): + raise ValueError("Zipformer CTC 仅支持中英文,请选择 Qwen 处理其他语言。") + stream = self.model.create_stream() + stream.accept_waveform(16000, audio) + self.model.decode_stream(stream) + return stream.result.text + + +class QwenOnnxBackend: + precision = "int4" + chunk_seconds = 25 + + def __init__(self, folder: Path, threads: int): + import numpy as np + import onnxruntime as ort + from tokenizers import Tokenizer + self.config = json.loads((folder / "config.json").read_text(encoding="utf-8")) + opts = ort.SessionOptions() + opts.intra_op_num_threads = threads + opts.inter_op_num_threads = 1 + self.sessions = {name: ort.InferenceSession(str(folder / f"{name}.int4.onnx"), opts, + providers=["CPUExecutionProvider"]) for name in ("encoder", "decoder_init", "decoder_step")} + decoder = self.config["decoder"] + self.embeddings = np.memmap(folder / "embed_tokens.bin", mode="r", dtype=self.config["embed_tokens_dtype"], + shape=(decoder["vocab_size"], decoder["hidden_size"])) + self.tokenizer = Tokenizer.from_file(str(folder / "tokenizer.json")) + self.filters = mel_filters() + + def transcribe(self, audio, language): + import numpy as np + features = self.sessions["encoder"].run(["audio_features"], {"mel": log_mel(audio, self.filters)})[0] + encode = lambda text: self.tokenizer.encode(text, add_special_tokens=False).ids + prefix = encode('<|im_start|>system\n<|im_end|>\n<|im_start|>user\n<|audio_start|>') + lang = language_name(language) + suffix = '<|audio_end|><|im_end|>\n<|im_start|>assistant\n' + if lang: + suffix += f'language {lang}' + prompt = prefix + [self.config["special_tokens"]["audio_pad_token_id"]] * features.shape[1] + encode(suffix) + positions = np.arange(len(prompt), dtype=np.int64)[None, :] + initial = self.sessions["decoder_init"] + if "input_ids" in {x.name for x in initial.get_inputs()}: + inputs = dict(input_ids=np.array([prompt], dtype=np.int64), position_ids=positions, + audio_features=features, audio_offset=np.array([len(prefix)], dtype=np.int64)) + else: + embeddings = np.asarray(self.embeddings[prompt], dtype=np.float32).copy() + embeddings[len(prefix):len(prefix) + features.shape[1]] = features[0] + inputs = dict(input_embeds=embeddings[None], position_ids=positions) + outputs = ["logits", "present_keys", "present_values"] + logits, keys, values = initial.run(outputs, inputs) + generated = [] + eos = self.config["special_tokens"]["eos_token_ids"] + for index in range(512): + token = int(np.argmax(logits[0, -1])) + if token in eos: + break + generated.append(token) + logits, keys, values = self.sessions["decoder_step"].run(outputs, dict( + input_embeds=np.asarray(self.embeddings[token], dtype=np.float32)[None, None], + position_ids=np.array([[len(prompt) + index]], dtype=np.int64), past_keys=keys, past_values=values)) + else: + raise RuntimeError("识别输出超过长度限制,请改用其他模型。") + return self.tokenizer.decode(generated, skip_special_tokens=True).split("")[-1].strip() + + +class QwenGpuBackend: + chunk_seconds = 25 + + def __init__(self, folder: Path, threads: int): + import torch + from transformers import AutoProcessor, AutoModelForMultimodalLM + if not torch.cuda.is_available(): + raise RuntimeError("Qwen GPU 需要可用的 PyTorch CUDA 运行环境,请改用 CPU 模型或安装 GPU 组件。") + torch.set_num_threads(threads) + dtype = torch.bfloat16 if torch.cuda.is_bf16_supported() else torch.float16 + self.precision = "bfloat16" if dtype == torch.bfloat16 else "float16" + self.processor = AutoProcessor.from_pretrained(folder, local_files_only=True) + self.model = AutoModelForMultimodalLM.from_pretrained(folder, dtype=dtype, + attn_implementation="sdpa", local_files_only=True).to("cuda").eval() + + def transcribe(self, audio, language): + import torch + with torch.inference_mode(): + inputs = self.processor.apply_transcription_request(audio=audio, language=language_name(language)) + inputs = inputs.to(self.model.device, self.model.dtype) + generated = self.model.generate(**inputs, max_new_tokens=512, do_sample=False) + tokens = generated[:, inputs["input_ids"].shape[1]:] + if tokens.shape[1] >= 512: + raise RuntimeError("识别输出超过长度限制,请改用其他模型。") + return self.processor.decode(tokens, return_format="transcription_only")[0] + + +def load_backend(backend: str, folder: str, threads: int): + return {"zipformer": ZipformerBackend, "qwen-onnx": QwenOnnxBackend, "qwen-hf": QwenGpuBackend}[backend](Path(folder), threads) diff --git a/src/wechat_decrypt_tool/asr_models.py b/src/wechat_decrypt_tool/asr_models.py new file mode 100644 index 00000000..7b705150 --- /dev/null +++ b/src/wechat_decrypt_tool/asr_models.py @@ -0,0 +1,77 @@ +"""新语音模型的固定版本、资产校验和能力声明;旧 Whisper ID 保持原含义。""" +from __future__ import annotations + +import hashlib +import importlib.util +import json +from pathlib import Path +from typing import Callable + +SPECS = json.loads((Path(__file__).parent / "resources/voice_models.json").read_text(encoding="utf-8")) +NEW_MODEL_CATALOG = ( + dict(id="zipformer-small-ctc-int8", name="Zipformer CTC", size="约 29 MB", speed="CPU 极速", + quality="低配优先", description="中英文短语音,低内存占用;长语音自动分段,输出不含标点。", recommended=True), + dict(id="qwen3-asr-06b-onnx-int4", name="Qwen3-ASR 0.6B · CPU", size="约 2.03 GB", speed="CPU", + quality="质量优先", description="无需独显;本机实测进程内存约 3.65 GiB,建议 16 GB 内存。"), + dict(id="qwen3-asr-06b-hf", name="Qwen3-ASR 0.6B · GPU", size="约 1.58 GB", speed="NVIDIA GPU", + quality="质量优先", description="需 Qwen GPU 运行组件;本机文本差异较少,速度慢于 Turbo。"), + dict(id="qwen3-asr-17b-hf", name="Qwen3-ASR 1.7B · GPU", size="约 4.09 GB", speed="NVIDIA GPU", + quality="大模型", description="需 Qwen GPU 运行组件;本机分配显存约 4 GiB,需另留运行余量。"), +) + + +def cache_identity(model: str) -> str: + spec = SPECS.get(model) + if not spec: + return model + return f"{model}@{spec['revision']}:{spec['backend']}:v{spec['cacheVersion']}" + + +def model_files_ready(path: Path, model: str) -> bool: + """状态轮询只检查路径和大小,完整哈希在模型安装前校验。""" + spec = SPECS[model] + try: + root = path.resolve() + if not path.is_dir() or path.is_symlink(): + return False + for name, entry in spec["files"].items(): + target = path / name + if not target.resolve().is_relative_to(root) or target.is_symlink(): + return False + if not target.is_file() or target.stat().st_size != entry["size"]: + return False + return True + except OSError: + return False + + +def verify_model_files(path: Path, model: str, checkpoint: Callable[[], None] = lambda: None) -> None: + if not model_files_ready(path, model): + raise ValueError("模型文件缺失或大小不符") + for name, entry in SPECS[model]["files"].items(): + digest = hashlib.sha256() + with (path / name).open("rb") as stream: + while chunk := stream.read(4 * 1024 * 1024): + checkpoint() + digest.update(chunk) + if digest.hexdigest() != entry["sha256"]: + raise ValueError(f"模型文件校验失败:{name}") + + +def dependency_status(model: str) -> tuple[bool, str]: + backend = SPECS.get(model, {}).get("backend", "whisper") + packages = { + "whisper": ("faster_whisper",), + "zipformer": ("sherpa_onnx", "av", "numpy"), + "qwen-onnx": ("onnxruntime", "tokenizers", "av", "numpy"), + "qwen-hf": ("torch", "transformers", "av", "numpy"), + }[backend] + try: + ready = all(importlib.util.find_spec(name) is not None for name in packages) + except (ImportError, ValueError): + ready = False + if ready: + return True, "" + if backend == "qwen-hf": + return False, "当前未安装 Qwen GPU 运行组件,请使用含 Qwen GPU 组件的版本,或选择 CPU 模型 / Turbo。" + return False, "缺少语音识别运行组件,请安装语音转文字可选依赖或更新应用。" diff --git a/src/wechat_decrypt_tool/asr_worker.py b/src/wechat_decrypt_tool/asr_worker.py new file mode 100644 index 00000000..e8551b85 --- /dev/null +++ b/src/wechat_decrypt_tool/asr_worker.py @@ -0,0 +1,162 @@ +"""可取消、空闲自动退出的本地语音工作进程;模型和语音不会联网。""" +from __future__ import annotations + +import multiprocessing +import os +import threading +import time + + +class AsrCancelled(RuntimeError): + pass + + +class AsrError(RuntimeError): + def __init__(self, code: str, message: str): + super().__init__(message) + self.code = code + + +def _worker(connection, backend: str, folder: str, threads: int): + os.environ.update(HF_HUB_OFFLINE="1", TRANSFORMERS_OFFLINE="1", HF_HUB_DISABLE_TELEMETRY="1", + HF_HUB_DISABLE_PROGRESS_BARS="1", OMP_NUM_THREADS=str(threads), + MKL_NUM_THREADS=str(threads), OPENBLAS_NUM_THREADS=str(threads)) + model = None + try: + from .asr_backends import audio_chunks, load_backend, read_audio + import numpy as np + while connection.poll(120): + request = connection.recv() + try: + if model is None: + model = load_backend(backend, folder, threads) + audio = read_audio(request["path"]) + texts = [] + for chunk in audio_chunks(audio, model.chunk_seconds): + # 跳过纯静音,避免自回归模型无中生有。 + if float(np.max(np.abs(chunk))) > 1e-6: + texts.append(model.transcribe(chunk, request["language"])) + separator = " " if request["language"] not in ("zh", "yue", "ja") else "" + connection.send(dict(text=separator.join(texts), duration=len(audio) / 16000, + language=request["language"], precision=model.precision)) + except Exception as exc: + code = "dependency_missing" if isinstance(exc, ImportError) else "transcription_failed" + connection.send(dict(error=str(exc)[:500], code=code)) + # 错误后退出,释放可能处于半初始化状态的模型和 CUDA 显存。 + break + except (EOFError, BrokenPipeError, OSError): + pass + finally: + connection.close() + + +class ProcessBackend: + def __init__(self, backend: str, folder: str, precision: str, threads: int = 4): + self.backend, self.folder, self.precision = backend, folder, precision + self.threads = max(1, min(threads, os.cpu_count() or 1)) + self.process = None + self.connection = None + + @property + def is_loaded(self): + return self.process is not None and self.process.is_alive() + + def _start(self): + if self.process is not None and self.process.is_alive(): + return + self.close() + context = multiprocessing.get_context("spawn") + self.connection, child = context.Pipe() + self.process = context.Process(target=_worker, args=(child, self.backend, self.folder, self.threads), daemon=True) + try: + self.process.start() + finally: + child.close() + + def transcribe_audio(self, path: str, language: str, cancel_event=None): + if cancel_event is not None and cancel_event.is_set(): + raise AsrCancelled() + self._start() + deadline = time.monotonic() + 900 + try: + self.connection.send(dict(path=path, language=language)) + while True: + if cancel_event is not None and cancel_event.is_set(): + raise AsrCancelled() + if time.monotonic() >= deadline: + raise AsrError("transcription_timeout", "识别等待超时,请改用 Zipformer CTC 或较小模型。") + if self.connection.poll(0.05): + result = self.connection.recv() + if "error" in result: + raise AsrError(result["code"], result["error"]) + self.precision = result["precision"] + return result + if not self.process.is_alive(): + raise AsrError("worker_exited", "语音工作进程已退出,可能内存不足;请重试或选择较小模型。") + except (EOFError, BrokenPipeError, OSError) as exc: + self.close() + raise AsrError("worker_exited", "语音工作进程中断,请重试或选择较小模型。") from exc + except BaseException: + self.close() + raise + + def close(self): + if self.connection is not None: + self.connection.close() + self.connection = None + if self.process is not None: + if self.process.pid is not None: + if self.process.is_alive(): + self.process.terminate() + self.process.join(timeout=5) + if self.process.is_alive(): + self.process.kill() + self.process.join(timeout=2) + self.process.close() + self.process = None + + +def _torch_probe(connection): + try: + import torch + count = torch.cuda.device_count() if torch.cuda.is_available() else 0 + connection.send(dict(available=count > 0, deviceCount=count, + devices=[dict(name=torch.cuda.get_device_name(i)) for i in range(count)], + reason="" if count else "PyTorch CUDA 不可用,请安装 GPU 运行组件和显卡驱动,或选择 CPU 模型。")) + except Exception: + connection.send(dict(available=False, deviceCount=0, devices=[], reason="Qwen GPU 运行组件无法加载,请更新 GPU 组件或选择 CPU 模型。")) + finally: + connection.close() + + +_PROBE_LOCK = threading.Lock() +_PROBE_CACHE = None + + +def probe_qwen_cuda(): + global _PROBE_CACHE + with _PROBE_LOCK: + if _PROBE_CACHE and time.monotonic() < _PROBE_CACHE[0]: + return dict(_PROBE_CACHE[1]) + context = multiprocessing.get_context("spawn") + parent, child = context.Pipe(duplex=False) + process = context.Process(target=_torch_probe, args=(child,), daemon=True) + result = dict(available=False, deviceCount=0, devices=[], reason="Qwen GPU 检测未完成,请检查运行组件或改用 CPU 模型。") + try: + process.start() + child.close() + if parent.poll(30): + result = parent.recv() + except (EOFError, OSError, RuntimeError): + pass + finally: + parent.close() + child.close() + if process.pid is not None: + process.join(timeout=1) + if process.is_alive(): + process.terminate() + process.join(timeout=2) + process.close() + _PROBE_CACHE = (time.monotonic() + 60, result) + return dict(result) diff --git a/src/wechat_decrypt_tool/resources/voice_models.json b/src/wechat_decrypt_tool/resources/voice_models.json new file mode 100644 index 00000000..dc870b01 --- /dev/null +++ b/src/wechat_decrypt_tool/resources/voice_models.json @@ -0,0 +1,146 @@ +{ + "zipformer-small-ctc-int8": { + "backend": "zipformer", + "repo": "pkufool/zipformer-small", + "revision": "e1764e4e54504721900d1e6b99c746e7331980af", + "cacheVersion": 1, + "devices": [ + "cpu" + ], + "precision": "int8", + "files": { + "ctc.int8.onnx": { + "size": 28747091, + "sha256": "2d430ee45ac05a23ff19eb8fe3e19effe33f7d0c7b627e96b2286df423047617" + }, + "data/tokens.txt": { + "size": 114134, + "sha256": "5beaddf82e078ef5644dcad130f3c4817a04016bfb0f31c0df28c6b4fa8df3af" + } + } + }, + "qwen3-asr-06b-onnx-int4": { + "backend": "qwen-onnx", + "repo": "andrewleech/qwen3-asr-0.6b-onnx", + "revision": "4fc24a1402e74db89c4d2ef256875e71680128c4", + "cacheVersion": 1, + "devices": [ + "cpu" + ], + "precision": "int4", + "files": { + "config.json": { + "size": 1250, + "sha256": "df31c4689abe9d782366fffd2454b546291d0205d082b3fd01b99fb76a45b11f" + }, + "decoder_init.int4.onnx": { + "size": 355390, + "sha256": "6d54633bbb9e6b1ecf372f218d04adb3bf6f2bfc442ce55c988526d532896faa" + }, + "decoder_step.int4.onnx": { + "size": 354888, + "sha256": "7eaa4f31d5eb6ae2937ff45222c32f692f2d5ac29b46cbae9046da332a9f317d" + }, + "decoder_weights.int4.data": { + "size": 962460672, + "sha256": "d68cd1c0695a7ba42651d06b1bc1158e2e58af5f9adbb6dba874fbbb8a4f22cf" + }, + "embed_tokens.bin": { + "size": 311164928, + "sha256": "e80150119fa5f7e56e85aed64c3a02d5c78eb7a37cfdcb973d0987316f15bee2" + }, + "encoder.int4.onnx": { + "size": 745762694, + "sha256": "3c027f880f677615de85e1f6934906e1d5d77624b724096d33accef99c753eed" + }, + "preprocessor_config.json": { + "size": 298, + "sha256": "9c7c558a05f326fe4365e5e59e486383ff127dfd93ce1cd5e6e23f883cd02281" + }, + "tokenizer.json": { + "size": 11429377, + "sha256": "bd2a97b55c8f7f9c328c73ee9b9178771037e9f566dfca8e238a063d41cbac92" + } + } + }, + "qwen3-asr-06b-hf": { + "backend": "qwen-hf", + "repo": "Qwen/Qwen3-ASR-0.6B-hf", + "revision": "7f1569a48a89f3e3f4dc3a5c9d28bddd903bc76c", + "cacheVersion": 1, + "devices": [ + "cuda" + ], + "precision": "bfloat16", + "files": { + "chat_template.jinja": { + "size": 1434, + "sha256": "f50e6b694fbf4a683206e37869990d68333fe95d285730f084c838a34b0d98c2" + }, + "config.json": { + "size": 2398, + "sha256": "9eecf6f1b383e343889c2e6010e632590fa57d4bc678e151c7d6a160a0dfb04a" + }, + "generation_config.json": { + "size": 165, + "sha256": "9939fc9388b79bd70757f938b87381e817173d6a6158f5af6506c0b73e775c3c" + }, + "model.safetensors": { + "size": 1564928088, + "sha256": "d3f212dd20abecd315d830bc54ae3865e56ebfc3276484e57b771288ba27fd35" + }, + "processor_config.json": { + "size": 487, + "sha256": "bc0b230081b44e629dd5b9045b78495615c1831b4b9f4cffe97bd37e82a6156a" + }, + "tokenizer.json": { + "size": 11429653, + "sha256": "fe1fad59be22a41ee293363fcf95fdedbc7c93f3b49270b1d2e18bd1399a7a05" + }, + "tokenizer_config.json": { + "size": 998, + "sha256": "945e980986de2ca7768f3326bfdbb4fbea3406f972b8ae0be233089f2b253c11" + } + } + }, + "qwen3-asr-17b-hf": { + "backend": "qwen-hf", + "repo": "Qwen/Qwen3-ASR-1.7B-hf", + "revision": "bcd2b5b7f32b480ab5790554cfa8347f246a14f3", + "cacheVersion": 1, + "devices": [ + "cuda" + ], + "precision": "bfloat16", + "files": { + "chat_template.jinja": { + "size": 1434, + "sha256": "f50e6b694fbf4a683206e37869990d68333fe95d285730f084c838a34b0d98c2" + }, + "config.json": { + "size": 2399, + "sha256": "117ac8e63e2af7cae3665e5a632d6eb03f5f384915519ceb6403c15ec6533f63" + }, + "generation_config.json": { + "size": 165, + "sha256": "9939fc9388b79bd70757f938b87381e817173d6a6158f5af6506c0b73e775c3c" + }, + "model.safetensors": { + "size": 4076193080, + "sha256": "2db53c7d81bd9b8cbc6a074e89be2c968a0d373fb4ee68bb1b1e14f7042dfee1" + }, + "processor_config.json": { + "size": 487, + "sha256": "bc0b230081b44e629dd5b9045b78495615c1831b4b9f4cffe97bd37e82a6156a" + }, + "tokenizer.json": { + "size": 11429653, + "sha256": "fe1fad59be22a41ee293363fcf95fdedbc7c93f3b49270b1d2e18bd1399a7a05" + }, + "tokenizer_config.json": { + "size": 998, + "sha256": "945e980986de2ca7768f3326bfdbb4fbea3406f972b8ae0be233089f2b253c11" + } + } + } +} diff --git a/src/wechat_decrypt_tool/routers/chat_media.py b/src/wechat_decrypt_tool/routers/chat_media.py index 13748be0..64448a40 100644 --- a/src/wechat_decrypt_tool/routers/chat_media.py +++ b/src/wechat_decrypt_tool/routers/chat_media.py @@ -125,12 +125,12 @@ class VoiceTranscriptionCacheLookupRequest(BaseModel): class VoiceTranscriptionSettingsRequest(BaseModel): device: Optional[str] = Field(None, description="推理设备:cpu 或 cuda") - model: Optional[str] = Field(None, description="Whisper 模型") + model: Optional[str] = Field(None, description="本地语音模型") class VoiceTranscriptionBatchRequest(BaseModel): account: Optional[str] = Field(None, description="账号目录名") - force: bool = Field(False, description="忽略现有 Whisper 缓存并重新识别") + force: bool = Field(False, description="忽略现有本地模型缓存并重新识别") engine: StrictStr = Field("local", description="批量转写方式:local 或 wechat-native") concurrency: Optional[conint(strict=True, ge=0)] = Field( # type: ignore[valid-type] None, @@ -3742,12 +3742,12 @@ async def get_chat_voice(server_id: int, account: Optional[str] = None): ) -@router.get("/api/chat/media/voice/transcription/status", summary="检查本地 Whisper 语音转文字能力") +@router.get("/api/chat/media/voice/transcription/status", summary="检查本地语音转文字能力") async def get_chat_voice_transcription_status(): return await asyncio.to_thread(get_voice_transcription_service().status) -@router.put("/api/chat/media/voice/transcription/settings", summary="设置本地 Whisper 模型或推理设备") +@router.put("/api/chat/media/voice/transcription/settings", summary="设置本地语音模型或推理设备") async def set_chat_voice_transcription_settings(req: VoiceTranscriptionSettingsRequest, request: Request): _require_local_voice_mutation(request) device = str(req.device or "").strip() @@ -3769,7 +3769,7 @@ async def set_chat_voice_transcription_settings(req: VoiceTranscriptionSettingsR return {"status": "success", "configuration": configuration} -@router.post("/api/chat/media/voice/transcription/models/{model}/download", summary="下载 Whisper 模型") +@router.post("/api/chat/media/voice/transcription/models/{model}/download", summary="下载本地语音模型") async def download_chat_voice_transcription_model(model: str, request: Request): _require_local_voice_mutation(request) try: @@ -3782,7 +3782,7 @@ async def download_chat_voice_transcription_model(model: str, request: Request): ) from exc -@router.get("/api/chat/media/voice/transcription/models/downloads/{job_id}", summary="查询 Whisper 模型下载任务") +@router.get("/api/chat/media/voice/transcription/models/downloads/{job_id}", summary="查询语音模型下载任务") async def get_chat_voice_transcription_model_download(job_id: str): try: return VOICE_MODEL_DOWNLOAD_MANAGER.get(job_id) @@ -3793,7 +3793,7 @@ async def get_chat_voice_transcription_model_download(job_id: str): ) from exc -@router.delete("/api/chat/media/voice/transcription/models/{model}", summary="删除 Whisper 模型") +@router.delete("/api/chat/media/voice/transcription/models/{model}", summary="删除本地语音模型") async def delete_chat_voice_transcription_model(model: str, request: Request): _require_local_voice_mutation(request) try: diff --git a/src/wechat_decrypt_tool/voice_transcription.py b/src/wechat_decrypt_tool/voice_transcription.py index 0619bc33..918c16f8 100644 --- a/src/wechat_decrypt_tool/voice_transcription.py +++ b/src/wechat_decrypt_tool/voice_transcription.py @@ -22,6 +22,7 @@ from dataclasses import dataclass, replace from pathlib import Path from typing import Any, Callable, Optional +from types import SimpleNamespace import httpx @@ -34,6 +35,30 @@ _CUDA_PROBE_CACHE_TTL_SECONDS = 5.0 _CUDA_PROBE_CACHE_LOCK = threading.Lock() _CUDA_PROBE_CACHE: Optional[tuple[float, dict[str, Any]]] = None +_VOICE_CUDA_DLL_HANDLES: list[Any] = [] +_VOICE_CUDA_DLL_LOCK = threading.Lock() + + +def _prepare_whisper_cuda_libraries() -> None: + """Windows 下复用可选 GPU 组件自带的 CUDA 12 库,避免只装 CUDA 13 时回退。""" + if os.name != "nt": + return + with _VOICE_CUDA_DLL_LOCK: + if _VOICE_CUDA_DLL_HANDLES: + return + try: + spec = importlib.util.find_spec("torch") + if not spec or not spec.origin: + return + folder = Path(spec.origin).parent / "lib" + if not (folder / "cublas64_12.dll").is_file(): + return + # CTranslate2 使用 LoadLibrary;PATH 和 Python DLL 搜索目录都需要设置。 + handle = os.add_dll_directory(str(folder)) + os.environ["PATH"] = str(folder) + os.pathsep + os.environ.get("PATH", "") + _VOICE_CUDA_DLL_HANDLES.append(handle) + except (ImportError, OSError, ValueError): + logger.debug("可选 CUDA 库目录不可用,继续使用系统运行库。", exc_info=True) from .runtime_settings import ( VOICE_TRANSCRIPTION_DEVICE_CPU, @@ -44,6 +69,11 @@ write_voice_transcription_model_setting, ) from .app_paths import get_data_dir, get_output_databases_dir, get_output_dir +from .asr_models import ( + SPECS as ASR_MODEL_SPECS, NEW_MODEL_CATALOG, cache_identity, + dependency_status, model_files_ready, verify_model_files, +) +from .asr_worker import AsrCancelled, AsrError, ProcessBackend, probe_qwen_cuda VOICE_MODEL_CATALOG: tuple[dict[str, Any], ...] = ( @@ -97,6 +127,15 @@ "description": "Large v3 的高速版本,推荐 NVIDIA GPU。", }, ) +_WHISPER_CATALOG = VOICE_MODEL_CATALOG +VOICE_MODEL_CATALOG = ( + *NEW_MODEL_CATALOG[:2], + next(item for item in _WHISPER_CATALOG if item["id"] == "turbo"), + *NEW_MODEL_CATALOG[2:], + *({**item, "legacy": True, "recommended": False, + "description": "保留原有 Whisper 识别方式,已有设置和缓存继续可用。", + "quality": "兼容模型"} for item in _WHISPER_CATALOG if item["id"] != "turbo"), +) VOICE_MODEL_IDS = frozenset(str(item["id"]) for item in VOICE_MODEL_CATALOG) VOICE_MODEL_STORAGE_DIRNAME = "voice_models" VOICE_MODEL_REPOSITORIES: dict[str, str] = { @@ -107,6 +146,7 @@ "large-v3": "Systran/faster-whisper-large-v3", "turbo": "mobiuslabsgmbh/faster-whisper-large-v3-turbo", } +VOICE_MODEL_REPOSITORIES.update({key: value["repo"] for key, value in ASR_MODEL_SPECS.items()}) VOICE_MODEL_DOWNLOAD_ALLOW_PATTERNS = ( "config.json", "preprocessor_config.json", @@ -155,14 +195,14 @@ def get_legacy_voice_model_storage_root() -> Path: def _managed_voice_model_dir(model: str) -> Path: model_id = str(model or "").strip() if model_id not in VOICE_MODEL_IDS: - raise VoiceTranscriptionError("invalid_model", "不支持该 Whisper 模型。") + raise VoiceTranscriptionError("invalid_model", "不支持该语音模型。") return get_voice_model_storage_root() / model_id def _legacy_voice_model_dir(model: str) -> Path: model_id = str(model or "").strip() if model_id not in VOICE_MODEL_IDS: - raise VoiceTranscriptionError("invalid_model", "不支持该 Whisper 模型。") + raise VoiceTranscriptionError("invalid_model", "不支持该语音模型。") return get_legacy_voice_model_storage_root() / model_id @@ -594,9 +634,14 @@ def from_env(cls) -> "VoiceTranscriptionConfig": model, model_source = read_effective_voice_transcription_model() language = str(os.environ.get("WECHAT_TOOL_WHISPER_LANGUAGE") or "zh").strip() or "zh" device, device_source = read_effective_voice_transcription_device() + spec = ASR_MODEL_SPECS.get(model) + if spec and device_source == "default": + device = spec["devices"][0] compute_type = str(os.environ.get("WECHAT_TOOL_WHISPER_COMPUTE_TYPE") or "").strip() if not compute_type: compute_type = "float16" if device == VOICE_TRANSCRIPTION_DEVICE_CUDA else "int8" + if spec: + compute_type = spec["precision"] allow_download = _env_bool("WECHAT_TOOL_WHISPER_ALLOW_DOWNLOAD", False) try: beam_size = max(1, min(10, int(os.environ.get("WECHAT_TOOL_WHISPER_BEAM_SIZE") or 5))) @@ -740,7 +785,9 @@ def _public_model_name(value: str) -> str: return raw -def _model_directory_is_ready(path: Path) -> bool: +def _model_directory_is_ready(path: Path, model: str = "") -> bool: + if model in ASR_MODEL_SPECS: + return model_files_ready(path, model) try: if not path.is_dir(): return False @@ -823,7 +870,7 @@ def inspect_model_readiness(model: str) -> dict[str, Any]: managed_dir = _managed_voice_model_dir(raw) managed_root = get_voice_model_storage_root() managed_owned = _managed_model_path_is_owned(managed_root, managed_dir) - if managed_owned and _model_directory_is_ready(managed_dir): + if managed_owned and _model_directory_is_ready(managed_dir, raw): return { "ready": True, "downloadable": True, @@ -854,7 +901,7 @@ def inspect_model_readiness(model: str) -> dict[str, Any]: same_location = legacy_dir.resolve() == managed_dir.resolve() except OSError: same_location = legacy_dir.absolute() == managed_dir.absolute() - if not same_location and _model_directory_is_ready(legacy_dir): + if not same_location and _model_directory_is_ready(legacy_dir, raw): return { "ready": True, "downloadable": True, @@ -864,6 +911,9 @@ def inspect_model_readiness(model: str) -> dict[str, Any]: "reason": "检测到旧 output 目录中的模型;可继续使用,但本应用不会在此处删除它。", } + if raw in ASR_MODEL_SPECS: + return dict(ready=False, downloadable=True, managed=managed_partial, deletable=managed_partial, + source="app-cache", reason="模型尚未下载完整,请在模型列表中下载。") try: from faster_whisper.utils import download_model except Exception: @@ -965,8 +1015,14 @@ def get_voice_model_catalog(*, selected_model: Optional[str] = None) -> list[dic readiness = inspect_model_readiness(model_id) job = jobs.get(model_id) or {} item = dict(definition) + runtime_ready, runtime_reason = dependency_status(model_id) + spec = ASR_MODEL_SPECS.get(model_id, {}) item.update( { + "backend": spec.get("backend", "whisper"), + "devices": spec.get("devices", ["cpu", "cuda"]), + "runtimeAvailable": runtime_ready, + "runtimeReason": runtime_reason, "selected": model_id == selected, "downloaded": bool(readiness.get("ready")), "downloadable": bool(readiness.get("downloadable")), @@ -1464,7 +1520,8 @@ def __init__( model_loader: Optional[Callable[[VoiceTranscriptionConfig], Any]] = None, ) -> None: self.config = config or VoiceTranscriptionConfig.from_env() - self._model_loader = model_loader or self._load_faster_whisper_model + self._model_loader = model_loader or self._load_backend_model + self._cache_model = cache_identity(self.config.model) self._model: Any = None self._active_device = "" self._active_compute_type = "" @@ -1475,7 +1532,7 @@ def __init__( self._active_inferences = 0 self._model_transitioning = False self._model_generation = 0 - self._model_num_workers = max(1, int(self.config.num_workers or 1)) + self._model_num_workers = 1 if self.config.model in ASR_MODEL_SPECS else max(1, int(self.config.num_workers or 1)) self._retired = False # Service replacement can overlap with a completed inference writing its # cache, so all service generations must serialize the SQLite file. @@ -1498,6 +1555,10 @@ def status(self) -> dict[str, Any]: dependency_available = importlib.util.find_spec("faster_whisper") is not None except Exception: dependency_available = False + spec = ASR_MODEL_SPECS.get(self.config.model) + runtime_reason = "" + if spec: + dependency_available, runtime_reason = dependency_status(self.config.model) try: text_normalizer_available = importlib.util.find_spec("opencc") is not None except Exception: @@ -1505,14 +1566,19 @@ def status(self) -> dict[str, Any]: model_readiness = self._model_readiness() model_ready = bool(model_readiness.get("ready")) model_downloadable = bool(model_readiness.get("downloadable")) - can_prepare_model = bool(self.config.allow_download and model_downloadable) + can_prepare_model = bool(not spec and self.config.allow_download and model_downloadable) cuda = probe_cuda() + if spec and spec["backend"] == "qwen-hf": + cuda = probe_qwen_cuda() if dependency_available else dict(available=False, deviceCount=0, devices=[], reason=runtime_reason) + device_supported = not spec or self.config.device in spec["devices"] + device_ready = device_supported and (not spec or self.config.device != "cuda" or cuda["available"]) fallback_reason = self._fallback_reason - if not fallback_reason and self.config.device == VOICE_TRANSCRIPTION_DEVICE_CUDA and not cuda["available"]: + if not spec and not fallback_reason and self.config.device == VOICE_TRANSCRIPTION_DEVICE_CUDA and not cuda["available"]: fallback_reason = f"{cuda['reason']} 首次识别会自动回退到 CPU。" available = bool( self.config.enabled and dependency_available + and device_ready and text_normalizer_available and self.config.model and (model_ready or can_prepare_model) @@ -1521,11 +1587,15 @@ def status(self) -> dict[str, Any]: if not self.config.enabled: reason = "语音转文字功能未启用。" elif not dependency_available: - reason = "未安装 faster-whisper,请安装语音转文字可选依赖。" + reason = runtime_reason or "未安装 faster-whisper,请安装语音转文字可选依赖。" elif not text_normalizer_available: reason = "未安装 OpenCC,无法保证输出为简体中文。" elif not str(self.config.model or "").strip(): reason = "未配置 Whisper 模型。" + elif not device_supported: + reason = "当前模型不支持所选设备,请在设置中选择匹配的模型和设备。" + elif not device_ready: + reason = str(cuda.get("reason") or "当前模型所需的 GPU 不可用,请选择 CPU 模型。") elif not model_ready: reason = str(model_readiness.get("reason") or "Whisper 模型尚未准备好。") if can_prepare_model: @@ -1541,6 +1611,8 @@ def status(self) -> dict[str, Any]: "modelSource": str(model_readiness.get("source") or "unavailable"), "modelDownloadRequired": bool(not model_ready and can_prepare_model), "model": _public_model_name(self.config.model), + "backend": spec["backend"] if spec else "whisper", + "supportedDevices": spec["devices"] if spec else ["cpu", "cuda"], "modelSettingSource": self.config.model_source, "models": get_voice_model_catalog(selected_model=self.config.model), "language": self.config.language, @@ -1551,11 +1623,11 @@ def status(self) -> dict[str, Any]: "deviceSource": self.config.device_source, "activeDevice": self._active_device or None, "activeComputeType": self._active_compute_type or None, - "modelLoaded": self._model is not None, + "modelLoaded": self._model is not None and getattr(self._model, "is_loaded", True), "numWorkers": self._model_num_workers, "cuda": cuda, "requestedDeviceAvailable": bool( - self.config.device != VOICE_TRANSCRIPTION_DEVICE_CUDA or cuda["available"] + device_supported and (self.config.device != VOICE_TRANSCRIPTION_DEVICE_CUDA or cuda["available"]) ), "usingFallback": bool(self._fallback_reason), "fallbackReason": fallback_reason, @@ -1642,7 +1714,7 @@ def _transcribe_voice_impl( str(account_path.absolute()), sid, source_hash, - str(self.config.model), + self._cache_model, str(self.config.language), ) with _voice_transcript_singleflight(flight_key, cancel_event): @@ -1654,7 +1726,7 @@ def _transcribe_voice_impl( self._raise_if_cancelled(cancel_event) payload, ext, _media_type = _convert_silk_to_browser_audio(data, preferred_format="wav") if not payload or ext == "silk": - raise VoiceTranscriptionError("voice_decode_failed", "语音解码失败,无法交给 Whisper 识别。") + raise VoiceTranscriptionError("voice_decode_failed", "语音解码失败,无法进行本地识别。") temp_path: Optional[Path] = None try: @@ -1716,6 +1788,8 @@ def configure_inference_concurrency( """Reload the model at a quiescent point with matching CTranslate2 workers.""" workers = max(1, int(concurrency or 1)) + if self.config.model in ASR_MODEL_SPECS: + workers = 1 with self._inference_condition: while self._model_transitioning and not self._retired: self._raise_if_cancelled(cancel_event) @@ -1815,6 +1889,7 @@ def _transcribe_with_fallback( try: try: text, info = self._transcribe_once(model, path, cancel_event=cancel_event) + compute_type = getattr(model, "precision", compute_type) except _VoiceTranscriptionCancelled as exc: try: exc.__traceback__ = None @@ -1831,7 +1906,7 @@ def _transcribe_with_fallback( raise except Exception as exc: inference_error_type = type(exc).__name__ - if device == VOICE_TRANSCRIPTION_DEVICE_CUDA and _is_cuda_runtime_error(exc): + if self.config.model not in ASR_MODEL_SPECS and device == VOICE_TRANSCRIPTION_DEVICE_CUDA and _is_cuda_runtime_error(exc): cuda_fallback_required = True with self._inference_condition: self._cuda_fallback_pending = True @@ -1922,6 +1997,17 @@ def _transcribe_once( cancel_event: Optional[threading.Event] = None, ) -> tuple[str, Any]: self._raise_if_cancelled(cancel_event) + if self.config.model in ASR_MODEL_SPECS: + try: + result = model.transcribe_audio(str(path), self.config.language, cancel_event) + except AsrCancelled: + raise _VoiceTranscriptionCancelled() from None + except AsrError as exc: + raise VoiceTranscriptionError(exc.code, str(exc)) from exc + self._raise_if_cancelled(cancel_event) + self._active_compute_type = result["precision"] + return normalize_transcript_text(result["text"]), SimpleNamespace( + language=result["language"], duration=result["duration"]) segments, info = model.transcribe( str(path), language=self.config.language, @@ -1946,6 +2032,8 @@ def _transcribe_once( return normalize_transcript_text(text), info def _release_loaded_model_unlocked(self) -> None: + if isinstance(self._model, ProcessBackend): + self._model.close() self._model = None self._active_device = "" self._active_compute_type = "" @@ -1978,6 +2066,9 @@ def _get_model(self) -> Any: return self._model runtime_config = replace(self.config, num_workers=self._model_num_workers) + if runtime_config.model in ASR_MODEL_SPECS: + # Qwen GPU 不会静默改用另一个 CPU 模型,错误交由用户选择处理。 + return self._load_model(runtime_config) if runtime_config.device == VOICE_TRANSCRIPTION_DEVICE_CUDA: if self._cuda_fallback_pending: return self._load_cpu_fallback(self._fallback_reason) @@ -2017,11 +2108,30 @@ def _load_model(self, config: VoiceTranscriptionConfig) -> Any: except Exception as exc: raise VoiceTranscriptionError( "model_load_failed", - f"Whisper 模型加载失败:{type(exc).__name__}", + f"语音模型加载失败:{type(exc).__name__}", ) from exc + @staticmethod + def _load_backend_model(config: VoiceTranscriptionConfig) -> Any: + spec = ASR_MODEL_SPECS.get(config.model) + if not spec: + return VoiceTranscriptionService._load_faster_whisper_model(config) + if config.device not in spec["devices"]: + raise VoiceTranscriptionError("invalid_device", "当前模型不支持所选设备,请选择对应的 CPU 或 GPU 模型。") + available, reason = dependency_status(config.model) + if not available: + raise VoiceTranscriptionError("dependency_missing", reason) + folder = _managed_voice_model_dir(config.model) + if not _managed_model_path_is_owned(get_voice_model_storage_root(), folder) or not _model_directory_is_ready(folder, config.model): + folder = _legacy_voice_model_dir(config.model) + if not _model_directory_is_ready(folder, config.model): + raise VoiceTranscriptionError("model_not_ready", "模型文件未准备好,请先在设置中下载模型。") + return ProcessBackend(spec["backend"], str(folder), spec["precision"]) + @staticmethod def _load_faster_whisper_model(config: VoiceTranscriptionConfig) -> Any: + if config.device == VOICE_TRANSCRIPTION_DEVICE_CUDA: + _prepare_whisper_cuda_libraries() try: from faster_whisper import WhisperModel except ImportError as exc: @@ -2084,7 +2194,7 @@ def _read_cache(self, account_dir: Path, server_id: int, source_hash: str) -> Op row = conn.execute( "SELECT text, detected_language, duration, text_version FROM transcript " "WHERE server_id = ? AND source_hash = ? AND model = ? AND language = ? LIMIT 1", - (int(server_id), source_hash, self.config.model, self.config.language), + (int(server_id), source_hash, self._cache_model, self.config.language), ).fetchone() if row: normalized_text, needs_update = self._normalize_cached_text(row[0], row[3]) @@ -2098,7 +2208,7 @@ def _read_cache(self, account_dir: Path, server_id: int, source_hash: str) -> Op time.time(), int(server_id), source_hash, - self.config.model, + self._cache_model, self.config.language, ), ) @@ -2163,7 +2273,7 @@ def lookup_cached_transcripts( "SELECT server_id, source_hash, text, detected_language, duration, text_version FROM transcript " f"WHERE model = ? AND language = ? AND server_id IN ({placeholders}) " "ORDER BY updated_at DESC", - (self.config.model, self.config.language, *ids), + (self._cache_model, self.config.language, *ids), ).fetchall() normalized_rows = [] for row in rows or []: @@ -2178,7 +2288,7 @@ def lookup_cached_transcripts( time.time(), int(row[0]), str(row[1]), - self.config.model, + self._cache_model, self.config.language, ), ) @@ -2243,7 +2353,7 @@ def _write_cache( ( int(server_id), source_hash, - self.config.model, + self._cache_model, self.config.language, normalized_text, str(result.get("language") or self.config.language), @@ -2277,6 +2387,9 @@ def _download_voice_model_snapshot( # A cancelled worker must not wait for other executor workers to drain. "max_workers": 1, } + if model_id in ASR_MODEL_SPECS: + spec = ASR_MODEL_SPECS[model_id] + common.update(revision=spec["revision"], allow_patterns=list(spec["files"])) class SilentTqdm(base_tqdm): def __init__(self, *args: Any, **kwargs: Any) -> None: @@ -2461,7 +2574,7 @@ def get(self, job_id: str) -> dict[str, Any]: def start(self, model: str) -> dict[str, Any]: model_id = str(model or "").strip() if model_id not in VOICE_MODEL_IDS: - raise VoiceTranscriptionError("invalid_model", "不支持该 Whisper 模型。") + raise VoiceTranscriptionError("invalid_model", "不支持该语音模型。") activity_key = "" cancel_event = threading.Event() completion_event = threading.Event() @@ -2522,7 +2635,7 @@ def begin_delete(self, model: str) -> bool: model_id = str(model or "").strip() if model_id not in VOICE_MODEL_IDS: - raise VoiceTranscriptionError("invalid_model", "不支持该 Whisper 模型。") + raise VoiceTranscriptionError("invalid_model", "不支持该语音模型。") job_id = "" completion_event: Optional[threading.Event] = None @@ -2634,7 +2747,7 @@ def _run( "model_download_refused", "拒绝写入应用模型目录之外的路径。", ) - if _model_directory_is_ready(model_dir): + if _model_directory_is_ready(model_dir, model_id): self._update(job_id, status="done", stage="done", percent=100, finishedAt=time.time()) return @@ -2668,8 +2781,13 @@ def update_progress(**progress: Any) -> None: total_bytes=0, force=True, ) - if downloaded_dir.resolve() != stage_dir.resolve() or not _model_directory_is_ready(stage_dir): + if downloaded_dir.resolve() != stage_dir.resolve() or not _model_directory_is_ready(stage_dir, model_id): raise VoiceTranscriptionError("model_download_incomplete", "模型下载完成,但缓存文件不完整。") + if model_id in ASR_MODEL_SPECS: + try: + verify_model_files(stage_dir, model_id, lambda: update_progress(stage="verifying", downloaded_bytes=0, total_bytes=0)) + except ValueError as exc: + raise VoiceTranscriptionError("model_download_corrupt", str(exc)) from exc if cancel_event.is_set(): raise _VoiceModelDownloadCancelled() self._update_progress( @@ -2692,7 +2810,7 @@ def update_progress(**progress: Any) -> None: stage_dir = None if cancel_event.is_set(): raise _VoiceModelDownloadCancelled() - if not _model_directory_is_ready(model_dir): + if not _model_directory_is_ready(model_dir, model_id): raise VoiceTranscriptionError("model_download_incomplete", "模型下载完成,但缓存文件不完整。") current_service = get_voice_transcription_service() if current_service.config.model == model_id: @@ -2775,6 +2893,9 @@ def resolve_voice_transcription_batch_concurrency( ) else 1 else: effective = requested_value + if config.model in ASR_MODEL_SPECS: + # 新后端串行复用单个进程,避免多份模型挤占低配内存或显存。 + effective = 1 return requested_value, effective @@ -3437,6 +3558,9 @@ def set_voice_transcription_device(device: str) -> dict[str, Any]: current_service = get_voice_transcription_service() current_model = current_service.config.model + spec = ASR_MODEL_SPECS.get(current_model) + if spec and normalized not in spec["devices"]: + raise VoiceTranscriptionError("invalid_device", "当前模型不支持该设备,请先选择对应的 CPU 或 GPU 模型。") _begin_voice_model_deletion(current_model) try: write_voice_transcription_device_setting(normalized) @@ -3451,7 +3575,7 @@ def set_voice_transcription_model(model: str) -> dict[str, Any]: normalized = str(model or "").strip() if normalized not in VOICE_MODEL_IDS: - raise VoiceTranscriptionError("invalid_model", "不支持该 Whisper 模型。") + raise VoiceTranscriptionError("invalid_model", "不支持该语音模型。") _configured, source = read_effective_voice_transcription_model() if source == "env": @@ -3462,8 +3586,24 @@ def set_voice_transcription_model(model: str) -> dict[str, Any]: current_service = get_voice_transcription_service() current_model = current_service.config.model + spec = ASR_MODEL_SPECS.get(normalized) + target_device = None + if spec: + target_device = spec["devices"][0] + configured_device, device_source = read_effective_voice_transcription_device() + if device_source == "env" and configured_device != target_device: + raise VoiceTranscriptionError("device_locked", "启动环境变量固定的设备与此模型不兼容,请选择匹配的模型。") + ready, reason = dependency_status(normalized) + if not ready: + raise VoiceTranscriptionError("dependency_missing", reason) + if not inspect_model_readiness(normalized)["ready"]: + raise VoiceTranscriptionError("model_not_ready", "请先下载完整模型,再选择使用。") + if target_device == "cuda" and not probe_qwen_cuda()["available"]: + raise VoiceTranscriptionError("gpu_unavailable", "Qwen GPU 运行环境不可用,请检查 GPU 组件和驱动,或选择 CPU 模型。") _begin_voice_model_deletion(current_model) try: + if target_device is not None: + write_voice_transcription_device_setting(target_device) write_voice_transcription_model_setting(normalized) return _reset_voice_transcription_service().status() finally: @@ -3475,7 +3615,7 @@ def delete_voice_model(model: str) -> dict[str, Any]: model_id = str(model or "").strip() if model_id not in VOICE_MODEL_IDS: - raise VoiceTranscriptionError("invalid_model", "不支持该 Whisper 模型。") + raise VoiceTranscriptionError("invalid_model", "不支持该语音模型。") if VOICE_TRANSCRIPTION_BATCH_MANAGER.has_active_model(model_id): raise VoiceTranscriptionError("model_busy", "该模型正在用于批量转写,暂时不能删除。") diff --git a/tests/test_asr_upgrade.py b/tests/test_asr_upgrade.py new file mode 100644 index 00000000..9e0ec213 --- /dev/null +++ b/tests/test_asr_upgrade.py @@ -0,0 +1,196 @@ +"""新后端的资产发布、配置兼容、缓存隔离和进程取消回归。""" +import hashlib +import threading +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import Mock + +import numpy as np +import pytest + +from wechat_decrypt_tool import asr_models as assets +from wechat_decrypt_tool import voice_transcription as voice +from wechat_decrypt_tool.asr_backends import audio_chunks, log_mel, mel_filters +from wechat_decrypt_tool.asr_worker import AsrCancelled, AsrError, ProcessBackend + +CTC = "zipformer-small-ctc-int8" +CPU = "qwen3-asr-06b-onnx-int4" +GPU = "qwen3-asr-06b-hf" + + +@pytest.fixture +def small_asset(monkeypatch): + files = {"ctc.int8.onnx": b"model", "data/tokens.txt": b"tokens"} + spec = {**assets.SPECS[CTC], "files": {name: dict(size=len(data), sha256=hashlib.sha256(data).hexdigest()) + for name, data in files.items()}} + monkeypatch.setitem(assets.SPECS, CTC, spec) + return files + + +def write_assets(path, files): + for name, data in files.items(): + target = path / name + target.parent.mkdir(parents=True, exist_ok=True) + target.write_bytes(data) + + +def test_new_asset_readiness_requires_exact_files(tmp_path, small_asset): + assert not voice._model_directory_is_ready(tmp_path, CTC) + write_assets(tmp_path, small_asset) + assert voice._model_directory_is_ready(tmp_path, CTC) + (tmp_path / "ctc.int8.onnx").write_bytes(b"bad") + assert not voice._model_directory_is_ready(tmp_path, CTC) + + +@pytest.mark.parametrize("corrupt", [False, True]) +def test_download_verifies_hash_before_publish(tmp_path, monkeypatch, small_asset, corrupt): + root = tmp_path / "models" + monkeypatch.setattr(voice, "get_voice_model_storage_root", lambda: root) + monkeypatch.setattr(voice, "get_voice_transcription_service", lambda: SimpleNamespace(config=voice.VoiceTranscriptionConfig())) + def download(model, *, output_dir, progress_callback): + write_assets(output_dir, small_asset) + if corrupt: + (output_dir / "ctc.int8.onnx").write_bytes(b"wrong") + return output_dir + monkeypatch.setattr(voice, "_download_voice_model_snapshot", download) + manager = voice.VoiceModelDownloadManager() + job = manager.start(CTC) + assert manager._completion_events[job["jobId"]].wait(5) + result = manager.get(job["jobId"]) + assert result["status"] == ("error" if corrupt else "done") + assert (root / CTC).exists() is not corrupt + if corrupt: + assert "校验失败" in result["error"] + + +def test_download_uses_pinned_revision_and_variant_files(tmp_path, monkeypatch): + calls = [] + def snapshot(repo, **kwargs): + calls.append((repo, kwargs)) + return [] if kwargs.get("dry_run") else str(tmp_path) + monkeypatch.setattr("huggingface_hub.snapshot_download", snapshot) + voice._download_voice_model_snapshot(CPU, output_dir=tmp_path, progress_callback=lambda **_: None) + for repo, kwargs in calls: + assert repo == assets.SPECS[CPU]["repo"] + assert kwargs["revision"] == assets.SPECS[CPU]["revision"] + assert set(kwargs["allow_patterns"]) == set(assets.SPECS[CPU]["files"]) + assert not any("fp16" in name for name in kwargs["allow_patterns"]) + + +def test_cache_isolated_by_model_and_revision(tmp_path, monkeypatch): + result = dict(text="测试文本", language="zh", duration=1) + service = voice.VoiceTranscriptionService(voice.VoiceTranscriptionConfig(model=CTC)) + service._write_cache(tmp_path, 1, "hash", result) + assert service._read_cache(tmp_path, 1, "hash")["text"] == "测试文本" + assert service.lookup_cached_transcripts(tmp_path, [1])[1]["model"] == CTC + whisper = voice.VoiceTranscriptionService(voice.VoiceTranscriptionConfig(model="tiny")) + assert whisper._read_cache(tmp_path, 1, "hash") is None + monkeypatch.setitem(assets.SPECS, CTC, {**assets.SPECS[CTC], "revision": "updated"}) + updated = voice.VoiceTranscriptionService(voice.VoiceTranscriptionConfig(model=CTC)) + assert updated._read_cache(tmp_path, 1, "hash") is None + assert updated.lookup_cached_transcripts(tmp_path, [1]) == {} + + +def test_select_new_model_matches_device_without_remapping_legacy(monkeypatch): + monkeypatch.setattr(voice, "read_effective_voice_transcription_model", lambda: ("medium", "settings")) + monkeypatch.setattr(voice, "read_effective_voice_transcription_device", lambda: ("cpu", "settings")) + monkeypatch.setattr(voice, "get_voice_transcription_service", lambda: SimpleNamespace(config=voice.VoiceTranscriptionConfig())) + monkeypatch.setattr(voice, "dependency_status", lambda _: (True, "")) + monkeypatch.setattr(voice, "inspect_model_readiness", lambda _: dict(ready=True)) + monkeypatch.setattr(voice, "probe_qwen_cuda", lambda: dict(available=True)) + monkeypatch.setattr(voice, "_reset_voice_transcription_service", lambda: SimpleNamespace(status=lambda: {})) + save_model, save_device = Mock(), Mock() + monkeypatch.setattr(voice, "write_voice_transcription_model_setting", save_model) + monkeypatch.setattr(voice, "write_voice_transcription_device_setting", save_device) + voice.set_voice_transcription_model(GPU) + save_model.assert_called_once_with(GPU) + save_device.assert_called_once_with("cuda") + voice.set_voice_transcription_model("tiny") + assert save_device.call_count == 1 + assert save_model.call_args.args == ("tiny",) + + +def test_env_device_lock_rejects_incompatible_model(monkeypatch): + monkeypatch.setattr(voice, "read_effective_voice_transcription_model", lambda: ("medium", "settings")) + monkeypatch.setattr(voice, "read_effective_voice_transcription_device", lambda: ("cpu", "env")) + monkeypatch.setattr(voice, "get_voice_transcription_service", lambda: SimpleNamespace(config=voice.VoiceTranscriptionConfig())) + with pytest.raises(voice.VoiceTranscriptionError, match="不兼容") as caught: + voice.set_voice_transcription_model(GPU) + assert caught.value.code == "device_locked" + + +def test_qwen_gpu_failure_does_not_fall_back_to_cpu(monkeypatch): + fake = SimpleNamespace(transcribe_audio=Mock(side_effect=AsrError("gpu_unavailable", "CUDA unavailable"))) + loader = Mock(return_value=fake) + service = voice.VoiceTranscriptionService(voice.VoiceTranscriptionConfig(model=GPU, device="cuda"), model_loader=loader) + fallback = Mock() + monkeypatch.setattr(service, "_load_cpu_fallback", fallback) + with pytest.raises(voice.VoiceTranscriptionError) as caught: + service._transcribe_with_fallback(Path("test.wav"), cancel_event=None) + assert caught.value.code == "gpu_unavailable" + assert not fallback.called + assert loader.call_count == 1 + service.retire() + + +def test_gpu_status_uses_torch_probe_not_ctranslate(monkeypatch): + monkeypatch.setattr(voice, "dependency_status", lambda _: (True, "")) + monkeypatch.setattr(voice, "probe_cuda", lambda: dict(available=True, devices=[], reason="")) + monkeypatch.setattr(voice, "probe_qwen_cuda", lambda: dict(available=False, devices=[], reason="PyTorch CUDA missing")) + monkeypatch.setattr(voice, "get_voice_model_catalog", lambda **_: []) + service = voice.VoiceTranscriptionService(voice.VoiceTranscriptionConfig(model=GPU, device="cuda")) + monkeypatch.setattr(service, "_model_readiness", lambda: dict(ready=True)) + status = service.status() + assert not status["available"] + assert not status["usingFallback"] + assert "PyTorch" in status["reason"] + + +@pytest.mark.parametrize("model", [CTC, CPU, GPU, "qwen3-asr-17b-hf"]) +def test_new_backends_bound_concurrency(model): + config = voice.VoiceTranscriptionConfig(model=model, num_workers=99) + assert voice.resolve_voice_transcription_batch_concurrency(99, config) == (99, 1) + service = voice.VoiceTranscriptionService(config) + assert service.configure_inference_concurrency(99) == 1 + + +def test_long_audio_chunks_preserve_every_sample_and_bound_length(): + audio = np.linspace(-0.2, 0.2, 16000 * 61, dtype=np.float32) + chunks = list(audio_chunks(audio, 15)) + assert all(0 < len(x) <= 15 * 16000 for x in chunks) + np.testing.assert_array_equal(np.concatenate(chunks), audio) + + +def test_silence_features_are_finite_and_expected_shape(): + features = log_mel(np.zeros(16000, dtype=np.float32), mel_filters()) + assert features.shape == (1, 128, 100) + assert np.isfinite(features).all() + np.testing.assert_allclose(features, -1.5) + + +def test_cancel_terminates_worker_during_pending_inference(monkeypatch): + backend = ProcessBackend("zipformer", "unused", "int8") + cancelled = threading.Event() + connection = Mock() + connection.poll.side_effect = lambda _: (cancelled.set() or False) + backend.connection = connection + backend.process = Mock() + monkeypatch.setattr(backend, "_start", lambda: None) + close = Mock() + monkeypatch.setattr(backend, "close", close) + with pytest.raises(AsrCancelled): + backend.transcribe_audio("test.wav", "zh", cancelled) + close.assert_called_once() + + +def test_close_reaps_worker_and_releases_connection(): + backend = ProcessBackend("zipformer", "unused", "int8") + process = Mock(pid=123) + process.is_alive.side_effect = [True, False] + connection = Mock() + backend.process, backend.connection = process, connection + backend.close() + process.terminate.assert_called_once() + process.join.assert_called_once_with(timeout=5) + connection.close.assert_called_once() + assert backend.process is None and backend.connection is None diff --git a/tests/test_voice_transcription_manager.py b/tests/test_voice_transcription_manager.py index c9b98480..50f4fecd 100644 --- a/tests/test_voice_transcription_manager.py +++ b/tests/test_voice_transcription_manager.py @@ -66,7 +66,10 @@ def test_catalog_lists_multilingual_models_and_local_readiness(self): ): models = get_voice_model_catalog(selected_model="medium") - self.assertEqual([item["id"] for item in models], ["tiny", "base", "small", "medium", "large-v3", "turbo"]) + self.assertEqual([item["id"] for item in models], [ + "zipformer-small-ctc-int8", "qwen3-asr-06b-onnx-int4", "turbo", + "qwen3-asr-06b-hf", "qwen3-asr-17b-hf", "tiny", "base", "small", "medium", "large-v3", + ]) selected = next(item for item in models if item["id"] == "medium") self.assertTrue(selected["selected"]) self.assertTrue(selected["downloaded"]) diff --git a/tools/benchmark_stt_local.py b/tools/benchmark_stt_local.py new file mode 100644 index 00000000..e86fa7b8 --- /dev/null +++ b/tools/benchmark_stt_local.py @@ -0,0 +1,230 @@ +"""在本机固定语音集上测 ASR,参考文本只用于事后评分,不传入模型。""" +from __future__ import annotations + +import argparse +import importlib.metadata +import json +import os +from pathlib import Path +import random +import statistics +import sys +import threading +import time +import unicodedata + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--root', type=Path, required=True) + parser.add_argument('--model', required=True) + parser.add_argument('--threads', type=int, default=4) + parser.add_argument('--rounds', type=int, default=2) + parser.add_argument('--limit', type=int, default=0) + parser.add_argument('--tag', default='') + parser.add_argument('--chunk-seconds', type=float, default=0) + args = parser.parse_args() + # 所有模型均先下载到本地;正式推理期间禁止模型库联网。 + os.environ.update(HF_HUB_OFFLINE='1', TRANSFORMERS_OFFLINE='1', HF_HUB_DISABLE_TELEMETRY='1', + OMP_NUM_THREADS=str(args.threads), MKL_NUM_THREADS=str(args.threads)) + import numpy as np + import psutil + import soundfile as sf + from opencc import OpenCC + from rapidfuzz.distance import Levenshtein + root = args.root.resolve() + samples = json.loads((root / 'manifest.private.json').read_text(encoding='utf-8'))['samples'] + if args.limit: + samples = samples[:args.limit] + audio = {s['id']: sf.read(s['path'], dtype='float32')[0] for s in samples} + converter = OpenCC('t2s') + + def normalize(text): + return ''.join(c for c in converter.convert(unicodedata.normalize('NFKC', text)).lower() + if unicodedata.category(c)[0] in 'LN') + + proc = psutil.Process() + baseline_rss = proc.memory_info().rss + peak_rss = [baseline_rss] + stop = threading.Event() + + def monitor(): + while not stop.wait(0.05): + try: + peak_rss[0] = max(peak_rss[0], proc.memory_info().rss) + except psutil.Error: + pass + + monitor_thread = threading.Thread(target=monitor, daemon=True) + monitor_thread.start() + load_start = time.perf_counter() + sync = lambda: None + details = {} + key = args.model + if key.startswith('whisper-'): + # 使用项目原有的 beam=5、中文、VAD 和关闭前文条件配置。 + from faster_whisper import WhisperModel + name, device = key[len('whisper-'):].rsplit('-', 1) + if device == 'cuda': + # CUDA DLL 仅添加到测试进程,不修改系统 PATH。 + torch_lib = Path(sys.prefix) / 'Lib/site-packages/torch/lib' + handles = [os.add_dll_directory(str(torch_lib))] if torch_lib.exists() else [] + os.environ['PATH'] = str(torch_lib) + os.pathsep + os.environ['PATH'] + model = WhisperModel(str(root / 'models' / ('whisper-' + name)), device=device, + compute_type='int8' if device == 'cpu' else 'float16', + cpu_threads=args.threads, num_workers=1, local_files_only=True) + def transcribe(waveform): + segments, _ = model.transcribe(waveform, language='zh', beam_size=5, + vad_filter=True, condition_on_previous_text=False) + return ''.join(s.text for s in segments) + details = dict(device=device, precision='int8' if device == 'cpu' else 'float16', + beam_size=5, vad_filter=True) + elif key.startswith('zipformer-'): + import sherpa_onnx + folder = root / 'models/zipformer' + common = dict(tokens=str(folder / 'data/tokens.txt'), num_threads=args.threads, + sample_rate=16000, feature_dim=80, provider='cpu') + if key == 'zipformer-ctc': + model = sherpa_onnx.OfflineRecognizer.from_zipformer_ctc( + model=str(folder / 'ctc.int8.onnx'), **common) + else: + model = sherpa_onnx.OfflineRecognizer.from_transducer( + encoder=str(folder / 'encoder.int8.onnx'), decoder=str(folder / 'decoder.onnx'), + joiner=str(folder / 'joiner.int8.onnx'), **common) + def transcribe(waveform): + stream = model.create_stream() + stream.accept_waveform(16000, waveform) + model.decode_stream(stream) + return stream.result.text + details = dict(device='cpu', precision='mixed-int8', decoder='greedy_search') + elif key == 'qwen-onnx-cpu': + import onnxruntime as ort + import librosa + from tokenizers import Tokenizer + sys.path.insert(0, str(root / 'qwen-onnx-source')) + from src.inference import greedy_decode_onnx + folder = root / 'models/qwen-onnx' + cfg = json.loads((folder / 'config.json').read_text()) + opts = ort.SessionOptions() + opts.intra_op_num_threads = args.threads + opts.inter_op_num_threads = 1 + sessions = {name: ort.InferenceSession(str(folder / (name + '.int4.onnx')), opts, + providers=['CPUExecutionProvider']) + for name in ['encoder', 'decoder_init', 'decoder_step']} + embedding = np.memmap(folder / 'embed_tokens.bin', mode='r', dtype=cfg['embed_tokens_dtype'], + shape=(cfg['decoder']['vocab_size'], cfg['decoder']['hidden_size'])) + class Embeddings: + def __getitem__(self, index): + return np.asarray(embedding[index], dtype=np.float32) + tokens = Tokenizer.from_file(str(folder / 'tokenizer.json')) + filters = librosa.filters.mel(sr=16000, n_fft=400, n_mels=128, fmin=0, fmax=8000, norm='slaney') + # 按实际分词器编码角色名;上游示例硬编码的 system/user ID 与本模型不符。 + prompt_prefix = tokens.encode('<|im_start|>system\n<|im_end|>\n<|im_start|>user\n<|audio_start|>', add_special_tokens=False).ids + prompt_suffix = tokens.encode('<|audio_end|><|im_end|>\n<|im_start|>assistant\nlanguage Chinese', add_special_tokens=False).ids + def transcribe(waveform): + stft = librosa.stft(waveform, n_fft=400, hop_length=160, window='hann', center=True, pad_mode='reflect') + mel = filters @ (np.abs(stft) ** 2) + mel = np.log10(np.maximum(mel, 1e-10)) + mel = (np.maximum(mel, mel.max() - 8) + 4) / 4 + features = sessions['encoder'].run(['audio_features'], {'mel': mel[None, :, :-1].astype(np.float32)})[0] + prompt = prompt_prefix + [cfg['special_tokens']['audio_pad_token_id']] * features.shape[1] + prompt_suffix + generated = greedy_decode_onnx(sessions, Embeddings(), features, prompt, max_tokens=512) + return tokens.decode(generated, skip_special_tokens=True).split('')[-1].strip() + details = dict(device='cpu', precision='fp32-encoder/int4-decoder', language='Chinese', max_tokens=512) + elif key in ('qwen-06-cuda', 'qwen-17-cuda'): + import torch + from transformers import AutoProcessor, AutoModelForMultimodalLM + assert torch.cuda.is_available(), '当前测试环境的 PyTorch CUDA 不可用' + torch.set_num_threads(args.threads) + folder = root / 'models' / key.removesuffix('-cuda') + processor = AutoProcessor.from_pretrained(folder, local_files_only=True) + model = AutoModelForMultimodalLM.from_pretrained(folder, dtype=torch.bfloat16, + attn_implementation='sdpa', local_files_only=True).to('cuda').eval() + sync = torch.cuda.synchronize + torch.cuda.reset_peak_memory_stats() + def transcribe(waveform): + with torch.inference_mode(): + inputs = processor.apply_transcription_request(audio=waveform, language='Chinese').to(model.device, model.dtype) + ids = model.generate(**inputs, max_new_tokens=512, do_sample=False) + generated = ids[:, inputs['input_ids'].shape[1]:] + return processor.decode(generated, return_format='transcription_only')[0] + details = dict(device='cuda', precision='bfloat16', language='Chinese', max_tokens=512, attention='sdpa') + else: + raise ValueError(key) + + if args.chunk_seconds: + original_transcribe = transcribe + def transcribe(waveform): + # 在窗口末端附近寻找低能量位置,限制导出模型的最大输入长度。 + remaining = waveform + texts = [] + maximum = int(args.chunk_seconds * 16000) + while len(remaining) > maximum: + candidates = range(int(maximum * 0.7), maximum - 320, 320) + cut = min(candidates, key=lambda p: float(np.mean(remaining[p:p+320] ** 2))) + texts.append(original_transcribe(remaining[:cut])) + remaining = remaining[cut:] + if len(remaining): + texts.append(original_transcribe(remaining)) + return ''.join(texts) + details['chunk_max_seconds'] = args.chunk_seconds + details['chunk_method'] = 'minimum RMS in last 30% of window' + sync() + load_seconds = time.perf_counter() - load_start + # 第一条单独预热;所有模型使用相同样本,不计入热运行速度。 + started = time.perf_counter() + warmup_text = transcribe(audio[samples[0]['id']]) + sync() + warmup_seconds = time.perf_counter() - started + print(json.dumps(dict(event='loaded', model=key, load_seconds=load_seconds, + first_inference_seconds=warmup_seconds)), flush=True) + rows = [] + result_dir = root / 'results' + result_dir.mkdir(exist_ok=True) + target = result_dir / (key + args.tag + '.private.json') + for round_index in range(args.rounds): + order = list(samples) + random.Random(20260921 + round_index).shuffle(order) + for sample in order: + sync() + started = time.perf_counter() + text = transcribe(audio[sample['id']]) + sync() + elapsed = time.perf_counter() - started + ref, hyp = normalize(sample['reference']), normalize(text) + row = dict(id=sample['id'], round=round_index, audio_seconds=sample['duration'], + seconds=elapsed, transcript=text, reference=sample['reference'], + edits=Levenshtein.distance(ref, hyp), reference_chars=len(ref)) + rows.append(row) + target.write_text(json.dumps(dict(model=key, complete=False, rows=rows), ensure_ascii=False, indent=2), encoding='utf-8') + print(json.dumps(dict(event='sample', model=key, round=round_index, + done=len(rows), seconds=round(elapsed, 3))), flush=True) + stop.set() + monitor_thread.join() + first_round = [x for x in rows if x['round'] == 0] + per_sample = [statistics.median([r['seconds'] for r in rows if r['id'] == s['id']]) for s in samples] + versions = {} + for package in ['faster-whisper', 'ctranslate2', 'sherpa-onnx', 'onnxruntime', 'torch', 'transformers', 'numpy']: + try: + versions[package] = importlib.metadata.version(package) + except importlib.metadata.PackageNotFoundError: + pass + result = dict(model=key, complete=True, sample_count=len(samples), rounds=args.rounds, + audio_seconds=sum(s['duration'] for s in samples), + hot_seconds=sum(per_sample), rtf=sum(per_sample)/sum(s['duration'] for s in samples), + p50_seconds=float(np.percentile(per_sample, 50)), p95_seconds=float(np.percentile(per_sample, 95)), + silver_cer=sum(r['edits'] for r in first_round)/max(1,sum(r['reference_chars'] for r in first_round)), + exact_matches=sum(r['edits']==0 for r in first_round), + load_seconds=load_seconds, first_inference_seconds=warmup_seconds, + peak_rss_bytes=peak_rss[0], baseline_rss_bytes=baseline_rss, + details=details, threads=args.threads, versions=versions, + reference_type='WeChat native ASR; not human verified; CER measures disagreement, not proven error rate', rows=rows) + if key.startswith('qwen-') and key.endswith('-cuda'): + result['torch_peak_allocated_bytes'] = torch.cuda.max_memory_allocated() + result['torch_peak_reserved_bytes'] = torch.cuda.max_memory_reserved() + target.write_text(json.dumps(result, ensure_ascii=False, indent=2), encoding='utf-8') + print(json.dumps({k:v for k,v in result.items() if k != 'rows'}, ensure_ascii=True), flush=True) + + +if __name__ == '__main__': + main() diff --git a/uv.lock b/uv.lock index 30223580..f509cf41 100644 --- a/uv.lock +++ b/uv.lock @@ -2,11 +2,14 @@ version = 1 revision = 2 requires-python = ">=3.11" resolution-markers = [ - "python_full_version >= '3.14' and sys_platform != 'darwin'", - "python_full_version >= '3.12' and python_full_version < '3.14' and sys_platform != 'darwin'", + "python_full_version >= '3.14' and sys_platform == 'win32'", + "python_full_version >= '3.14' and sys_platform != 'darwin' and sys_platform != 'win32'", + "python_full_version >= '3.12' and python_full_version < '3.14' and sys_platform == 'win32'", + "python_full_version >= '3.12' and python_full_version < '3.14' and sys_platform != 'darwin' and sys_platform != 'win32'", "python_full_version >= '3.14' and sys_platform == 'darwin'", "python_full_version >= '3.12' and python_full_version < '3.14' and sys_platform == 'darwin'", - "python_full_version < '3.12' and sys_platform != 'darwin'", + "python_full_version < '3.12' and sys_platform == 'win32'", + "python_full_version < '3.12' and sys_platform != 'darwin' and sys_platform != 'win32'", "python_full_version < '3.12' and sys_platform == 'darwin'", ] @@ -37,6 +40,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/a9/ba/000a1996d4308bc65120167c21241a3b205464a2e0b58deda26ae8ac21d1/altgraph-0.17.5-py2.py3-none-any.whl", hash = "sha256:f3a22400bce1b0c701683820ac4f3b159cd301acab067c51c653e06961600597", size = 21228, upload-time = "2025-11-21T20:35:49.444Z" }, ] +[[package]] +name = "annotated-doc" +version = "0.0.5" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/5a/8e/38aa427ed5402449e226975b649c5dc73ccadfefeb95e6aecb8f8ea4b6b6/annotated_doc-0.0.5.tar.gz", hash = "sha256:c7e58ce09192557605d8bbd92836d7e1d520ac9580096042c0bfd197efacf1bb", size = 10758, upload-time = "2026-07-28T13:50:58.129Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/3e/30/e900b21425a860e195f32e37657aa1f7c7f2b1bfb26f03ca209b90933c06/annotated_doc-0.0.5-py3-none-any.whl", hash = "sha256:117bac03a25ede5df5440e855b32d556049ca169ead221505badf432fed4b101", size = 5302, upload-time = "2026-07-28T13:50:57.239Z" }, +] + [[package]] name = "annotated-types" version = "0.7.0" @@ -655,6 +667,18 @@ version = "0.42.1" source = { registry = "https://pypi.org/simple" } sdist = { url = "https://files.pythonhosted.org/packages/c6/cb/18eeb235f833b726522d7ebed54f2278ce28ba9438e3135ab0278d9792a2/jieba-0.42.1.tar.gz", hash = "sha256:055ca12f62674fafed09427f176506079bc135638a14e23e25be909131928db2", size = 19214172, upload-time = "2020-01-20T14:27:23.5Z" } +[[package]] +name = "jinja2" +version = "3.1.6" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "markupsafe" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/df/bf/f7da0350254c0ed7c72f3e33cef02e048281fec7ecec5f032d4aac52226b/jinja2-3.1.6.tar.gz", hash = "sha256:0137fb05990d35f1275a587e9aee6d56da821fc83491a0fb838183be43f66d6d", size = 245115, upload-time = "2025-03-05T20:05:02.478Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/62/a1/3d680cbfd5f4b8f15abc1d571870c5fc3e594bb582bc3b64ea099db13e56/jinja2-3.1.6-py3-none-any.whl", hash = "sha256:85ece4451f492d0c13c5dd7c13a64681a86afae63a5f347908daf103ce6d2f67", size = 134899, upload-time = "2025-03-05T20:05:00.369Z" }, +] + [[package]] name = "jiter" version = "0.16.0" @@ -1108,6 +1132,101 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/c7/d1/a9f36f8ecdf0fb7c9b1e78c8d7af12b8c8754e74851ac7b94a8305540fc7/macholib-1.16.4-py2.py3-none-any.whl", hash = "sha256:da1a3fa8266e30f0ce7e97c6a54eefaae8edd1e5f86f3eb8b95457cae90265ea", size = 38117, upload-time = "2025-11-22T08:28:36.939Z" }, ] +[[package]] +name = "markdown-it-py" +version = "4.2.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "mdurl" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/06/ff/7841249c247aa650a76b9ee4bbaeae59370dc8bfd2f6c01f3630c35eb134/markdown_it_py-4.2.0.tar.gz", hash = "sha256:04a21681d6fbb623de53f6f364d352309d4094dd4194040a10fd51833e418d49", size = 82454, upload-time = "2026-05-07T12:08:28.36Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/b3/81/4da04ced5a082363ecfa159c010d200ecbd959ae410c10c0264a38cac0f5/markdown_it_py-4.2.0-py3-none-any.whl", hash = "sha256:9f7ebbcd14fe59494226453aed97c1070d83f8d24b6fc3a3bcf9a38092641c4a", size = 91687, upload-time = "2026-05-07T12:08:27.182Z" }, +] + +[[package]] +name = "markupsafe" +version = "3.0.3" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/7e/99/7690b6d4034fffd95959cbe0c02de8deb3098cc577c67bb6a24fe5d7caa7/markupsafe-3.0.3.tar.gz", hash = "sha256:722695808f4b6457b320fdc131280796bdceb04ab50fe1795cd540799ebe1698", size = 80313, upload-time = "2025-09-27T18:37:40.426Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/08/db/fefacb2136439fc8dd20e797950e749aa1f4997ed584c62cfb8ef7c2be0e/markupsafe-3.0.3-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:1cc7ea17a6824959616c525620e387f6dd30fec8cb44f649e31712db02123dad", size = 11631, upload-time = "2025-09-27T18:36:18.185Z" }, + { url = "https://files.pythonhosted.org/packages/e1/2e/5898933336b61975ce9dc04decbc0a7f2fee78c30353c5efba7f2d6ff27a/markupsafe-3.0.3-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:4bd4cd07944443f5a265608cc6aab442e4f74dff8088b0dfc8238647b8f6ae9a", size = 12058, upload-time = "2025-09-27T18:36:19.444Z" }, + { url = "https://files.pythonhosted.org/packages/1d/09/adf2df3699d87d1d8184038df46a9c80d78c0148492323f4693df54e17bb/markupsafe-3.0.3-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:6b5420a1d9450023228968e7e6a9ce57f65d148ab56d2313fcd589eee96a7a50", size = 24287, upload-time = "2025-09-27T18:36:20.768Z" }, + { url = "https://files.pythonhosted.org/packages/30/ac/0273f6fcb5f42e314c6d8cd99effae6a5354604d461b8d392b5ec9530a54/markupsafe-3.0.3-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:0bf2a864d67e76e5c9a34dc26ec616a66b9888e25e7b9460e1c76d3293bd9dbf", size = 22940, upload-time = "2025-09-27T18:36:22.249Z" }, + { url = "https://files.pythonhosted.org/packages/19/ae/31c1be199ef767124c042c6c3e904da327a2f7f0cd63a0337e1eca2967a8/markupsafe-3.0.3-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:bc51efed119bc9cfdf792cdeaa4d67e8f6fcccab66ed4bfdd6bde3e59bfcbb2f", size = 21887, upload-time = "2025-09-27T18:36:23.535Z" }, + { url = "https://files.pythonhosted.org/packages/b2/76/7edcab99d5349a4532a459e1fe64f0b0467a3365056ae550d3bcf3f79e1e/markupsafe-3.0.3-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:068f375c472b3e7acbe2d5318dea141359e6900156b5b2ba06a30b169086b91a", size = 23692, upload-time = "2025-09-27T18:36:24.823Z" }, + { url = "https://files.pythonhosted.org/packages/a4/28/6e74cdd26d7514849143d69f0bf2399f929c37dc2b31e6829fd2045b2765/markupsafe-3.0.3-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:7be7b61bb172e1ed687f1754f8e7484f1c8019780f6f6b0786e76bb01c2ae115", size = 21471, upload-time = "2025-09-27T18:36:25.95Z" }, + { url = "https://files.pythonhosted.org/packages/62/7e/a145f36a5c2945673e590850a6f8014318d5577ed7e5920a4b3448e0865d/markupsafe-3.0.3-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:f9e130248f4462aaa8e2552d547f36ddadbeaa573879158d721bbd33dfe4743a", size = 22923, upload-time = "2025-09-27T18:36:27.109Z" }, + { url = "https://files.pythonhosted.org/packages/0f/62/d9c46a7f5c9adbeeeda52f5b8d802e1094e9717705a645efc71b0913a0a8/markupsafe-3.0.3-cp311-cp311-win32.whl", hash = "sha256:0db14f5dafddbb6d9208827849fad01f1a2609380add406671a26386cdf15a19", size = 14572, upload-time = "2025-09-27T18:36:28.045Z" }, + { url = "https://files.pythonhosted.org/packages/83/8a/4414c03d3f891739326e1783338e48fb49781cc915b2e0ee052aa490d586/markupsafe-3.0.3-cp311-cp311-win_amd64.whl", hash = "sha256:de8a88e63464af587c950061a5e6a67d3632e36df62b986892331d4620a35c01", size = 15077, upload-time = "2025-09-27T18:36:29.025Z" }, + { url = "https://files.pythonhosted.org/packages/35/73/893072b42e6862f319b5207adc9ae06070f095b358655f077f69a35601f0/markupsafe-3.0.3-cp311-cp311-win_arm64.whl", hash = "sha256:3b562dd9e9ea93f13d53989d23a7e775fdfd1066c33494ff43f5418bc8c58a5c", size = 13876, upload-time = "2025-09-27T18:36:29.954Z" }, + { url = "https://files.pythonhosted.org/packages/5a/72/147da192e38635ada20e0a2e1a51cf8823d2119ce8883f7053879c2199b5/markupsafe-3.0.3-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:d53197da72cc091b024dd97249dfc7794d6a56530370992a5e1a08983ad9230e", size = 11615, upload-time = "2025-09-27T18:36:30.854Z" }, + { url = "https://files.pythonhosted.org/packages/9a/81/7e4e08678a1f98521201c3079f77db69fb552acd56067661f8c2f534a718/markupsafe-3.0.3-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:1872df69a4de6aead3491198eaf13810b565bdbeec3ae2dc8780f14458ec73ce", size = 12020, upload-time = "2025-09-27T18:36:31.971Z" }, + { url = "https://files.pythonhosted.org/packages/1e/2c/799f4742efc39633a1b54a92eec4082e4f815314869865d876824c257c1e/markupsafe-3.0.3-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:3a7e8ae81ae39e62a41ec302f972ba6ae23a5c5396c8e60113e9066ef893da0d", size = 24332, upload-time = "2025-09-27T18:36:32.813Z" }, + { url = "https://files.pythonhosted.org/packages/3c/2e/8d0c2ab90a8c1d9a24f0399058ab8519a3279d1bd4289511d74e909f060e/markupsafe-3.0.3-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d6dd0be5b5b189d31db7cda48b91d7e0a9795f31430b7f271219ab30f1d3ac9d", size = 22947, upload-time = "2025-09-27T18:36:33.86Z" }, + { url = "https://files.pythonhosted.org/packages/2c/54/887f3092a85238093a0b2154bd629c89444f395618842e8b0c41783898ea/markupsafe-3.0.3-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:94c6f0bb423f739146aec64595853541634bde58b2135f27f61c1ffd1cd4d16a", size = 21962, upload-time = "2025-09-27T18:36:35.099Z" }, + { url = "https://files.pythonhosted.org/packages/c9/2f/336b8c7b6f4a4d95e91119dc8521402461b74a485558d8f238a68312f11c/markupsafe-3.0.3-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:be8813b57049a7dc738189df53d69395eba14fb99345e0a5994914a3864c8a4b", size = 23760, upload-time = "2025-09-27T18:36:36.001Z" }, + { url = "https://files.pythonhosted.org/packages/32/43/67935f2b7e4982ffb50a4d169b724d74b62a3964bc1a9a527f5ac4f1ee2b/markupsafe-3.0.3-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:83891d0e9fb81a825d9a6d61e3f07550ca70a076484292a70fde82c4b807286f", size = 21529, upload-time = "2025-09-27T18:36:36.906Z" }, + { url = "https://files.pythonhosted.org/packages/89/e0/4486f11e51bbba8b0c041098859e869e304d1c261e59244baa3d295d47b7/markupsafe-3.0.3-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:77f0643abe7495da77fb436f50f8dab76dbc6e5fd25d39589a0f1fe6548bfa2b", size = 23015, upload-time = "2025-09-27T18:36:37.868Z" }, + { url = "https://files.pythonhosted.org/packages/2f/e1/78ee7a023dac597a5825441ebd17170785a9dab23de95d2c7508ade94e0e/markupsafe-3.0.3-cp312-cp312-win32.whl", hash = "sha256:d88b440e37a16e651bda4c7c2b930eb586fd15ca7406cb39e211fcff3bf3017d", size = 14540, upload-time = "2025-09-27T18:36:38.761Z" }, + { url = "https://files.pythonhosted.org/packages/aa/5b/bec5aa9bbbb2c946ca2733ef9c4ca91c91b6a24580193e891b5f7dbe8e1e/markupsafe-3.0.3-cp312-cp312-win_amd64.whl", hash = "sha256:26a5784ded40c9e318cfc2bdb30fe164bdb8665ded9cd64d500a34fb42067b1c", size = 15105, upload-time = "2025-09-27T18:36:39.701Z" }, + { url = "https://files.pythonhosted.org/packages/e5/f1/216fc1bbfd74011693a4fd837e7026152e89c4bcf3e77b6692fba9923123/markupsafe-3.0.3-cp312-cp312-win_arm64.whl", hash = "sha256:35add3b638a5d900e807944a078b51922212fb3dedb01633a8defc4b01a3c85f", size = 13906, upload-time = "2025-09-27T18:36:40.689Z" }, + { url = "https://files.pythonhosted.org/packages/38/2f/907b9c7bbba283e68f20259574b13d005c121a0fa4c175f9bed27c4597ff/markupsafe-3.0.3-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:e1cf1972137e83c5d4c136c43ced9ac51d0e124706ee1c8aa8532c1287fa8795", size = 11622, upload-time = "2025-09-27T18:36:41.777Z" }, + { url = "https://files.pythonhosted.org/packages/9c/d9/5f7756922cdd676869eca1c4e3c0cd0df60ed30199ffd775e319089cb3ed/markupsafe-3.0.3-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:116bb52f642a37c115f517494ea5feb03889e04df47eeff5b130b1808ce7c219", size = 12029, upload-time = "2025-09-27T18:36:43.257Z" }, + { url = "https://files.pythonhosted.org/packages/00/07/575a68c754943058c78f30db02ee03a64b3c638586fba6a6dd56830b30a3/markupsafe-3.0.3-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:133a43e73a802c5562be9bbcd03d090aa5a1fe899db609c29e8c8d815c5f6de6", size = 24374, upload-time = "2025-09-27T18:36:44.508Z" }, + { url = "https://files.pythonhosted.org/packages/a9/21/9b05698b46f218fc0e118e1f8168395c65c8a2c750ae2bab54fc4bd4e0e8/markupsafe-3.0.3-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:ccfcd093f13f0f0b7fdd0f198b90053bf7b2f02a3927a30e63f3ccc9df56b676", size = 22980, upload-time = "2025-09-27T18:36:45.385Z" }, + { url = "https://files.pythonhosted.org/packages/7f/71/544260864f893f18b6827315b988c146b559391e6e7e8f7252839b1b846a/markupsafe-3.0.3-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:509fa21c6deb7a7a273d629cf5ec029bc209d1a51178615ddf718f5918992ab9", size = 21990, upload-time = "2025-09-27T18:36:46.916Z" }, + { url = "https://files.pythonhosted.org/packages/c2/28/b50fc2f74d1ad761af2f5dcce7492648b983d00a65b8c0e0cb457c82ebbe/markupsafe-3.0.3-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:a4afe79fb3de0b7097d81da19090f4df4f8d3a2b3adaa8764138aac2e44f3af1", size = 23784, upload-time = "2025-09-27T18:36:47.884Z" }, + { url = "https://files.pythonhosted.org/packages/ed/76/104b2aa106a208da8b17a2fb72e033a5a9d7073c68f7e508b94916ed47a9/markupsafe-3.0.3-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:795e7751525cae078558e679d646ae45574b47ed6e7771863fcc079a6171a0fc", size = 21588, upload-time = "2025-09-27T18:36:48.82Z" }, + { url = "https://files.pythonhosted.org/packages/b5/99/16a5eb2d140087ebd97180d95249b00a03aa87e29cc224056274f2e45fd6/markupsafe-3.0.3-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:8485f406a96febb5140bfeca44a73e3ce5116b2501ac54fe953e488fb1d03b12", size = 23041, upload-time = "2025-09-27T18:36:49.797Z" }, + { url = "https://files.pythonhosted.org/packages/19/bc/e7140ed90c5d61d77cea142eed9f9c303f4c4806f60a1044c13e3f1471d0/markupsafe-3.0.3-cp313-cp313-win32.whl", hash = "sha256:bdd37121970bfd8be76c5fb069c7751683bdf373db1ed6c010162b2a130248ed", size = 14543, upload-time = "2025-09-27T18:36:51.584Z" }, + { url = "https://files.pythonhosted.org/packages/05/73/c4abe620b841b6b791f2edc248f556900667a5a1cf023a6646967ae98335/markupsafe-3.0.3-cp313-cp313-win_amd64.whl", hash = "sha256:9a1abfdc021a164803f4d485104931fb8f8c1efd55bc6b748d2f5774e78b62c5", size = 15113, upload-time = "2025-09-27T18:36:52.537Z" }, + { url = "https://files.pythonhosted.org/packages/f0/3a/fa34a0f7cfef23cf9500d68cb7c32dd64ffd58a12b09225fb03dd37d5b80/markupsafe-3.0.3-cp313-cp313-win_arm64.whl", hash = "sha256:7e68f88e5b8799aa49c85cd116c932a1ac15caaa3f5db09087854d218359e485", size = 13911, upload-time = "2025-09-27T18:36:53.513Z" }, + { url = "https://files.pythonhosted.org/packages/e4/d7/e05cd7efe43a88a17a37b3ae96e79a19e846f3f456fe79c57ca61356ef01/markupsafe-3.0.3-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:218551f6df4868a8d527e3062d0fb968682fe92054e89978594c28e642c43a73", size = 11658, upload-time = "2025-09-27T18:36:54.819Z" }, + { url = "https://files.pythonhosted.org/packages/99/9e/e412117548182ce2148bdeacdda3bb494260c0b0184360fe0d56389b523b/markupsafe-3.0.3-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:3524b778fe5cfb3452a09d31e7b5adefeea8c5be1d43c4f810ba09f2ceb29d37", size = 12066, upload-time = "2025-09-27T18:36:55.714Z" }, + { url = "https://files.pythonhosted.org/packages/bc/e6/fa0ffcda717ef64a5108eaa7b4f5ed28d56122c9a6d70ab8b72f9f715c80/markupsafe-3.0.3-cp313-cp313t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:4e885a3d1efa2eadc93c894a21770e4bc67899e3543680313b09f139e149ab19", size = 25639, upload-time = "2025-09-27T18:36:56.908Z" }, + { url = "https://files.pythonhosted.org/packages/96/ec/2102e881fe9d25fc16cb4b25d5f5cde50970967ffa5dddafdb771237062d/markupsafe-3.0.3-cp313-cp313t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:8709b08f4a89aa7586de0aadc8da56180242ee0ada3999749b183aa23df95025", size = 23569, upload-time = "2025-09-27T18:36:57.913Z" }, + { url = "https://files.pythonhosted.org/packages/4b/30/6f2fce1f1f205fc9323255b216ca8a235b15860c34b6798f810f05828e32/markupsafe-3.0.3-cp313-cp313t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:b8512a91625c9b3da6f127803b166b629725e68af71f8184ae7e7d54686a56d6", size = 23284, upload-time = "2025-09-27T18:36:58.833Z" }, + { url = "https://files.pythonhosted.org/packages/58/47/4a0ccea4ab9f5dcb6f79c0236d954acb382202721e704223a8aafa38b5c8/markupsafe-3.0.3-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:9b79b7a16f7fedff2495d684f2b59b0457c3b493778c9eed31111be64d58279f", size = 24801, upload-time = "2025-09-27T18:36:59.739Z" }, + { url = "https://files.pythonhosted.org/packages/6a/70/3780e9b72180b6fecb83a4814d84c3bf4b4ae4bf0b19c27196104149734c/markupsafe-3.0.3-cp313-cp313t-musllinux_1_2_riscv64.whl", hash = "sha256:12c63dfb4a98206f045aa9563db46507995f7ef6d83b2f68eda65c307c6829eb", size = 22769, upload-time = "2025-09-27T18:37:00.719Z" }, + { url = "https://files.pythonhosted.org/packages/98/c5/c03c7f4125180fc215220c035beac6b9cb684bc7a067c84fc69414d315f5/markupsafe-3.0.3-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:8f71bc33915be5186016f675cd83a1e08523649b0e33efdb898db577ef5bb009", size = 23642, upload-time = "2025-09-27T18:37:01.673Z" }, + { url = "https://files.pythonhosted.org/packages/80/d6/2d1b89f6ca4bff1036499b1e29a1d02d282259f3681540e16563f27ebc23/markupsafe-3.0.3-cp313-cp313t-win32.whl", hash = "sha256:69c0b73548bc525c8cb9a251cddf1931d1db4d2258e9599c28c07ef3580ef354", size = 14612, upload-time = "2025-09-27T18:37:02.639Z" }, + { url = "https://files.pythonhosted.org/packages/2b/98/e48a4bfba0a0ffcf9925fe2d69240bfaa19c6f7507b8cd09c70684a53c1e/markupsafe-3.0.3-cp313-cp313t-win_amd64.whl", hash = "sha256:1b4b79e8ebf6b55351f0d91fe80f893b4743f104bff22e90697db1590e47a218", size = 15200, upload-time = "2025-09-27T18:37:03.582Z" }, + { url = "https://files.pythonhosted.org/packages/0e/72/e3cc540f351f316e9ed0f092757459afbc595824ca724cbc5a5d4263713f/markupsafe-3.0.3-cp313-cp313t-win_arm64.whl", hash = "sha256:ad2cf8aa28b8c020ab2fc8287b0f823d0a7d8630784c31e9ee5edea20f406287", size = 13973, upload-time = "2025-09-27T18:37:04.929Z" }, + { url = "https://files.pythonhosted.org/packages/33/8a/8e42d4838cd89b7dde187011e97fe6c3af66d8c044997d2183fbd6d31352/markupsafe-3.0.3-cp314-cp314-macosx_10_13_x86_64.whl", hash = "sha256:eaa9599de571d72e2daf60164784109f19978b327a3910d3e9de8c97b5b70cfe", size = 11619, upload-time = "2025-09-27T18:37:06.342Z" }, + { url = "https://files.pythonhosted.org/packages/b5/64/7660f8a4a8e53c924d0fa05dc3a55c9cee10bbd82b11c5afb27d44b096ce/markupsafe-3.0.3-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:c47a551199eb8eb2121d4f0f15ae0f923d31350ab9280078d1e5f12b249e0026", size = 12029, upload-time = "2025-09-27T18:37:07.213Z" }, + { url = "https://files.pythonhosted.org/packages/da/ef/e648bfd021127bef5fa12e1720ffed0c6cbb8310c8d9bea7266337ff06de/markupsafe-3.0.3-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f34c41761022dd093b4b6896d4810782ffbabe30f2d443ff5f083e0cbbb8c737", size = 24408, upload-time = "2025-09-27T18:37:09.572Z" }, + { url = "https://files.pythonhosted.org/packages/41/3c/a36c2450754618e62008bf7435ccb0f88053e07592e6028a34776213d877/markupsafe-3.0.3-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:457a69a9577064c05a97c41f4e65148652db078a3a509039e64d3467b9e7ef97", size = 23005, upload-time = "2025-09-27T18:37:10.58Z" }, + { url = "https://files.pythonhosted.org/packages/bc/20/b7fdf89a8456b099837cd1dc21974632a02a999ec9bf7ca3e490aacd98e7/markupsafe-3.0.3-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:e8afc3f2ccfa24215f8cb28dcf43f0113ac3c37c2f0f0806d8c70e4228c5cf4d", size = 22048, upload-time = "2025-09-27T18:37:11.547Z" }, + { url = "https://files.pythonhosted.org/packages/9a/a7/591f592afdc734f47db08a75793a55d7fbcc6902a723ae4cfbab61010cc5/markupsafe-3.0.3-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:ec15a59cf5af7be74194f7ab02d0f59a62bdcf1a537677ce67a2537c9b87fcda", size = 23821, upload-time = "2025-09-27T18:37:12.48Z" }, + { url = "https://files.pythonhosted.org/packages/7d/33/45b24e4f44195b26521bc6f1a82197118f74df348556594bd2262bda1038/markupsafe-3.0.3-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:0eb9ff8191e8498cca014656ae6b8d61f39da5f95b488805da4bb029cccbfbaf", size = 21606, upload-time = "2025-09-27T18:37:13.485Z" }, + { url = "https://files.pythonhosted.org/packages/ff/0e/53dfaca23a69fbfbbf17a4b64072090e70717344c52eaaaa9c5ddff1e5f0/markupsafe-3.0.3-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:2713baf880df847f2bece4230d4d094280f4e67b1e813eec43b4c0e144a34ffe", size = 23043, upload-time = "2025-09-27T18:37:14.408Z" }, + { url = "https://files.pythonhosted.org/packages/46/11/f333a06fc16236d5238bfe74daccbca41459dcd8d1fa952e8fbd5dccfb70/markupsafe-3.0.3-cp314-cp314-win32.whl", hash = "sha256:729586769a26dbceff69f7a7dbbf59ab6572b99d94576a5592625d5b411576b9", size = 14747, upload-time = "2025-09-27T18:37:15.36Z" }, + { url = "https://files.pythonhosted.org/packages/28/52/182836104b33b444e400b14f797212f720cbc9ed6ba34c800639d154e821/markupsafe-3.0.3-cp314-cp314-win_amd64.whl", hash = "sha256:bdc919ead48f234740ad807933cdf545180bfbe9342c2bb451556db2ed958581", size = 15341, upload-time = "2025-09-27T18:37:16.496Z" }, + { url = "https://files.pythonhosted.org/packages/6f/18/acf23e91bd94fd7b3031558b1f013adfa21a8e407a3fdb32745538730382/markupsafe-3.0.3-cp314-cp314-win_arm64.whl", hash = "sha256:5a7d5dc5140555cf21a6fefbdbf8723f06fcd2f63ef108f2854de715e4422cb4", size = 14073, upload-time = "2025-09-27T18:37:17.476Z" }, + { url = "https://files.pythonhosted.org/packages/3c/f0/57689aa4076e1b43b15fdfa646b04653969d50cf30c32a102762be2485da/markupsafe-3.0.3-cp314-cp314t-macosx_10_13_x86_64.whl", hash = "sha256:1353ef0c1b138e1907ae78e2f6c63ff67501122006b0f9abad68fda5f4ffc6ab", size = 11661, upload-time = "2025-09-27T18:37:18.453Z" }, + { url = "https://files.pythonhosted.org/packages/89/c3/2e67a7ca217c6912985ec766c6393b636fb0c2344443ff9d91404dc4c79f/markupsafe-3.0.3-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:1085e7fbddd3be5f89cc898938f42c0b3c711fdcb37d75221de2666af647c175", size = 12069, upload-time = "2025-09-27T18:37:19.332Z" }, + { url = "https://files.pythonhosted.org/packages/f0/00/be561dce4e6ca66b15276e184ce4b8aec61fe83662cce2f7d72bd3249d28/markupsafe-3.0.3-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:1b52b4fb9df4eb9ae465f8d0c228a00624de2334f216f178a995ccdcf82c4634", size = 25670, upload-time = "2025-09-27T18:37:20.245Z" }, + { url = "https://files.pythonhosted.org/packages/50/09/c419f6f5a92e5fadde27efd190eca90f05e1261b10dbd8cbcb39cd8ea1dc/markupsafe-3.0.3-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:fed51ac40f757d41b7c48425901843666a6677e3e8eb0abcff09e4ba6e664f50", size = 23598, upload-time = "2025-09-27T18:37:21.177Z" }, + { url = "https://files.pythonhosted.org/packages/22/44/a0681611106e0b2921b3033fc19bc53323e0b50bc70cffdd19f7d679bb66/markupsafe-3.0.3-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:f190daf01f13c72eac4efd5c430a8de82489d9cff23c364c3ea822545032993e", size = 23261, upload-time = "2025-09-27T18:37:22.167Z" }, + { url = "https://files.pythonhosted.org/packages/5f/57/1b0b3f100259dc9fffe780cfb60d4be71375510e435efec3d116b6436d43/markupsafe-3.0.3-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:e56b7d45a839a697b5eb268c82a71bd8c7f6c94d6fd50c3d577fa39a9f1409f5", size = 24835, upload-time = "2025-09-27T18:37:23.296Z" }, + { url = "https://files.pythonhosted.org/packages/26/6a/4bf6d0c97c4920f1597cc14dd720705eca0bf7c787aebc6bb4d1bead5388/markupsafe-3.0.3-cp314-cp314t-musllinux_1_2_riscv64.whl", hash = "sha256:f3e98bb3798ead92273dc0e5fd0f31ade220f59a266ffd8a4f6065e0a3ce0523", size = 22733, upload-time = "2025-09-27T18:37:24.237Z" }, + { url = "https://files.pythonhosted.org/packages/14/c7/ca723101509b518797fedc2fdf79ba57f886b4aca8a7d31857ba3ee8281f/markupsafe-3.0.3-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:5678211cb9333a6468fb8d8be0305520aa073f50d17f089b5b4b477ea6e67fdc", size = 23672, upload-time = "2025-09-27T18:37:25.271Z" }, + { url = "https://files.pythonhosted.org/packages/fb/df/5bd7a48c256faecd1d36edc13133e51397e41b73bb77e1a69deab746ebac/markupsafe-3.0.3-cp314-cp314t-win32.whl", hash = "sha256:915c04ba3851909ce68ccc2b8e2cd691618c4dc4c4232fb7982bca3f41fd8c3d", size = 14819, upload-time = "2025-09-27T18:37:26.285Z" }, + { url = "https://files.pythonhosted.org/packages/1a/8a/0402ba61a2f16038b48b39bccca271134be00c5c9f0f623208399333c448/markupsafe-3.0.3-cp314-cp314t-win_amd64.whl", hash = "sha256:4faffd047e07c38848ce017e8725090413cd80cbc23d86e55c587bf979e579c9", size = 15426, upload-time = "2025-09-27T18:37:27.316Z" }, + { url = "https://files.pythonhosted.org/packages/70/bc/6f1c2f612465f5fa89b95bead1f44dcb607670fd42891d8fdcd5d039f4f4/markupsafe-3.0.3-cp314-cp314t-win_arm64.whl", hash = "sha256:32001d6a8fc98c8cb5c947787c5d08b0a50663d139f1305bac5885d98d9b40fa", size = 14146, upload-time = "2025-09-27T18:37:28.327Z" }, +] + +[[package]] +name = "mdurl" +version = "0.1.2" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/d6/54/cfe61301667036ec958cb99bd3efefba235e65cdeb9c84d24a8293ba1d90/mdurl-0.1.2.tar.gz", hash = "sha256:bb413d29f5eea38f31dd4754dd7377d4465116fb207585f97bf925588687c1ba", size = 8729, upload-time = "2022-08-14T12:40:10.846Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/b3/38/89ba8ad64ae25be8de66a6d463314cf1eb366222074cfda9ee839c56a4b4/mdurl-0.1.2-py3-none-any.whl", hash = "sha256:84008a41e51615a49fc9966191ff91509e3c40b939176e643fd50a5c2196b8f8", size = 9979, upload-time = "2022-08-14T12:40:09.779Z" }, +] + [[package]] name = "mpmath" version = "1.3.0" @@ -1117,12 +1236,22 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/43/e3/7d92a15f894aa0c9c4b49b8ee9ac9850d6e63b03c9c32c0367a13ae62209/mpmath-1.3.0-py3-none-any.whl", hash = "sha256:a0b2b9fe80bbcd81a6647ff13108738cfb482d481d826cc0e02f5b35e5c88d2c", size = 536198, upload-time = "2023-03-07T16:47:09.197Z" }, ] +[[package]] +name = "networkx" +version = "3.6.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/6a/51/63fe664f3908c97be9d2e4f1158eb633317598cfa6e1fc14af5383f17512/networkx-3.6.1.tar.gz", hash = "sha256:26b7c357accc0c8cde558ad486283728b65b6a95d85ee1cd66bafab4c8168509", size = 2517025, upload-time = "2025-12-08T17:02:39.908Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/9e/c9/b2622292ea83fbb4ec318f5b9ab867d0a28ab43c5717bb85b0a5f6b3b0a4/networkx-3.6.1-py3-none-any.whl", hash = "sha256:d47fbf302e7d9cbbb9e2555a0d267983d2aa476bac30e90dfbe5669bd57f3762", size = 2068504, upload-time = "2025-12-08T17:02:38.159Z" }, +] + [[package]] name = "numpy" version = "2.4.6" source = { registry = "https://pypi.org/simple" } resolution-markers = [ - "python_full_version < '3.12' and sys_platform != 'darwin'", + "python_full_version < '3.12' and sys_platform == 'win32'", + "python_full_version < '3.12' and sys_platform != 'darwin' and sys_platform != 'win32'", "python_full_version < '3.12' and sys_platform == 'darwin'", ] sdist = { url = "https://files.pythonhosted.org/packages/d0/ad/fed0499ce6a338d2a03ebae59cd15093910c8875328855781952abf6c2fe/numpy-2.4.6.tar.gz", hash = "sha256:f3a3570c4a2a16746ac2c31a7c7c7b0c186b95ce902e33db6f28094ed7387dda", size = 20735807, upload-time = "2026-05-18T23:37:14.07Z" } @@ -1205,8 +1334,10 @@ name = "numpy" version = "2.5.2" source = { registry = "https://pypi.org/simple" } resolution-markers = [ - "python_full_version >= '3.14' and sys_platform != 'darwin'", - "python_full_version >= '3.12' and python_full_version < '3.14' and sys_platform != 'darwin'", + "python_full_version >= '3.14' and sys_platform == 'win32'", + "python_full_version >= '3.14' and sys_platform != 'darwin' and sys_platform != 'win32'", + "python_full_version >= '3.12' and python_full_version < '3.14' and sys_platform == 'win32'", + "python_full_version >= '3.12' and python_full_version < '3.14' and sys_platform != 'darwin' and sys_platform != 'win32'", "python_full_version >= '3.14' and sys_platform == 'darwin'", "python_full_version >= '3.12' and python_full_version < '3.14' and sys_platform == 'darwin'", ] @@ -1279,6 +1410,140 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/b4/07/458c344f0f0c178f4481dad5cca790626ffe4c34eabf9467069d06ee4999/numpy-2.5.2-cp315-cp315t-win_arm64.whl", hash = "sha256:5f8e00be2ec6f45f4e8a41a527f68d44a7d96fee92a650e4d8b1326f77f61e6e", size = 10748103, upload-time = "2026-08-09T13:48:24.21Z" }, ] +[[package]] +name = "nvidia-cublas-cu12" +version = "12.8.4.1" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/dc/61/e24b560ab2e2eaeb3c839129175fb330dfcfc29e5203196e5541a4c44682/nvidia_cublas_cu12-12.8.4.1-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:8ac4e771d5a348c551b2a426eda6193c19aa630236b418086020df5ba9667142", size = 594346921, upload-time = "2025-03-07T01:44:31.254Z" }, +] + +[[package]] +name = "nvidia-cuda-cupti-cu12" +version = "12.8.90" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/f8/02/2adcaa145158bf1a8295d83591d22e4103dbfd821bcaf6f3f53151ca4ffa/nvidia_cuda_cupti_cu12-12.8.90-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:ea0cb07ebda26bb9b29ba82cda34849e73c166c18162d3913575b0c9db9a6182", size = 10248621, upload-time = "2025-03-07T01:40:21.213Z" }, +] + +[[package]] +name = "nvidia-cuda-nvrtc-cu12" +version = "12.8.93" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/05/6b/32f747947df2da6994e999492ab306a903659555dddc0fbdeb9d71f75e52/nvidia_cuda_nvrtc_cu12-12.8.93-py3-none-manylinux2010_x86_64.manylinux_2_12_x86_64.whl", hash = "sha256:a7756528852ef889772a84c6cd89d41dfa74667e24cca16bb31f8f061e3e9994", size = 88040029, upload-time = "2025-03-07T01:42:13.562Z" }, +] + +[[package]] +name = "nvidia-cuda-runtime-cu12" +version = "12.8.90" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/0d/9b/a997b638fcd068ad6e4d53b8551a7d30fe8b404d6f1804abf1df69838932/nvidia_cuda_runtime_cu12-12.8.90-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:adade8dcbd0edf427b7204d480d6066d33902cab2a4707dcfc48a2d0fd44ab90", size = 954765, upload-time = "2025-03-07T01:40:01.615Z" }, +] + +[[package]] +name = "nvidia-cudnn-cu12" +version = "9.10.2.21" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "nvidia-cublas-cu12", marker = "sys_platform != 'darwin' and sys_platform != 'win32'" }, +] +wheels = [ + { url = "https://files.pythonhosted.org/packages/ba/51/e123d997aa098c61d029f76663dedbfb9bc8dcf8c60cbd6adbe42f76d049/nvidia_cudnn_cu12-9.10.2.21-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:949452be657fa16687d0930933f032835951ef0892b37d2d53824d1a84dc97a8", size = 706758467, upload-time = "2025-06-06T21:54:08.597Z" }, +] + +[[package]] +name = "nvidia-cufft-cu12" +version = "11.3.3.83" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "nvidia-nvjitlink-cu12", marker = "sys_platform != 'darwin' and sys_platform != 'win32'" }, +] +wheels = [ + { url = "https://files.pythonhosted.org/packages/1f/13/ee4e00f30e676b66ae65b4f08cb5bcbb8392c03f54f2d5413ea99a5d1c80/nvidia_cufft_cu12-11.3.3.83-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:4d2dd21ec0b88cf61b62e6b43564355e5222e4a3fb394cac0db101f2dd0d4f74", size = 193118695, upload-time = "2025-03-07T01:45:27.821Z" }, +] + +[[package]] +name = "nvidia-cufile-cu12" +version = "1.13.1.3" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/bb/fe/1bcba1dfbfb8d01be8d93f07bfc502c93fa23afa6fd5ab3fc7c1df71038a/nvidia_cufile_cu12-1.13.1.3-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:1d069003be650e131b21c932ec3d8969c1715379251f8d23a1860554b1cb24fc", size = 1197834, upload-time = "2025-03-07T01:45:50.723Z" }, +] + +[[package]] +name = "nvidia-curand-cu12" +version = "10.3.9.90" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/fb/aa/6584b56dc84ebe9cf93226a5cde4d99080c8e90ab40f0c27bda7a0f29aa1/nvidia_curand_cu12-10.3.9.90-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:b32331d4f4df5d6eefa0554c565b626c7216f87a06a4f56fab27c3b68a830ec9", size = 63619976, upload-time = "2025-03-07T01:46:23.323Z" }, +] + +[[package]] +name = "nvidia-cusolver-cu12" +version = "11.7.3.90" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "nvidia-cublas-cu12", marker = "sys_platform != 'darwin' and sys_platform != 'win32'" }, + { name = "nvidia-cusparse-cu12", marker = "sys_platform != 'darwin' and sys_platform != 'win32'" }, + { name = "nvidia-nvjitlink-cu12", marker = "sys_platform != 'darwin' and sys_platform != 'win32'" }, +] +wheels = [ + { url = "https://files.pythonhosted.org/packages/85/48/9a13d2975803e8cf2777d5ed57b87a0b6ca2cc795f9a4f59796a910bfb80/nvidia_cusolver_cu12-11.7.3.90-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:4376c11ad263152bd50ea295c05370360776f8c3427b30991df774f9fb26c450", size = 267506905, upload-time = "2025-03-07T01:47:16.273Z" }, +] + +[[package]] +name = "nvidia-cusparse-cu12" +version = "12.5.8.93" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "nvidia-nvjitlink-cu12", marker = "sys_platform != 'darwin' and sys_platform != 'win32'" }, +] +wheels = [ + { url = "https://files.pythonhosted.org/packages/c2/f5/e1854cb2f2bcd4280c44736c93550cc300ff4b8c95ebe370d0aa7d2b473d/nvidia_cusparse_cu12-12.5.8.93-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:1ec05d76bbbd8b61b06a80e1eaf8cf4959c3d4ce8e711b65ebd0443bb0ebb13b", size = 288216466, upload-time = "2025-03-07T01:48:13.779Z" }, +] + +[[package]] +name = "nvidia-cusparselt-cu12" +version = "0.7.1" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/56/79/12978b96bd44274fe38b5dde5cfb660b1d114f70a65ef962bcbbed99b549/nvidia_cusparselt_cu12-0.7.1-py3-none-manylinux2014_x86_64.whl", hash = "sha256:f1bb701d6b930d5a7cea44c19ceb973311500847f81b634d802b7b539dc55623", size = 287193691, upload-time = "2025-02-26T00:15:44.104Z" }, +] + +[[package]] +name = "nvidia-nccl-cu12" +version = "2.27.5" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/6e/89/f7a07dc961b60645dbbf42e80f2bc85ade7feb9a491b11a1e973aa00071f/nvidia_nccl_cu12-2.27.5-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:ad730cf15cb5d25fe849c6e6ca9eb5b76db16a80f13f425ac68d8e2e55624457", size = 322348229, upload-time = "2025-06-26T04:11:28.385Z" }, +] + +[[package]] +name = "nvidia-nvjitlink-cu12" +version = "12.8.93" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/f6/74/86a07f1d0f42998ca31312f998bd3b9a7eff7f52378f4f270c8679c77fb9/nvidia_nvjitlink_cu12-12.8.93-py3-none-manylinux2010_x86_64.manylinux_2_12_x86_64.whl", hash = "sha256:81ff63371a7ebd6e6451970684f916be2eab07321b73c9d244dc2b4da7f73b88", size = 39254836, upload-time = "2025-03-07T01:49:55.661Z" }, +] + +[[package]] +name = "nvidia-nvshmem-cu12" +version = "3.3.20" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/3b/6c/99acb2f9eb85c29fc6f3a7ac4dccfd992e22666dd08a642b303311326a97/nvidia_nvshmem_cu12-3.3.20-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:d00f26d3f9b2e3c3065be895e3059d6479ea5c638a3f38c9fec49b1b9dd7c1e5", size = 124657145, upload-time = "2025-08-04T20:25:19.995Z" }, +] + +[[package]] +name = "nvidia-nvtx-cu12" +version = "12.8.90" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a2/eb/86626c1bbc2edb86323022371c39aa48df6fd8b0a1647bc274577f72e90b/nvidia_nvtx_cu12-12.8.90-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:5b17e2001cc0d751a5bc2c6ec6d26ad95913324a4adb86788c944f8ce9ba441f", size = 89954, upload-time = "2025-03-07T01:42:44.131Z" }, +] + [[package]] name = "onnxruntime" version = "1.23.2" @@ -1311,9 +1576,12 @@ name = "onnxruntime" version = "1.28.0" source = { registry = "https://pypi.org/simple" } resolution-markers = [ - "python_full_version >= '3.14' and sys_platform != 'darwin'", - "python_full_version >= '3.12' and python_full_version < '3.14' and sys_platform != 'darwin'", - "python_full_version < '3.12' and sys_platform != 'darwin'", + "python_full_version >= '3.14' and sys_platform == 'win32'", + "python_full_version >= '3.14' and sys_platform != 'darwin' and sys_platform != 'win32'", + "python_full_version >= '3.12' and python_full_version < '3.14' and sys_platform == 'win32'", + "python_full_version >= '3.12' and python_full_version < '3.14' and sys_platform != 'darwin' and sys_platform != 'win32'", + "python_full_version < '3.12' and sys_platform == 'win32'", + "python_full_version < '3.12' and sys_platform != 'darwin' and sys_platform != 'win32'", ] dependencies = [ { name = "flatbuffers", marker = "sys_platform != 'darwin'" }, @@ -2163,6 +2431,43 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/3f/51/d4db610ef29373b879047326cbf6fa98b6c1969d6f6dc423279de2b1be2c/requests_toolbelt-1.0.0-py2.py3-none-any.whl", hash = "sha256:cccfdd665f0a24fcf4726e690f65639d272bb0637b9b92dfd91a5568ccf6bd06", size = 54481, upload-time = "2023-05-01T04:11:28.427Z" }, ] +[[package]] +name = "rich" +version = "15.0.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "markdown-it-py" }, + { name = "pygments" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/c0/8f/0722ca900cc807c13a6a0c696dacf35430f72e0ec571c4275d2371fca3e9/rich-15.0.0.tar.gz", hash = "sha256:edd07a4824c6b40189fb7ac9bc4c52536e9780fbbfbddf6f1e2502c31b068c36", size = 230680, upload-time = "2026-04-12T08:24:00.75Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/82/3b/64d4899d73f91ba49a8c18a8ff3f0ea8f1c1d75481760df8c68ef5235bf5/rich-15.0.0-py3-none-any.whl", hash = "sha256:33bd4ef74232fb73fe9279a257718407f169c09b78a87ad3d296f548e27de0bb", size = 310654, upload-time = "2026-04-12T08:24:02.83Z" }, +] + +[[package]] +name = "safetensors" +version = "0.8.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/45/06/f955dbbb1859e3bd23c8ac6141af5106e7ad5fedec4a3a6e3d60f94b7001/safetensors-0.8.0.tar.gz", hash = "sha256:fabaf3e0f18a6618d9b36560682562157f77c2b71fcffc7b432be2baed9d753d", size = 325846, upload-time = "2026-06-09T07:52:25.563Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/39/a0/f718cda65b05407d228f97602cf60dca269c979867aa5beb25410de26cd3/safetensors-0.8.0-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:c554f85858e05226d3c2828e32395e677434685d6d94594a41643361c5e837f0", size = 473568, upload-time = "2026-06-09T07:52:18.829Z" }, + { url = "https://files.pythonhosted.org/packages/f5/b1/fa7c600e7dceae12e9606c7578cbc9ff1e1ed55844883ee5c92205e86226/safetensors-0.8.0-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:c80201d22cbf405b80647a60ada77bba06c8fba2da2743ba1e89cdcc39a81f25", size = 484562, upload-time = "2026-06-09T07:52:17.518Z" }, + { url = "https://files.pythonhosted.org/packages/09/7d/65a7de0af421317bb36a067241e4235fff194eed60b961ed6d3f59a3fc60/safetensors-0.8.0-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:7a46e5ff292c356d6991e60942ba7f79817682d3a2cef0702136448cb9c4d235", size = 502844, upload-time = "2026-06-09T07:52:07.624Z" }, + { url = "https://files.pythonhosted.org/packages/91/4f/3175c9d75634e0e0dda0082794193521035edd7c70a6f212bf33ca06ddf4/safetensors-0.8.0-cp310-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:4124502b78f03534117c848f87a39b8f31e577b15eff423bf8bfb95f2a8c30d0", size = 511823, upload-time = "2026-06-09T07:52:09.565Z" }, + { url = "https://files.pythonhosted.org/packages/20/87/846c289e7aa2299eff406335717cf43ce8777194ece8aad75772e0411615/safetensors-0.8.0-cp310-abi3-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:7bc0a787ba8a35be368ee3574edfa2b1ad389eebd0a72e482ae275490e3f6c98", size = 633461, upload-time = "2026-06-09T07:52:11.128Z" }, + { url = "https://files.pythonhosted.org/packages/76/22/8d64d9df2c45d5ded401df889d0ad90882804ca172d79ec4f0df8f727fe0/safetensors-0.8.0-cp310-abi3-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:040070828e36dc8e122178bbbd5830ff9e97920affb84cbe0f46442497bed358", size = 545148, upload-time = "2026-06-09T07:52:13.603Z" }, + { url = "https://files.pythonhosted.org/packages/28/50/f203ff3a3ddfe19308efc83c5a3a29ed02bf786732ec35e68bf9162f3365/safetensors-0.8.0-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:fd6f3f93c9a0a7cc2788ee63fb763353d4bd2e89b0751bc78fcf7dda00bea774", size = 516040, upload-time = "2026-06-09T07:52:16.29Z" }, + { url = "https://files.pythonhosted.org/packages/46/fb/cdaed17ceb2948784fd9c36b6fd3e951b608547cea81a48e8ee6f8cfdfcb/safetensors-0.8.0-cp310-abi3-manylinux_2_31_riscv64.whl", hash = "sha256:fcdd41ec4628fee5799f807c73c353629130fbd942aa23d83c623dd6c9d52d78", size = 513832, upload-time = "2026-06-09T07:52:12.37Z" }, + { url = "https://files.pythonhosted.org/packages/0d/49/1e15de264dcc3b77943d2d0c56a95809956883b1c2d6d585c792523f180b/safetensors-0.8.0-cp310-abi3-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:8e9f537aa183a38ace122d27303dcd986b26bd2a7591f9181d7f0c396f4677ca", size = 559930, upload-time = "2026-06-09T07:52:14.743Z" }, + { url = "https://files.pythonhosted.org/packages/2a/43/bf38443278eab4b1be1fce2931e2b012ad9cb7df52ada751d0aab8f7659a/safetensors-0.8.0-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:87eec7ffed2b809f05a398a8becb7d013f19f7837cd15d9748580d6cf30dbaf4", size = 678670, upload-time = "2026-06-09T07:52:20.032Z" }, + { url = "https://files.pythonhosted.org/packages/72/e3/68cd3fa5b48488e84add63e04cb12f3bc28ae4638c06d4508c6e88823d0e/safetensors-0.8.0-cp310-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:4a95ae2b05d7726d751da4ebf626a2ca782b706e101bd894c95bc2450b1cffcc", size = 786679, upload-time = "2026-06-09T07:52:21.322Z" }, + { url = "https://files.pythonhosted.org/packages/29/4b/1c19c509d56e01f4fbb3d0a2e597450f6cc04d1d56cf52defb0a62dfd715/safetensors-0.8.0-cp310-abi3-musllinux_1_2_i686.whl", hash = "sha256:3ae091f16662658bdc019a4ff6cb4c085bb7d725eb5978b183ffd265863b6d2d", size = 765683, upload-time = "2026-06-09T07:52:22.594Z" }, + { url = "https://files.pythonhosted.org/packages/27/43/41c1621732edd934d868a00d1b891584c892a7b62a9aab82ea5a0a5623ee/safetensors-0.8.0-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:8e080062fcde23be189565e1c3305d16751a218ecf9412c8601e64204eb6f846", size = 722361, upload-time = "2026-06-09T07:52:23.924Z" }, + { url = "https://files.pythonhosted.org/packages/8e/3f/73ccf82579412b4a71c4ca673f10b5f1f888d7cf5af7fe24f27d30307be4/safetensors-0.8.0-cp310-abi3-win32.whl", hash = "sha256:2ddf52eac562eda224f99acfa7889d02968c1fd59a5b011ae7d8137c37e9c02d", size = 342401, upload-time = "2026-06-09T07:52:28.895Z" }, + { url = "https://files.pythonhosted.org/packages/1b/6d/3fba214c1e5e0f69991677ec3bc17023f0421776975e1de0c682dca475e2/safetensors-0.8.0-cp310-abi3-win_amd64.whl", hash = "sha256:096ec1a98435df7beb08853bb5aa9081a84f23d0adc67ed1a0a10550f608373f", size = 355540, upload-time = "2026-06-09T07:52:27.832Z" }, + { url = "https://files.pythonhosted.org/packages/8d/fc/7eedc3510d97878876e32774eebbeb61c43f148a96e915c84229a3e967aa/safetensors-0.8.0-cp310-abi3-win_arm64.whl", hash = "sha256:f7838e5135a406ad3e02efdcb8cf2e5397d368b0154537c4fec682dbc544d452", size = 340500, upload-time = "2026-06-09T07:52:26.745Z" }, +] + [[package]] name = "setuptools" version = "80.9.0" @@ -2172,6 +2477,74 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/a3/dc/17031897dae0efacfea57dfd3a82fdd2a2aeb58e0ff71b77b87e44edc772/setuptools-80.9.0-py3-none-any.whl", hash = "sha256:062d34222ad13e0cc312a4c02d73f059e86a4acbfbdea8f8f76b28c99f306922", size = 1201486, upload-time = "2025-05-27T00:56:49.664Z" }, ] +[[package]] +name = "shellingham" +version = "1.5.4" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/58/15/8b3609fd3830ef7b27b655beb4b4e9c62313a4e8da8c676e142cc210d58e/shellingham-1.5.4.tar.gz", hash = "sha256:8dbca0739d487e5bd35ab3ca4b36e11c4078f3a234bfce294b0a0291363404de", size = 10310, upload-time = "2023-10-24T04:13:40.426Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e0/f9/0595336914c5619e5f28a1fb793285925a8cd4b432c9da0a987836c7f822/shellingham-1.5.4-py2.py3-none-any.whl", hash = "sha256:7ecfff8f2fd72616f7481040475a65b2bf8af90a56c89140852d1120324e8686", size = 9755, upload-time = "2023-10-24T04:13:38.866Z" }, +] + +[[package]] +name = "sherpa-onnx" +version = "1.13.8" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/f9/5d/9c19acbe7eebd09cee4b69408532f44dc09e71b6025518ed7dd9c0509d6d/sherpa_onnx-1.13.8.tar.gz", hash = "sha256:68e638f745df120a7fae268b7bae0eaab2362b5ad76fe3389bece1d3cd63272c", size = 1057541, upload-time = "2026-09-10T15:05:43.696Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/10/43/ac0418d7336b38df63a098bbb4d2c78d8d6ea981454de535ee94b1da2816/sherpa_onnx-1.13.8-cp311-cp311-linux_armv7l.whl", hash = "sha256:5a324650a38f2d1dfac12305d5ee5f83265c805513e2e2b6dae7c52e3bf9eee6", size = 12229581, upload-time = "2026-09-10T16:21:45.287Z" }, + { url = "https://files.pythonhosted.org/packages/3a/53/1248cf11cbba23e2b9b9e0cfed8945b73956398d331e89b61c8e2297c4ec/sherpa_onnx-1.13.8-cp311-cp311-macosx_10_15_universal2.whl", hash = "sha256:b4c2e9c5fd12fe3ac6a21d76b45f16442d3b1d80691d81fbb0f8acfef250f5a3", size = 4434699, upload-time = "2026-09-10T14:54:30.83Z" }, + { url = "https://files.pythonhosted.org/packages/d6/be/c2b2a42ffcb22224fa95c62280cdc1f955f30f21805a2fda7fd182539628/sherpa_onnx-1.13.8-cp311-cp311-macosx_10_15_x86_64.whl", hash = "sha256:a00d9ceeb4d9531d2f5bd90d194d53408461c5aa6ff46e65f8776ed37007f3d0", size = 2332701, upload-time = "2026-09-10T15:01:40.938Z" }, + { url = "https://files.pythonhosted.org/packages/a9/74/dafb3c1c1ff82fc00abd098ade44811e48f3af88b4e3c2f1628567d0e80b/sherpa_onnx-1.13.8-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:8bb2b86ce44b5c5bb9949977177ddc36954c51097f81284d5ac48588e821ec07", size = 2141306, upload-time = "2026-09-10T15:36:50.677Z" }, + { url = "https://files.pythonhosted.org/packages/c6/02/f2300e5cb07a611afcac5699a97c8d6f229f53a7c751882818ed40a7f72f/sherpa_onnx-1.13.8-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:b56808d19a79368dcaa507d02a4c1ce537ca713fab9fedd91256b4ec599c6bf7", size = 4177610, upload-time = "2026-09-10T15:11:29.024Z" }, + { url = "https://files.pythonhosted.org/packages/ae/c6/1fe91047af08b30806f18fac55639037d3cf0247aeec97b7b7884407d6bf/sherpa_onnx-1.13.8-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:94fd1476b56ed36b851da8db9bd5144e92f669007ad54e0d8094913e4fbf1418", size = 4398381, upload-time = "2026-09-10T15:05:05.363Z" }, + { url = "https://files.pythonhosted.org/packages/2a/d2/b7601292ccc18f35cdb39a82f1bbb11e4ffe210bd235179c9da3715761f1/sherpa_onnx-1.13.8-cp311-cp311-win32.whl", hash = "sha256:de271d3338358dfb7cea1f780e7d99738437c0d9969f11b5bcf5d6660c6734b7", size = 1964883, upload-time = "2026-09-10T15:23:22.705Z" }, + { url = "https://files.pythonhosted.org/packages/a8/95/9325a66149d5a54c53161eba9ba77f8ea7068dc8f921c3d8b75174261211/sherpa_onnx-1.13.8-cp311-cp311-win_amd64.whl", hash = "sha256:171e6fac715bae20e11829e8dbfc70ed990ede1b35e6332f8121b27691a002de", size = 2283357, upload-time = "2026-09-10T15:52:44.166Z" }, + { url = "https://files.pythonhosted.org/packages/86/e6/e86545b64d55e134f9eb15641d69943bc34b68d9db187043c6e95d002123/sherpa_onnx-1.13.8-cp311-cp311-win_arm64.whl", hash = "sha256:41c3433fc044936808a9799b89a5e4c4f7b79d68eda8600d900a3f3724074099", size = 2242706, upload-time = "2026-09-10T15:16:42.968Z" }, + { url = "https://files.pythonhosted.org/packages/e0/74/f88231aa67e7713b2509da3aa53756b9c85f42fc7a3abe3a4b825e2974bc/sherpa_onnx-1.13.8-cp312-cp312-linux_armv7l.whl", hash = "sha256:6e50efbe69ff07f8b62d2d01f9d7f6e5854033be4b8d940fe9b575e873b22584", size = 12230846, upload-time = "2026-09-10T16:56:15.576Z" }, + { url = "https://files.pythonhosted.org/packages/a0/35/ed6cc5d2e29f29c9778975bec44ac5e8069e1a3528f7ff1306cbdcdab585/sherpa_onnx-1.13.8-cp312-cp312-macosx_10_15_universal2.whl", hash = "sha256:17ff957b1b33849671262ddbac51815a5a82286f963091bddf85540c9e728c12", size = 4457476, upload-time = "2026-09-10T15:45:38.598Z" }, + { url = "https://files.pythonhosted.org/packages/36/e6/a19257d7c60bf04c07f8d0772b943b6596b6b10cfb0694b7794d36630d40/sherpa_onnx-1.13.8-cp312-cp312-macosx_10_15_x86_64.whl", hash = "sha256:ef218d6545a9dcc2aff7043f39026c7d78d6cf46a3876d450d508feebffc00a9", size = 2350796, upload-time = "2026-09-10T14:23:58.591Z" }, + { url = "https://files.pythonhosted.org/packages/78/6e/8d9b92e95896ec500084e706529f5637eff797246e74c498922f6aa13a53/sherpa_onnx-1.13.8-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:04b9268c348e9f4bd61315754ad06b02a3c185d3d7770acd8e2dcee9c2e039f9", size = 2146858, upload-time = "2026-09-10T14:49:23.355Z" }, + { url = "https://files.pythonhosted.org/packages/1f/5f/22e1571146b2c0b581657562d5919b69da13ed93dc70049f42d45fe9e84c/sherpa_onnx-1.13.8-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:2520b1e7b779a29a493a8cec893612218dfad7304786bf3f64df2ff7ac6c3302", size = 4176945, upload-time = "2026-09-10T15:21:43.965Z" }, + { url = "https://files.pythonhosted.org/packages/13/78/2f712b7408ddbb64614c20388912414705b728516224e737b63f2adec94a/sherpa_onnx-1.13.8-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:6949773017647febc0c3696dffb2c67dd3febd4737b87e3dd135a42773704a06", size = 4400575, upload-time = "2026-09-10T14:49:42.265Z" }, + { url = "https://files.pythonhosted.org/packages/39/62/a6cc995ea40fd88a14b5d312d7b86b49ab65213e501f15fabe00cf79e1e5/sherpa_onnx-1.13.8-cp312-cp312-win32.whl", hash = "sha256:90e0c7b16fe6be32361dc3f6d2a66e8f74e8085079e91da1bee1acea772b769d", size = 1965883, upload-time = "2026-09-10T15:11:47.827Z" }, + { url = "https://files.pythonhosted.org/packages/24/af/33e9dcaa527bbc0c2526959713626607a46da2d7b19871b7773d8607d356/sherpa_onnx-1.13.8-cp312-cp312-win_amd64.whl", hash = "sha256:566e49ad3fb2aa8ceb723a52f9809aa6c2ee0a5eb3b0946c3fd2374af807a5c6", size = 2287856, upload-time = "2026-09-10T15:39:28.995Z" }, + { url = "https://files.pythonhosted.org/packages/ea/0e/237f9baaa3da5c029a06803a7587d4717ec191284d6f3972c679683ad6d8/sherpa_onnx-1.13.8-cp312-cp312-win_arm64.whl", hash = "sha256:1613f14db38b6936098c660f610baee95541378edbdec39a973a591d8df58f5b", size = 2249666, upload-time = "2026-09-10T14:54:55.475Z" }, + { url = "https://files.pythonhosted.org/packages/c8/c5/172fddb0ab34bed7f645df376d4c2e19deb5b3bc336ca06e1f7a52777e48/sherpa_onnx-1.13.8-cp313-cp313-linux_armv7l.whl", hash = "sha256:bbd93a09fa01fb32c222614a7f939495d8966fc15e1cb3943d7a79d3fe7291bd", size = 12230180, upload-time = "2026-09-10T16:28:11.636Z" }, + { url = "https://files.pythonhosted.org/packages/cf/7c/21316b8e438598da6f54cc5eea6994fbb9d508caf4afd8292b76802b244c/sherpa_onnx-1.13.8-cp313-cp313-macosx_10_15_universal2.whl", hash = "sha256:d2f50ad1b1e1918c0358270846288bdc1636b130bea4d9d85088a14bb7de69b1", size = 4458011, upload-time = "2026-09-10T16:02:34.034Z" }, + { url = "https://files.pythonhosted.org/packages/7d/4d/b442afd30eff64b83112b54fa866c5d2e3a5aeec13ee8bd665c7b146fba1/sherpa_onnx-1.13.8-cp313-cp313-macosx_10_15_x86_64.whl", hash = "sha256:5f8df54514af5025b9d428403dd9a27f6fe252b5e162311bf5c47228a50d0cc5", size = 2351015, upload-time = "2026-09-10T14:37:10.823Z" }, + { url = "https://files.pythonhosted.org/packages/ce/7b/9829c8e222e6103fd3571523252974c1095c49ae7d1511babf46c4f0d6ea/sherpa_onnx-1.13.8-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:2c5a1799c326715b1ee69dae3be48883ddc6fb922fb813067606f7e9a29ea234", size = 2147122, upload-time = "2026-09-10T15:47:29.125Z" }, + { url = "https://files.pythonhosted.org/packages/13/e3/115476caa9f80cd5f55e4e7b777a6c2d01a133d10f8fae574ba869fc48c3/sherpa_onnx-1.13.8-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:916ec38242e779ee3a99f7b161bd1bf3b02a7fad34cb073d091021c0469e0b58", size = 4177700, upload-time = "2026-09-10T14:27:20.564Z" }, + { url = "https://files.pythonhosted.org/packages/8f/eb/b77acde02d9eee359436ade9239d9ade4351b09fbe85387a293f341c19ed/sherpa_onnx-1.13.8-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:0e5d870fb0648befb94e260c11de4cf4b293b2a90a63e3a6a2869be698927bc4", size = 4401242, upload-time = "2026-09-10T15:28:26.857Z" }, + { url = "https://files.pythonhosted.org/packages/04/3d/a5f41332fe672e139f542541cb1afecf96addcbb72ee4bf84b0f9ba3dd22/sherpa_onnx-1.13.8-cp313-cp313-win32.whl", hash = "sha256:4bbc4dbbd539fd83c28aa7a076e0e5e1c5c4c4cc9e31fc5bd956b4e8b032fcfc", size = 1966530, upload-time = "2026-09-10T15:51:56.826Z" }, + { url = "https://files.pythonhosted.org/packages/52/1f/a96759b33dab99647e57f94040d678c332e0d7eb595127d1e0b4e0b9af27/sherpa_onnx-1.13.8-cp313-cp313-win_amd64.whl", hash = "sha256:195455d8d6cc49f616e9d459b7d08f10a32daf2ddcbe90d4d479a8ba9aab28bf", size = 2288187, upload-time = "2026-09-10T15:59:35.179Z" }, + { url = "https://files.pythonhosted.org/packages/71/96/86a9571b95336e9968519acedcb33f17fb4c3512e961344f6daed47636cc/sherpa_onnx-1.13.8-cp313-cp313-win_arm64.whl", hash = "sha256:15e2d2ac2524e42efb40057f6e1d7cc53529417e791f5cf1bdd721a0ec2b1f7e", size = 2245867, upload-time = "2026-09-10T14:52:43.668Z" }, + { url = "https://files.pythonhosted.org/packages/58/e9/e7669a9f7a927e33f20b9bf98ffbe8d11c7f60c6e8f492bd0bc59dcb94d7/sherpa_onnx-1.13.8-cp314-cp314-android_24_arm64_v8a.whl", hash = "sha256:533440358bbfc5796bc8cf85fe2bc08d0c91de5200fc03da2d3eb656caec705b", size = 2364130, upload-time = "2026-09-10T14:04:10.482Z" }, + { url = "https://files.pythonhosted.org/packages/83/68/d76f8dd62b7475a0583c57cfaaebbb1de693b42405a17796613fa7592227/sherpa_onnx-1.13.8-cp314-cp314-android_24_armeabi_v7a.whl", hash = "sha256:4757a983a0b2757ae55e6179b0f5369381316b1efcdfd5a35b6ccb341fae3daa", size = 2433241, upload-time = "2026-09-10T14:04:22.379Z" }, + { url = "https://files.pythonhosted.org/packages/a5/fe/86cbd751f3b36f59ac0333b1ebbf91312c2fc5031c6cf54095a54735b690/sherpa_onnx-1.13.8-cp314-cp314-android_24_x86.whl", hash = "sha256:c0bd1249e4ff8a4f93c8b7eb7e906b8d894738556379b28e78817cbdf996eb72", size = 2527240, upload-time = "2026-09-10T14:10:43.529Z" }, + { url = "https://files.pythonhosted.org/packages/a4/00/bd069882f9a707a7e0cdad32f9447b95e756d4fe1421419ba0e6d9a2fcd2/sherpa_onnx-1.13.8-cp314-cp314-android_24_x86_64.whl", hash = "sha256:e787ab8e2c859bae1589b9dc729ca2d66096505db028f2969b1c3a380a986a38", size = 2588008, upload-time = "2026-09-10T14:35:08.781Z" }, + { url = "https://files.pythonhosted.org/packages/76/b9/832bab8dc5116bb29b0ff7cc1aed5122ab2115c6bb947140073f6930fe67/sherpa_onnx-1.13.8-cp314-cp314-linux_armv7l.whl", hash = "sha256:40a7a26e4af197f6b46e73e73cc6792c3401b12377e7d8d674766a5cb67ce4ac", size = 12228389, upload-time = "2026-09-10T16:44:33.115Z" }, + { url = "https://files.pythonhosted.org/packages/91/2c/4aee534542f2de43c3dadfbe7d436fdecfb28e4c6d5a12f75febdc3f5f8e/sherpa_onnx-1.13.8-cp314-cp314-macosx_10_15_universal2.whl", hash = "sha256:c8121c87eb80c86a8dbe2cf54ed11ca1150ae0001223bac902471e7b4608e1a9", size = 4461555, upload-time = "2026-09-10T15:58:23.745Z" }, + { url = "https://files.pythonhosted.org/packages/f5/fb/e8a289419f38ed9eb1882d22c12a2680254e5eb945e8e2517c5b2281dd23/sherpa_onnx-1.13.8-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:22f699e1fb2e8dafe02b1521ce74e0928b38cb68ad19f481380acf51e6a6965b", size = 2351239, upload-time = "2026-09-10T15:07:53.4Z" }, + { url = "https://files.pythonhosted.org/packages/33/f5/ff7f45799a1a77f165410fce5080be2cb59cfc98375bf4a56d7d1fe80540/sherpa_onnx-1.13.8-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:7fa6ec595a8b4bba2b91e6f8e58647777050ac4eb9845f5a5026c20636c0e2be", size = 2149334, upload-time = "2026-09-10T15:19:02.491Z" }, + { url = "https://files.pythonhosted.org/packages/4f/9a/51821829b5735b3d7ce607992a62fe93d5f7215cda324c11191c77d49b9b/sherpa_onnx-1.13.8-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:45f16c9fabc3d7ce31ba9722ff8196627ddcaa5ffd02413ad0664ad1952ba419", size = 4182835, upload-time = "2026-09-10T14:39:02.405Z" }, + { url = "https://files.pythonhosted.org/packages/2d/e2/b1af63c1f6a9e0025cbc9aa3b1814ba97ede63db0ad9f297bb9f268fdfc2/sherpa_onnx-1.13.8-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:97309e4b5d9850490085da61e35b0fca002a39ec9da93cd9e21d2d0dcc1cc62a", size = 4403669, upload-time = "2026-09-10T14:53:18.178Z" }, + { url = "https://files.pythonhosted.org/packages/b3/34/48d01cc8581f3d4354eb14f1c91a04b52c4c2024ae7c3ce465355aa6d227/sherpa_onnx-1.13.8-cp314-cp314-win32.whl", hash = "sha256:bf5f95bfd992f5dd3f897f57fc13f9d0174e16e48c19d43a24923cc5fb61ded1", size = 2006786, upload-time = "2026-09-10T15:39:06.995Z" }, + { url = "https://files.pythonhosted.org/packages/24/a5/100a7c200e696b2756838b0bdb83e9cf96f62e524d44d43daba0a8e9d6fe/sherpa_onnx-1.13.8-cp314-cp314-win_amd64.whl", hash = "sha256:1ced230bdd02f4d50106ce5c41e5f9142b45248e67ca532c84439d0475932bb8", size = 2353481, upload-time = "2026-09-10T15:51:06.243Z" }, + { url = "https://files.pythonhosted.org/packages/1f/38/ce056f2cdcfa9abcf40ea572cc8ece2fac4fe19a5fe2f1ec589cf71e412c/sherpa_onnx-1.13.8-cp314-cp314-win_arm64.whl", hash = "sha256:ecce26fe96e4b733356568fac1e99df7038e9871b94b70d9f731e2ea8656db67", size = 2314981, upload-time = "2026-09-10T15:08:22.649Z" }, +] + +[[package]] +name = "sherpa-onnx-core" +version = "1.13.8" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/be/56/85f4f0830a34e44ff09490db07fddc971dad718b0379ac493f4e615e5ee1/sherpa_onnx_core-1.13.8-py3-none-android_24_x86_64.whl", hash = "sha256:8ac9a05109105cd263ca16f09ffc7ec077979c95ddf185ff3d62b83dd929ce29", size = 11133563, upload-time = "2026-09-10T14:35:13.398Z" }, + { url = "https://files.pythonhosted.org/packages/de/7d/e5a1de0269ed3d646f430beb1156f17b39899a1cae612dcf6e29a2867248/sherpa_onnx_core-1.13.8-py3-none-win32.whl", hash = "sha256:44f7daed251d96da4a26486fcc42142cf863d8ba760c6bfececd0f210de40414", size = 14772677, upload-time = "2026-09-10T14:54:26.666Z" }, + { url = "https://files.pythonhosted.org/packages/94/38/64356ad97f68fffcf01fe7545407ad18a2ea86c1ffae5b0950f7fac73638/sherpa_onnx_core-1.13.8-py3-none-win_amd64.whl", hash = "sha256:5579e80196d516e6dae23c8f629292ce3142ab8869925d32b94612fd86f93733", size = 16903581, upload-time = "2026-09-10T15:10:50.349Z" }, + { url = "https://files.pythonhosted.org/packages/e4/e0/236bde7b6b9909d2f7f4282b9e18f5fa8d64b841e63265b508a86c3687a8/sherpa_onnx_core-1.13.8-py3-none-win_arm64.whl", hash = "sha256:85e9b12e9e985d73ad9165a1122c396e2c9384256a4023cd7cf9c91bc8d32073", size = 16067491, upload-time = "2026-09-10T14:16:26.19Z" }, +] + [[package]] name = "sniffio" version = "1.3.1" @@ -2210,7 +2583,7 @@ name = "sympy" version = "1.14.0" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "mpmath", marker = "sys_platform == 'darwin'" }, + { name = "mpmath" }, ] sdist = { url = "https://files.pythonhosted.org/packages/83/d3/803453b36afefb7c2bb238361cd4ae6125a569b4db67cd9e79846ba2d68c/sympy-1.14.0.tar.gz", hash = "sha256:d3d3fe8df1e5a0b42f0e7bdf50541697dbe7d23746e894990c030e2b05e72517", size = 7793921, upload-time = "2025-04-27T18:05:01.611Z" } wheels = [ @@ -2314,6 +2687,91 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/cd/2b/2be299bab55fc595e3d38567edb1a87f86e594842968fa9515a07bdcf422/tokenizers-0.23.1-cp310-abi3-win_arm64.whl", hash = "sha256:a26197957d8e4425dfba746315f3c425ea00cfa8367c5fbc4ec73447893dcea9", size = 2664127, upload-time = "2026-04-27T14:43:26.949Z" }, ] +[[package]] +name = "torch" +version = "2.9.1" +source = { registry = "https://pypi.org/simple" } +resolution-markers = [ + "python_full_version >= '3.14' and sys_platform != 'darwin' and sys_platform != 'win32'", + "python_full_version >= '3.12' and python_full_version < '3.14' and sys_platform != 'darwin' and sys_platform != 'win32'", + "python_full_version >= '3.14' and sys_platform == 'darwin'", + "python_full_version >= '3.12' and python_full_version < '3.14' and sys_platform == 'darwin'", + "python_full_version < '3.12' and sys_platform != 'darwin' and sys_platform != 'win32'", + "python_full_version < '3.12' and sys_platform == 'darwin'", +] +dependencies = [ + { name = "filelock", marker = "sys_platform != 'win32'" }, + { name = "fsspec", marker = "sys_platform != 'win32'" }, + { name = "jinja2", marker = "sys_platform != 'win32'" }, + { name = "networkx", marker = "sys_platform != 'win32'" }, + { name = "nvidia-cublas-cu12", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "nvidia-cuda-cupti-cu12", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "nvidia-cuda-nvrtc-cu12", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "nvidia-cuda-runtime-cu12", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "nvidia-cudnn-cu12", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "nvidia-cufft-cu12", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "nvidia-cufile-cu12", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "nvidia-curand-cu12", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "nvidia-cusolver-cu12", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "nvidia-cusparse-cu12", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "nvidia-cusparselt-cu12", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "nvidia-nccl-cu12", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "nvidia-nvjitlink-cu12", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "nvidia-nvshmem-cu12", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "nvidia-nvtx-cu12", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "setuptools", marker = "python_full_version >= '3.12' and sys_platform != 'win32'" }, + { name = "sympy", marker = "sys_platform != 'win32'" }, + { name = "triton", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, + { name = "typing-extensions", marker = "sys_platform != 'win32'" }, +] +wheels = [ + { url = "https://files.pythonhosted.org/packages/15/db/c064112ac0089af3d2f7a2b5bfbabf4aa407a78b74f87889e524b91c5402/torch-2.9.1-cp311-cp311-manylinux_2_28_aarch64.whl", hash = "sha256:62b3fd888277946918cba4478cf849303da5359f0fb4e3bfb86b0533ba2eaf8d", size = 104220430, upload-time = "2025-11-12T15:20:31.705Z" }, + { url = "https://files.pythonhosted.org/packages/56/be/76eaa36c9cd032d3b01b001e2c5a05943df75f26211f68fae79e62f87734/torch-2.9.1-cp311-cp311-manylinux_2_28_x86_64.whl", hash = "sha256:d033ff0ac3f5400df862a51bdde9bad83561f3739ea0046e68f5401ebfa67c1b", size = 899821446, upload-time = "2025-11-12T15:20:15.544Z" }, + { url = "https://files.pythonhosted.org/packages/1e/ce/7d251155a783fb2c1bb6837b2b7023c622a2070a0a72726ca1df47e7ea34/torch-2.9.1-cp311-none-macosx_11_0_arm64.whl", hash = "sha256:52347912d868653e1528b47cafaf79b285b98be3f4f35d5955389b1b95224475", size = 74463887, upload-time = "2025-11-12T15:20:36.611Z" }, + { url = "https://files.pythonhosted.org/packages/0f/27/07c645c7673e73e53ded71705045d6cb5bae94c4b021b03aa8d03eee90ab/torch-2.9.1-cp312-cp312-manylinux_2_28_aarch64.whl", hash = "sha256:da5f6f4d7f4940a173e5572791af238cb0b9e21b1aab592bd8b26da4c99f1cd6", size = 104126592, upload-time = "2025-11-12T15:20:41.62Z" }, + { url = "https://files.pythonhosted.org/packages/19/17/e377a460603132b00760511299fceba4102bd95db1a0ee788da21298ccff/torch-2.9.1-cp312-cp312-manylinux_2_28_x86_64.whl", hash = "sha256:27331cd902fb4322252657f3902adf1c4f6acad9dcad81d8df3ae14c7c4f07c4", size = 899742281, upload-time = "2025-11-12T15:22:17.602Z" }, + { url = "https://files.pythonhosted.org/packages/6e/ab/07739fd776618e5882661d04c43f5b5586323e2f6a2d7d84aac20d8f20bd/torch-2.9.1-cp312-none-macosx_11_0_arm64.whl", hash = "sha256:c0d25d1d8e531b8343bea0ed811d5d528958f1dcbd37e7245bc686273177ad7e", size = 74479191, upload-time = "2025-11-12T15:21:25.816Z" }, + { url = "https://files.pythonhosted.org/packages/20/60/8fc5e828d050bddfab469b3fe78e5ab9a7e53dda9c3bdc6a43d17ce99e63/torch-2.9.1-cp313-cp313-manylinux_2_28_aarch64.whl", hash = "sha256:c29455d2b910b98738131990394da3e50eea8291dfeb4b12de71ecf1fdeb21cb", size = 104135743, upload-time = "2025-11-12T15:21:34.936Z" }, + { url = "https://files.pythonhosted.org/packages/f2/b7/6d3f80e6918213babddb2a37b46dbb14c15b14c5f473e347869a51f40e1f/torch-2.9.1-cp313-cp313-manylinux_2_28_x86_64.whl", hash = "sha256:524de44cd13931208ba2c4bde9ec7741fd4ae6bfd06409a604fc32f6520c2bc9", size = 899749493, upload-time = "2025-11-12T15:24:36.356Z" }, + { url = "https://files.pythonhosted.org/packages/28/0e/2a37247957e72c12151b33a01e4df651d9d155dd74d8cfcbfad15a79b44a/torch-2.9.1-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:5be4bf7496f1e3ffb1dd44b672adb1ac3f081f204c5ca81eba6442f5f634df8e", size = 74830751, upload-time = "2025-11-12T15:21:43.792Z" }, + { url = "https://files.pythonhosted.org/packages/4b/f7/7a18745edcd7b9ca2381aa03353647bca8aace91683c4975f19ac233809d/torch-2.9.1-cp313-cp313t-manylinux_2_28_aarch64.whl", hash = "sha256:30a3e170a84894f3652434b56d59a64a2c11366b0ed5776fab33c2439396bf9a", size = 104142929, upload-time = "2025-11-12T15:21:48.319Z" }, + { url = "https://files.pythonhosted.org/packages/f4/dd/f1c0d879f2863ef209e18823a988dc7a1bf40470750e3ebe927efdb9407f/torch-2.9.1-cp313-cp313t-manylinux_2_28_x86_64.whl", hash = "sha256:8301a7b431e51764629208d0edaa4f9e4c33e6df0f2f90b90e261d623df6a4e2", size = 899748978, upload-time = "2025-11-12T15:23:04.568Z" }, + { url = "https://files.pythonhosted.org/packages/40/60/71c698b466dd01e65d0e9514b5405faae200c52a76901baf6906856f17e4/torch-2.9.1-cp313-none-macosx_11_0_arm64.whl", hash = "sha256:2c14b3da5df416cf9cb5efab83aa3056f5b8cd8620b8fde81b4987ecab730587", size = 74480347, upload-time = "2025-11-12T15:21:57.648Z" }, + { url = "https://files.pythonhosted.org/packages/48/50/c4b5112546d0d13cc9eaa1c732b823d676a9f49ae8b6f97772f795874a03/torch-2.9.1-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:1edee27a7c9897f4e0b7c14cfc2f3008c571921134522d5b9b5ec4ebbc69041a", size = 74433245, upload-time = "2025-11-12T15:22:39.027Z" }, + { url = "https://files.pythonhosted.org/packages/81/c9/2628f408f0518b3bae49c95f5af3728b6ab498c8624ab1e03a43dd53d650/torch-2.9.1-cp314-cp314-manylinux_2_28_aarch64.whl", hash = "sha256:19d144d6b3e29921f1fc70503e9f2fc572cde6a5115c0c0de2f7ca8b1483e8b6", size = 104134804, upload-time = "2025-11-12T15:22:35.222Z" }, + { url = "https://files.pythonhosted.org/packages/28/fc/5bc91d6d831ae41bf6e9e6da6468f25330522e92347c9156eb3f1cb95956/torch-2.9.1-cp314-cp314-manylinux_2_28_x86_64.whl", hash = "sha256:c432d04376f6d9767a9852ea0def7b47a7bbc8e7af3b16ac9cf9ce02b12851c9", size = 899747132, upload-time = "2025-11-12T15:23:36.068Z" }, + { url = "https://files.pythonhosted.org/packages/bd/b2/2d15a52516b2ea3f414643b8de68fa4cb220d3877ac8b1028c83dc8ca1c4/torch-2.9.1-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:cb10896a1f7fedaddbccc2017ce6ca9ecaaf990f0973bdfcf405439750118d2c", size = 74823558, upload-time = "2025-11-12T15:22:43.392Z" }, + { url = "https://files.pythonhosted.org/packages/86/5c/5b2e5d84f5b9850cd1e71af07524d8cbb74cba19379800f1f9f7c997fc70/torch-2.9.1-cp314-cp314t-manylinux_2_28_aarch64.whl", hash = "sha256:0a2bd769944991c74acf0c4ef23603b9c777fdf7637f115605a4b2d8023110c7", size = 104145788, upload-time = "2025-11-12T15:23:52.109Z" }, + { url = "https://files.pythonhosted.org/packages/a9/8c/3da60787bcf70add986c4ad485993026ac0ca74f2fc21410bc4eb1bb7695/torch-2.9.1-cp314-cp314t-manylinux_2_28_x86_64.whl", hash = "sha256:07c8a9660bc9414c39cac530ac83b1fb1b679d7155824144a40a54f4a47bfa73", size = 899735500, upload-time = "2025-11-12T15:24:08.788Z" }, +] + +[[package]] +name = "torch" +version = "2.9.1+cu128" +source = { registry = "https://download.pytorch.org/whl/cu128" } +resolution-markers = [ + "python_full_version >= '3.14' and sys_platform == 'win32'", + "python_full_version >= '3.12' and python_full_version < '3.14' and sys_platform == 'win32'", + "python_full_version < '3.12' and sys_platform == 'win32'", +] +dependencies = [ + { name = "filelock", marker = "sys_platform == 'win32'" }, + { name = "fsspec", marker = "sys_platform == 'win32'" }, + { name = "jinja2", marker = "sys_platform == 'win32'" }, + { name = "networkx", marker = "sys_platform == 'win32'" }, + { name = "setuptools", marker = "python_full_version >= '3.12' and sys_platform == 'win32'" }, + { name = "sympy", marker = "sys_platform == 'win32'" }, + { name = "typing-extensions", marker = "sys_platform == 'win32'" }, +] +wheels = [ + { url = "https://download-r2.pytorch.org/whl/cu128/torch-2.9.1%2Bcu128-cp311-cp311-win_amd64.whl", hash = "sha256:633005a3700e81b5be0df2a7d3c1d48aced23ed927653797a3bd2b144a3aeeb6", upload-time = "2026-01-26T16:54:12Z" }, + { url = "https://download-r2.pytorch.org/whl/cu128/torch-2.9.1%2Bcu128-cp312-cp312-win_amd64.whl", hash = "sha256:3a01f0b64c10a82d444d9fd06b3e8c567b1158b76b2764b8f51bfd8f535064b0", upload-time = "2026-01-26T16:54:32Z" }, + { url = "https://download-r2.pytorch.org/whl/cu128/torch-2.9.1%2Bcu128-cp313-cp313-win_amd64.whl", hash = "sha256:ad9183864acdd99fc5143d7ca9d3d2e7ddfc9a9600ff43217825d4e5e9855ccc", upload-time = "2026-01-26T16:55:00Z" }, + { url = "https://download-r2.pytorch.org/whl/cu128/torch-2.9.1%2Bcu128-cp313-cp313t-win_amd64.whl", hash = "sha256:24420e430e77136f7079354134b34e7ba9d87e539f5ac84c33b08e5c13412ebe", upload-time = "2026-01-26T16:55:48Z" }, + { url = "https://download-r2.pytorch.org/whl/cu128/torch-2.9.1%2Bcu128-cp314-cp314-win_amd64.whl", hash = "sha256:7bcd40cbffac475b478d6ce812f03da84e9a4894956efb89c3b7bcca5dbd4f91", upload-time = "2026-01-26T16:56:12Z" }, + { url = "https://download-r2.pytorch.org/whl/cu128/torch-2.9.1%2Bcu128-cp314-cp314t-win_amd64.whl", hash = "sha256:0c784b600959ec70ee01cb23e8bc870a0e0475af30378ff5e39f4abed8b7c1cc", upload-time = "2026-01-26T16:56:38Z" }, +] + [[package]] name = "tqdm" version = "4.70.0" @@ -2326,6 +2784,40 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/f9/1c/01bfd571a64e7f270e6bab5e33777debe0edc56759233ce84f27dec92d14/tqdm-4.70.0-py3-none-any.whl", hash = "sha256:7f585706bfddbdebf89daac705b2dfcc16890130727d3197ca62c732b4310953", size = 80184, upload-time = "2026-07-27T11:33:13.167Z" }, ] +[[package]] +name = "transformers" +version = "5.17.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "huggingface-hub" }, + { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.12'" }, + { name = "numpy", version = "2.5.2", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.12'" }, + { name = "packaging" }, + { name = "pyyaml" }, + { name = "regex" }, + { name = "safetensors" }, + { name = "tokenizers" }, + { name = "tqdm" }, + { name = "typer" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/0e/9e/750649904a065007a838981785b2bd8d9ff26154c6c341ac67d0b7f82c68/transformers-5.17.0.tar.gz", hash = "sha256:a153be279169b55b92d8000bf4af294aed684503d091cca7804da2dd8a9de000", size = 9817878, upload-time = "2026-09-09T15:39:56.886Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e8/d0/c502b60d684adbd98a8dc7d5bb866842772b816ac4354e4608be240041ae/transformers-5.17.0-py3-none-any.whl", hash = "sha256:78ec1ce21579b38dfb83950a0658cd119f87212a2fcfdff478096ce9d6c03801", size = 12295140, upload-time = "2026-09-09T15:39:53.746Z" }, +] + +[[package]] +name = "triton" +version = "3.5.1" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/b0/72/ec90c3519eaf168f22cb1757ad412f3a2add4782ad3a92861c9ad135d886/triton-3.5.1-cp311-cp311-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:61413522a48add32302353fdbaaf92daaaab06f6b5e3229940d21b5207f47579", size = 170425802, upload-time = "2025-11-11T17:40:53.209Z" }, + { url = "https://files.pythonhosted.org/packages/f2/50/9a8358d3ef58162c0a415d173cfb45b67de60176e1024f71fbc4d24c0b6d/triton-3.5.1-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d2c6b915a03888ab931a9fd3e55ba36785e1fe70cbea0b40c6ef93b20fc85232", size = 170470207, upload-time = "2025-11-11T17:41:00.253Z" }, + { url = "https://files.pythonhosted.org/packages/27/46/8c3bbb5b0a19313f50edcaa363b599e5a1a5ac9683ead82b9b80fe497c8d/triton-3.5.1-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:f3f4346b6ebbd4fad18773f5ba839114f4826037c9f2f34e0148894cd5dd3dba", size = 170470410, upload-time = "2025-11-11T17:41:06.319Z" }, + { url = "https://files.pythonhosted.org/packages/37/92/e97fcc6b2c27cdb87ce5ee063d77f8f26f19f06916aa680464c8104ef0f6/triton-3.5.1-cp313-cp313t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:0b4d2c70127fca6a23e247f9348b8adde979d2e7a20391bfbabaac6aebc7e6a8", size = 170579924, upload-time = "2025-11-11T17:41:12.455Z" }, + { url = "https://files.pythonhosted.org/packages/a4/e6/c595c35e5c50c4bc56a7bac96493dad321e9e29b953b526bbbe20f9911d0/triton-3.5.1-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d0637b1efb1db599a8e9dc960d53ab6e4637db7d4ab6630a0974705d77b14b60", size = 170480488, upload-time = "2025-11-11T17:41:18.222Z" }, + { url = "https://files.pythonhosted.org/packages/16/b5/b0d3d8b901b6a04ca38df5e24c27e53afb15b93624d7fd7d658c7cd9352a/triton-3.5.1-cp314-cp314t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:bac7f7d959ad0f48c0e97d6643a1cc0fd5786fe61cb1f83b537c6b2d54776478", size = 170582192, upload-time = "2025-11-11T17:41:23.963Z" }, +] + [[package]] name = "truststore" version = "0.10.4" @@ -2335,6 +2827,21 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/19/97/56608b2249fe206a67cd573bc93cd9896e1efb9e98bce9c163bcdc704b88/truststore-0.10.4-py3-none-any.whl", hash = "sha256:adaeaecf1cbb5f4de3b1959b42d41f6fab57b2b1666adb59e89cb0b53361d981", size = 18660, upload-time = "2025-08-12T18:49:01.46Z" }, ] +[[package]] +name = "typer" +version = "0.27.2" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "annotated-doc" }, + { name = "colorama", marker = "sys_platform == 'win32'" }, + { name = "rich" }, + { name = "shellingham" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/16/f7/57713ba479fd405eb76de31404b2c744c289e336b2d999511ebf51e496f7/typer-0.27.2.tar.gz", hash = "sha256:269b7eb9d3c202ca84b4bc9618cb04ebb43d3d4d1e567e4c768607232c05f945", size = 204045, upload-time = "2026-08-28T10:26:55.046Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/dc/bf/205d0004930ede8f542fb58f601526fccf4ae7626075ca1e6c4de5d3d652/typer-0.27.2-py3-none-any.whl", hash = "sha256:b3a5fc4342d5fc8fda8fc3010b1cf117e9249aab7fae800c2eff62fd3842d97d", size = 123130, upload-time = "2026-08-28T10:26:53.752Z" }, +] + [[package]] name = "typing-extensions" version = "4.14.0" @@ -2694,6 +3201,13 @@ build = [ voice-transcription = [ { name = "faster-whisper" }, { name = "opencc-python-reimplemented" }, + { name = "sherpa-onnx" }, + { name = "sherpa-onnx-core", marker = "sys_platform == 'win32'" }, +] +voice-transcription-gpu = [ + { name = "torch", version = "2.9.1", source = { registry = "https://pypi.org/simple" }, marker = "sys_platform != 'win32'" }, + { name = "torch", version = "2.9.1+cu128", source = { registry = "https://download.pytorch.org/whl/cu128" }, marker = "sys_platform == 'win32'" }, + { name = "transformers" }, ] [package.dev-dependencies] @@ -2737,8 +3251,13 @@ requires-dist = [ { name = "python-pptx", specifier = ">=1.0" }, { name = "pywin32", marker = "sys_platform == 'win32'", specifier = ">=310" }, { name = "requests", specifier = ">=2.32.4" }, + { name = "sherpa-onnx", marker = "extra == 'voice-transcription'", specifier = "==1.13.8" }, + { name = "sherpa-onnx-core", marker = "sys_platform == 'win32' and extra == 'voice-transcription'", specifier = "==1.13.8" }, { name = "sqlite-vec", specifier = "==0.1.9" }, { name = "tokenizers", specifier = "==0.23.1" }, + { name = "torch", marker = "sys_platform == 'win32' and extra == 'voice-transcription-gpu'", specifier = "==2.9.1", index = "https://download.pytorch.org/whl/cu128" }, + { name = "torch", marker = "sys_platform != 'win32' and extra == 'voice-transcription-gpu'", specifier = "==2.9.1" }, + { name = "transformers", marker = "extra == 'voice-transcription-gpu'", specifier = "==5.17.0" }, { name = "typing-extensions", specifier = ">=4.8.0" }, { name = "uvicorn", extras = ["standard"], specifier = ">=0.24.0" }, { name = "watchfiles", specifier = ">=1.1.0" }, @@ -2746,7 +3265,7 @@ requires-dist = [ { name = "yara-python", marker = "sys_platform == 'win32'", specifier = ">=4.5.2" }, { name = "zstandard", specifier = ">=0.23.0" }, ] -provides-extras = ["build", "voice-transcription"] +provides-extras = ["build", "voice-transcription", "voice-transcription-gpu"] [package.metadata.requires-dev] dev = [{ name = "pytest", specifier = ">=8.0.0" }] From 60685df61cbc792d7dfb378daca7b3f46393b31b Mon Sep 17 00:00:00 2001 From: xiaosheng <73678111+xiaoshengbao@users.noreply.github.com> Date: Tue, 22 Sep 2026 00:27:20 +0800 Subject: [PATCH 2/4] feat(stt): retain four model profiles and clarify settings copy --- docs/stt-integration-2026-09-21.md | 16 +- frontend/components/SettingsDialog.vue | 21 +-- src/wechat_decrypt_tool/asr_models.py | 10 +- .../resources/voice_models.json | 40 ----- src/wechat_decrypt_tool/runtime_settings.py | 4 +- .../voice_transcription.py | 86 +++------- tests/test_asr_upgrade.py | 62 ++++++- tests/test_voice_transcription.py | 8 +- tests/test_voice_transcription_manager.py | 156 +++++++++--------- tests/test_voice_transcription_settings.py | 5 +- 10 files changed, 190 insertions(+), 218 deletions(-) diff --git a/docs/stt-integration-2026-09-21.md b/docs/stt-integration-2026-09-21.md index 80f21f5f..5324ddd2 100644 --- a/docs/stt-integration-2026-09-21.md +++ b/docs/stt-integration-2026-09-21.md @@ -1,21 +1,21 @@ # 语音识别升级接入说明 -2026-09-21:源码已接入新模型,保留原 Whisper 模型 ID、用户选择及旧缓存。此次未生成或替换正式安装包。 +2026-09-22:软件仅提供 Zipformer CTC、Qwen3-ASR 0.6B CPU、Turbo、Qwen3-ASR 0.6B GPU 四个选项,默认选择 CTC。此次未生成或替换正式安装包。 ## 软件中的入口 -在设置的“语音识别模型”中下载、选择模型;聊天页的语音转写侧栏使用同一组选项。旧模型通过“兼容模型”展开,当前已选的旧模型始终可见。未安装的运行组件会显示原因,不能误选成可用模型。 +在设置的“语音识别模型”中下载、选择模型;聊天页的语音转写侧栏使用同一组选项。Tiny、Base、Small、Medium、Large v3 和 Qwen 1.7B 已从列表及下载入口移除。未安装的运行组件会显示原因,不能误选成可用模型。 | 档位 | 选项 | 运行设备 | | --- | --- | --- | | 低配极速 | Zipformer CTC INT8 | CPU;中英文、无标点 | | 中配质量优先 | Qwen3-ASR 0.6B ONNX INT4 | CPU;建议 16 GB 内存 | | GPU 速度优先 | 原 Whisper Turbo | NVIDIA GPU,保留原 CPU 回退逻辑 | -| GPU 质量优先 | Qwen3-ASR 0.6B / 1.7B | NVIDIA GPU,需单独的 Qwen GPU 运行组件 | +| GPU 质量优先 | Qwen3-ASR 0.6B | NVIDIA GPU,需单独的 Qwen GPU 运行组件 | CPU/GPU 版本是独立选项。选中新模型会设置匹配的设备;环境变量锁定设备时不会覆盖。Qwen GPU 失败会给出错误,不会悄悄切换另一模型。新后端单进程串行复用,避免批量并发创建多份模型;取消时终止工作进程,下一条任务可以重新加载。空闲 120 秒后进程自动释放。 -本机四个新模型和 Turbo 的文件已安装到 `%APPDATA%/wechat-data-analysis-desktop/voice_models/`,新模型复制前已校验固定版本的 SHA-256。首次接入保留原选择;后续应用户要求在桌面应用验收,最终启用 Qwen3-ASR 1.7B GPU。 +此前验收的四个新模型和 Turbo 的文件已安装到 `%APPDATA%/wechat-data-analysis-desktop/voice_models/`,新模型复制前已校验固定版本的 SHA-256。停用模型的文件及历史转写不会自动删除。旧 Whisper CPU 配置读取时转到 CTC,CUDA 配置转到 Turbo;Qwen 1.7B 转到 0.6B GPU。界面显示迁移提示,用户可重新选择。迁移不改写原设置,环境变量固定的设备不会被覆盖;若与模型冲突,界面会提示选择匹配设备。 ## 启动及构建 @@ -51,7 +51,7 @@ uv sync --extra voice-transcription --extra voice-transcription-gpu - Qwen ONNX 按实际 tokenizer 编码角色提示,避免社区示例的固定 token ID 不匹配;CPU 特征提取不依赖 PyTorch。 - Windows Whisper CUDA 可以复用已安装 PyTorch 中的 CUDA 12 DLL,解决只有系统 CUDA 13 时的依赖缺失。 -## 本机验收 +## 初次接入验收(历史数据,包含现已停用的 1.7B) 通过项目正式 `VoiceTranscriptionService.transcribe_voice` 读取并解码 20 条真实 SILK,四模型共 80 次成功;写缓存和批量缓存查询均验证通过。缓存写入测试目录,没有改写原会话的转写缓存。音频总长 202.54 秒。 @@ -77,3 +77,9 @@ uv sync --extra voice-transcription --extra voice-transcription-gpu PR 基于上游 `main` 的 `2646cfdf`,只移入本次语音升级。上游尚无开发分支上的导出语音选项,因此没有带入相关导出界面和导出状态改动。 独立工作区后端相关用例为 131 通过、1 跳过、1 失败;唯一失败是 `test_export_option_is_wired_from_dialog_to_backend`,其要求的 `exportTranscribeVoice` 控件在上游不存在。已从未修改的 `origin/main` 提取该测试及其全部输入文件,独立复现相同断言失败;本 PR 不修改这项测试。语音组件 63 项、设置与桌面启动契约 28 项均通过。使用符合前端版本要求的 Node 重新安装锁定依赖后,Nuxt 生产静态构建成功,预渲染 34 个路由。 + +## 四模型收敛验收(2026-09-22) + +移除旧模型折叠入口,模型卡片改为用途及配置说明,不再使用本机测试数据作为产品文案。新增旧设置迁移、停用模型选择及下载拒绝、四项模型目录一致性回归。 + +独立 PR 工作区后端 152 项通过,另有 14 项子测试通过;原有导出控件契约失败仍可复现。设置及桌面契约 28 项通过,语音组件 63 项通过,Nuxt 静态构建成功。启动实际 Electron 应用检查四项卡片与说明;四个保留模型各完成两条真实语音的应用 HTTP 转写,并逐条验证缓存,旧六项选择接口均返回 invalid_model。 diff --git a/frontend/components/SettingsDialog.vue b/frontend/components/SettingsDialog.vue index 023e5f2a..0240f11d 100644 --- a/frontend/components/SettingsDialog.vue +++ b/frontend/components/SettingsDialog.vue @@ -313,7 +313,7 @@ 语音识别模型 - 低配选 Zipformer;CPU 质量优先选 Qwen INT4;GPU 极速选 Turbo,质量优先选 Qwen。下载后选择,语音在本机处理。 + 低配电脑选 CTC;无独显选 Qwen CPU;有 NVIDIA 显卡可选 Turbo 或 Qwen GPU。下载并选择模型后,语音在本机处理。 当前:{{ voiceModelText }} @@ -324,7 +324,6 @@ {{ model.name }} 推荐 - 旧版兼容 已选择 {{ model.size }} · {{ model.speed }} · {{ model.quality }} @@ -406,13 +404,8 @@