Files
kefu/2026-08-19-18-27-34/wxwork_voice2text.py
2026-08-27 14:04:28 +08:00

125 lines
5.0 KiB
Python

# -*- coding: utf-8 -*-
"""企业微信语音转文字
两条路径 (可叠加):
1. 读本地缓存 (零成本, 推荐):
企微客户端里点过"转文字"的语音消息, 转写文本会缓存在
解密后 message.db 的 msg_voice2text 表。
关联键: msg_voice2text.message_id == message_table.server_id
2. 本地 ASR 兜底 (离线免费, 需先装依赖):
无本地缓存的语音 (.silk) 用 pilk 解码为 WAV, 再用 faster-whisper 识别。
首次运行会下载模型 (~150-460MB, 走 HuggingFace 镜像加速)。
"""
import os
import sqlite3
def load_voice2text(decrypted_dbs, log=print):
"""从解密后的 message.db 读取企微本地语音转写缓存
decrypted_dbs: [(db_path, db_name, user_dir), ...] 与导出链路一致的列表
返回: {(user_dir, str(server_id)): text}
"""
result = {}
for db_path, db_name, user_dir in decrypted_dbs:
if db_name != "message.db":
continue
try:
conn = sqlite3.connect(db_path)
has = conn.execute(
"SELECT 1 FROM sqlite_master WHERE type='table' AND name='msg_voice2text'"
).fetchone()
if not has:
conn.close()
continue
rows = conn.execute(
"SELECT message_id, text FROM msg_voice2text "
"WHERE text IS NOT NULL AND trim(text) != ''"
).fetchall()
n = 0
for message_id, text in rows:
if message_id is None:
continue
result[(user_dir, str(message_id))] = str(text)
n += 1
if n:
log(f"[+] {user_dir}: 读取到 {n} 条语音转写缓存 (msg_voice2text)")
conn.close()
except Exception as e:
log(f" [警告] 读取语音转写缓存失败 {db_path}: {e}")
return result
def decode_silk_to_wav(silk_path, wav_path, sample_rate=24000):
"""把 SILK v3 语音解码为 WAV (依赖 pilk)
pilk 0.2.x 提供 silk_to_wav() 便捷函数; 旧版本用 decode(..., sample_rate=)
"""
try:
import pilk
except ImportError:
raise RuntimeError("缺少 pilk 库, 无法解码语音: pip install pilk")
try:
pilk.silk_to_wav(silk_path, wav_path, rate=sample_rate)
except TypeError:
# 兼容旧版 pilk API
pilk.decode(silk_path, wav_path, sample_rate=sample_rate)
return wav_path
def asr_missing(voice2text, media_map, log=print, model_size="base"):
"""对 media_map 中无本地缓存的语音文件做本地识别, 结果合并进 voice2text
voice2text: {(user_dir, str(server_id)): text} (会被就地合并)
media_map: {str(server_id): {path, url}} (导出媒体阶段的产物, key 为字符串)
返回: voice2text (合并后)
"""
missing = []
for sid, info in media_map.items():
p = (info or {}).get("path", "")
if p and p.lower().endswith((".silk", ".amr")):
if not any(k[1] == sid for k in voice2text):
missing.append((p, sid))
if not missing:
return voice2text
try:
import pilk
from faster_whisper import WhisperModel
except ImportError:
log("[-] 未安装 pilk / faster-whisper, 跳过无缓存语音的本地识别")
log("[-] (pip install pilk faster-whisper 后可用; 本次仅导出已有转写缓存)")
return voice2text
# 禁用 xet 存储 (部分网络环境不稳定, 回退普通 HTTP 下载更可靠)
# 注意: 新版 huggingface_hub 对 hf-mirror 等镜像域名校验更严, 不强制设置
# HF_ENDPOINT, 使用官方源直连; 如网络受限可自行设置环境变量 HF_ENDPOINT 指向镜像
os.environ.setdefault("HF_HUB_DISABLE_XET", "1")
log(f"[*] 本地识别 {len(missing)} 条无缓存语音...")
log(f"[*] 加载语音识别模型 ({model_size})... 首次使用需下载模型 (~{model_size} 数百MB)")
try:
model = WhisperModel(model_size, device="cpu", compute_type="int8")
except Exception as e:
log(f"[-] 模型加载失败, 跳过本地识别: {e}")
return voice2text
import tempfile
with tempfile.TemporaryDirectory() as tmp:
for idx, (silk_path, sid) in enumerate(missing, 1):
try:
wav_path = os.path.join(tmp, f"v{idx}.wav")
decode_silk_to_wav(silk_path, wav_path)
segments, _info = model.transcribe(wav_path, language="zh", vad_filter=True)
text = "".join(s.text for s in segments).strip()
if text:
# 结果按 (账号占位, server_id) 存储, 账号维度在合并处处理
voice2text[("", sid)] = text
log(f" [{idx}/{len(missing)}] {os.path.basename(silk_path)} -> "
f"{text[:50] if text else '(无有效语音)'}")
except Exception as e:
log(f" [警告] 识别失败 {os.path.basename(silk_path)}: {e}")
return voice2text