125 lines
5.0 KiB
Python
125 lines
5.0 KiB
Python
# -*- coding: utf-8 -*-
|
|
"""企业微信语音转文字
|
|
|
|
两条路径 (可叠加):
|
|
|
|
1. 读本地缓存 (零成本, 推荐):
|
|
企微客户端里点过"转文字"的语音消息, 转写文本会缓存在
|
|
解密后 message.db 的 msg_voice2text 表。
|
|
关联键: msg_voice2text.message_id == message_table.server_id
|
|
|
|
2. 本地 ASR 兜底 (离线免费, 需先装依赖):
|
|
无本地缓存的语音 (.silk) 用 pilk 解码为 WAV, 再用 faster-whisper 识别。
|
|
首次运行会下载模型 (~150-460MB, 走 HuggingFace 镜像加速)。
|
|
"""
|
|
|
|
import os
|
|
import sqlite3
|
|
|
|
|
|
def load_voice2text(decrypted_dbs, log=print):
|
|
"""从解密后的 message.db 读取企微本地语音转写缓存
|
|
|
|
decrypted_dbs: [(db_path, db_name, user_dir), ...] 与导出链路一致的列表
|
|
返回: {(user_dir, str(server_id)): text}
|
|
"""
|
|
result = {}
|
|
for db_path, db_name, user_dir in decrypted_dbs:
|
|
if db_name != "message.db":
|
|
continue
|
|
try:
|
|
conn = sqlite3.connect(db_path)
|
|
has = conn.execute(
|
|
"SELECT 1 FROM sqlite_master WHERE type='table' AND name='msg_voice2text'"
|
|
).fetchone()
|
|
if not has:
|
|
conn.close()
|
|
continue
|
|
rows = conn.execute(
|
|
"SELECT message_id, text FROM msg_voice2text "
|
|
"WHERE text IS NOT NULL AND trim(text) != ''"
|
|
).fetchall()
|
|
n = 0
|
|
for message_id, text in rows:
|
|
if message_id is None:
|
|
continue
|
|
result[(user_dir, str(message_id))] = str(text)
|
|
n += 1
|
|
if n:
|
|
log(f"[+] {user_dir}: 读取到 {n} 条语音转写缓存 (msg_voice2text)")
|
|
conn.close()
|
|
except Exception as e:
|
|
log(f" [警告] 读取语音转写缓存失败 {db_path}: {e}")
|
|
return result
|
|
|
|
|
|
def decode_silk_to_wav(silk_path, wav_path, sample_rate=24000):
|
|
"""把 SILK v3 语音解码为 WAV (依赖 pilk)
|
|
|
|
pilk 0.2.x 提供 silk_to_wav() 便捷函数; 旧版本用 decode(..., sample_rate=)
|
|
"""
|
|
try:
|
|
import pilk
|
|
except ImportError:
|
|
raise RuntimeError("缺少 pilk 库, 无法解码语音: pip install pilk")
|
|
try:
|
|
pilk.silk_to_wav(silk_path, wav_path, rate=sample_rate)
|
|
except TypeError:
|
|
# 兼容旧版 pilk API
|
|
pilk.decode(silk_path, wav_path, sample_rate=sample_rate)
|
|
return wav_path
|
|
|
|
|
|
def asr_missing(voice2text, media_map, log=print, model_size="base"):
|
|
"""对 media_map 中无本地缓存的语音文件做本地识别, 结果合并进 voice2text
|
|
|
|
voice2text: {(user_dir, str(server_id)): text} (会被就地合并)
|
|
media_map: {str(server_id): {path, url}} (导出媒体阶段的产物, key 为字符串)
|
|
返回: voice2text (合并后)
|
|
"""
|
|
missing = []
|
|
for sid, info in media_map.items():
|
|
p = (info or {}).get("path", "")
|
|
if p and p.lower().endswith((".silk", ".amr")):
|
|
if not any(k[1] == sid for k in voice2text):
|
|
missing.append((p, sid))
|
|
if not missing:
|
|
return voice2text
|
|
|
|
try:
|
|
import pilk
|
|
from faster_whisper import WhisperModel
|
|
except ImportError:
|
|
log("[-] 未安装 pilk / faster-whisper, 跳过无缓存语音的本地识别")
|
|
log("[-] (pip install pilk faster-whisper 后可用; 本次仅导出已有转写缓存)")
|
|
return voice2text
|
|
|
|
# 禁用 xet 存储 (部分网络环境不稳定, 回退普通 HTTP 下载更可靠)
|
|
# 注意: 新版 huggingface_hub 对 hf-mirror 等镜像域名校验更严, 不强制设置
|
|
# HF_ENDPOINT, 使用官方源直连; 如网络受限可自行设置环境变量 HF_ENDPOINT 指向镜像
|
|
os.environ.setdefault("HF_HUB_DISABLE_XET", "1")
|
|
log(f"[*] 本地识别 {len(missing)} 条无缓存语音...")
|
|
log(f"[*] 加载语音识别模型 ({model_size})... 首次使用需下载模型 (~{model_size} 数百MB)")
|
|
try:
|
|
model = WhisperModel(model_size, device="cpu", compute_type="int8")
|
|
except Exception as e:
|
|
log(f"[-] 模型加载失败, 跳过本地识别: {e}")
|
|
return voice2text
|
|
|
|
import tempfile
|
|
with tempfile.TemporaryDirectory() as tmp:
|
|
for idx, (silk_path, sid) in enumerate(missing, 1):
|
|
try:
|
|
wav_path = os.path.join(tmp, f"v{idx}.wav")
|
|
decode_silk_to_wav(silk_path, wav_path)
|
|
segments, _info = model.transcribe(wav_path, language="zh", vad_filter=True)
|
|
text = "".join(s.text for s in segments).strip()
|
|
if text:
|
|
# 结果按 (账号占位, server_id) 存储, 账号维度在合并处处理
|
|
voice2text[("", sid)] = text
|
|
log(f" [{idx}/{len(missing)}] {os.path.basename(silk_path)} -> "
|
|
f"{text[:50] if text else '(无有效语音)'}")
|
|
except Exception as e:
|
|
log(f" [警告] 识别失败 {os.path.basename(silk_path)}: {e}")
|
|
return voice2text
|