这是一份音频检索代码,根据一段查询语音,在本地音频库中快速找出最相似的录音文件。
```python
-*- coding: utf-8 -*-
"""
voice_match.py ------ 根据一段语音,在本地音频库中检索最相似的音频
原理(与说话内容无关的通用声学匹配):
-
每段音频切成 2s 滑窗,每个窗口提取定长声学向量(MFCC + Δ + ΔΔ,CMN 归一化,分段池化)
-
所有窗口向量 L2 归一化后堆成矩阵,即"声学指纹库"
-
查询语音同样切窗提特征,与库矩阵做余弦相似度
-
按文件聚合(每个查询窗口取最佳命中,再取 top-n 平均),返回 Top-K
用法:
python voice_match.py build ./audios -i ./index
python voice_match.py search ./query.wav -i ./index -k 5
"""
from future import annotations
import argparse
import json
from pathlib import Path
import numpy as np
import librosa
SR = 16000
N_MFCC = 20
N_PARTS = 4 # 时间轴分几段池化
AUDIO_EXTS = {".wav", ".mp3", ".flac", ".m4a", ".ogg", ".aac", ".opus", ".wma"}
特征维度 = N_MFCC * 3(原+Δ+ΔΔ) * N_PARTS * 2(mean+std) = 480
FEAT_DIM = N_MFCC * 3 * N_PARTS * 2
--------------------------------------------------------------------------
音频加载 / 切窗
--------------------------------------------------------------------------
def load_audio(path: str | Path, sr: int = SR) -> np.ndarray:
"""读音频 -> 单声道 -> 重采样 -> 去静音 -> 峰值归一化"""
y, _ = librosa.load(str(path), sr=sr, mono=True)
if y.size == 0:
return y.astype(np.float32)
y, _ = librosa.effects.trim(y, top_db=30)
peak = float(np.abs(y).max())
if peak > 0:
y = y / peak
return y.astype(np.float32)
def make_windows(y: np.ndarray, win: float = 2.0, hop: float = 1.0,
sr: int = SR) -> listnp.ndarray:
"""滑窗切分,保证短音频也能返回至少一个窗口"""
w, h = int(win * sr), int(hop * sr)
if y.size == 0:
return \[\]
if y.size <= w:
return y
out = y\[s:s + w for s in range(0, y.size - w + 1, h)]
if (y.size - w) % h: # 补上尾部残留
out.append(y-w:)
return out
--------------------------------------------------------------------------
特征提取(可替换为声纹模型,见文末说明)
--------------------------------------------------------------------------
def embed(y: np.ndarray, sr: int = SR) -> np.ndarray | None:
"""一段语音(建议 1~3s) -> L2 归一化的 480 维向量"""
if y.size < sr * 0.25: # 太短,丢弃
return None
mfcc = librosa.feature.mfcc(
y=y, sr=sr, n_mfcc=N_MFCC, n_fft=400,
hop_length=160, n_mels=40, fmin=20, fmax=7600,
)
倒谱均值归一化(CMN):抑制麦克风/信道带来的整体偏移
mfcc = mfcc - mfcc.mean(axis=1, keepdims=True)
d1 = librosa.feature.delta(mfcc)
d2 = librosa.feature.delta(mfcc, order=2)
feat = np.vstack(mfcc, d1, d2) # (60, T)
时间轴均分 N_PARTS 段,每段取 mean + std,保留粗时序信息
parts = np.array_split(feat, N_PARTS, axis=1)
pooled = np.concatenate(
np.concatenate(\[p.mean(axis=1), p.std(axis=1)\]) for p in parts
)
norm = float(np.linalg.norm(pooled))
if norm <= 1e-9:
return None
return (pooled / norm).astype(np.float32)
--------------------------------------------------------------------------
索引
--------------------------------------------------------------------------
class VoiceIndex:
def init(self, index_dir: str | Path):
self.dir = Path(index_dir)
self.dir.mkdir(parents=True, exist_ok=True)
self.vec_path = self.dir / "vectors.npy"
self.meta_path = self.dir / "meta.json"
self.vectors: np.ndarray | None = None
self.meta: listdict = \[\]
def save(self) -> None:
np.save(self.vec_path, self.vectors)
self.meta_path.write_text(
json.dumps(self.meta, ensure_ascii=False), encoding="utf-8"
)
def load(self) -> "VoiceIndex":
if not self.vec_path.exists() or not self.meta_path.exists():
raise FileNotFoundError(f"索引不存在:{self.dir},请先执行 build")
self.vectors = np.load(self.vec_path)
self.meta = json.loads(self.meta_path.read_text(encoding="utf-8"))
return self
--------------------------------------------------------------------------
建库
--------------------------------------------------------------------------
def build(audio_dir: str | Path, index_dir: str | Path,
win: float = 2.0, hop: float = 1.0) -> VoiceIndex:
audio_dir = Path(audio_dir)
files = sorted(
p for p in audio_dir.rglob("*") if p.suffix.lower() in AUDIO_EXTS
)
if not files:
raise SystemExit(f"在 {audio_dir} 下没找到音频文件")
vecs: listnp.ndarray = \[\]
meta: listdict = \[\]
for i, f in enumerate(files, 1):
try:
y = load_audio(f)
except Exception as e: # 损坏/不支持的格式
print(f"skip {f.name}: {e}")
continue
kept = 0
for j, seg in enumerate(make_windows(y, win, hop)):
v = embed(seg)
if v is None:
continue
vecs.append(v)
meta.append({"file": str(f), "win": j})
kept += 1
print(f"{i}/{len(files)} {f.name} 窗口={kept}")
idx = VoiceIndex(index_dir)
idx.vectors = (np.vstack(vecs).astype(np.float32)
if vecs else np.zeros((0, FEAT_DIM), np.float32))
idx.meta = meta
idx.save()
print(f"\n完成:{len(files)} 个文件,{len(meta)} 个窗口向量 -> {index_dir}")
return idx
--------------------------------------------------------------------------
检索
--------------------------------------------------------------------------
def search(query: str | Path, index_dir: str | Path,
top_k: int = 5, win: float = 2.0, hop: float = 1.0) -> listtuple:
idx = VoiceIndex(index_dir).load()
if idx.vectors is None or idx.vectors.shape0 == 0:
raise RuntimeError("索引为空,请先 build")
y = load_audio(query)
q_vecs = v for v in (embed(s) for s in make_windows(y, win, hop)) if v is not None
if not q_vecs:
raise RuntimeError("查询语音太短或全是静音")
Q = np.vstack(q_vecs) # (nq, D)
S = Q @ idx.vectors.T # (nq, nv) 余弦相似度,越大越像
files = np.array(m\["file" for m in idx.meta])
results = \[\]
for f in np.unique(files):
sub = S:, files == f # (nq, nf)
best = np.sort(sub.max(axis=1))::-1 # 每个查询窗口的最佳命中
n = min(3, best.size)
score = float(best:n.mean()) # top-3 平均,抗单点噪声
hits = int((sub > 0.8).sum()) # 强命中窗口数(参考)
results.append((f, score, hits))
results.sort(key=lambda x: -x1)
return results:top_k
--------------------------------------------------------------------------
CLI
--------------------------------------------------------------------------
def main() -> None:
ap = argparse.ArgumentParser(description="基于语音的音频库检索")
sub = ap.add_subparsers(dest="cmd", required=True)
b = sub.add_parser("build", help="扫描目录建索引")
b.add_argument("audio_dir")
b.add_argument("-i", "--index", default="voice_index")
b.add_argument("--win", type=float, default=2.0, help="窗口秒数")
b.add_argument("--hop", type=float, default=1.0, help="滑动步长秒数")
s = sub.add_parser("search", help="用一段语音检索")
s.add_argument("query")
s.add_argument("-i", "--index", default="voice_index")
s.add_argument("-k", type=int, default=5)
s.add_argument("--win", type=float, default=2.0)
s.add_argument("--hop", type=float, default=1.0)
a = ap.parse_args()
if a.cmd == "build":
build(a.audio_dir, a.index, a.win, a.hop)
else:
rows = search(a.query, a.index, a.k, a.win, a.hop)
print(f"\n{'相似度':>8} {'强命中':>6} 文件")
print("-" * 60)
for f, sc, hits in rows:
print(f"{sc:8.4f} {hits:6d} {f}")
if name == "main":
main()
```
语音匹配检索的完整流程拆解
从建立声学指纹到检索排序,整个流程围绕几个关键步骤展开:
音频加载与预处理:统一转为单声道、重采样到16kHz,自动去除首尾静音,并对音量做归一化,让后续特征更稳定。
滑窗切分与特征提取:默认每2秒切一个窗口,用1秒步长滑动。每个窗口提取MFCC及其一阶、二阶差分,再做倒谱均值归一化和分段池化,最终压缩成480维的L2归一化向量。
索引构建与保存:把所有音频的窗口向量堆成矩阵,连同文件路径信息一起保存到索引目录。
查询检索与排序:对查询语音做同样的特征提取,计算它与库中所有窗口向量的余弦相似度,再按文件聚合(每个查询窗口取最佳命中,再取top-3平均),返回相似度最高的结果。
优化建议: 当前默认按说话人声学特征匹配,若您更关注"同一段录音"的精确查找,可以替换为音频指纹方案(如pyacoustid)并调整相似度计算方式。
仅供参考学习用。