refactor: QQ → OBv11 重命名 + 平台格式统一抽象

- 所有对外称呼从 QQ 改为 OBv11(注释/提示词/日志/配置项)
- 新增 PlatformFormat 结构体,统一管理平台消息标记格式
- defaultPlatformFormats() 注册表替代硬编码 qqTargetRe
- extractProactiveMessage 改为 Thinker 方法,遍历格式注册表匹配
- 配置项重命名: QQ_BOT_PORT → OBV11_BOT_PORT, QQBotPort → OBv11BotPort
- 标记格式: 【QQ群聊】→【OBv11群聊】、【QQ私聊】→【OBv11私聊】

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
2026-06-22 20:48:07 +08:00
parent 70dbb23234
commit 41f653b672
25 changed files with 1609 additions and 82 deletions
+172
View File
@@ -0,0 +1,172 @@
#!/usr/bin/env python3
"""
用 Group 01 的 33 个文件作为声纹模板,在全量 11K 文件中搜昔涟。
"""
import os, sys, json, warnings
import numpy as np
import librosa
warnings.filterwarnings('ignore')
SEARCH_DIR = r"D:\Project\Code\Uni\Cyrene-Voice-Model\data\cleaned"
OUTPUT_DIR = r"D:\Project\Code\Uni\Cyrene-Voice-Model\data\cyrene_voice"
# Group 01 的文件列表 (从聚类结果获取)
GROUP01_DIR = r"D:\Project\Code\Uni\Cyrene-Voice-Model\data\cleaned"
# 我们需要重建 Group 01 的成员——从之前的聚类结果
# 先手动提取 Group 01 所有文件
# 最简单: 用聚类 center 最近的 N 个文件
print("Step 1: 重建 Group 01 成员...")
import subprocess
# Re-run clustering focused on VoBanks to get exact Group 01 members
VOICEPRINTS = {}
ref_files = []
print(" 提取所有 VoBanks 声纹...")
wav_files = []
for root, dirs, files in os.walk(SEARCH_DIR):
for f in files:
if f.endswith('.wav') and 'VoBanks' in root:
wav_files.append(os.path.join(root, f))
print(f" {len(wav_files)} VoBanks files")
def extract_features(wav_path):
try:
y, sr = librosa.load(wav_path, sr=22050, mono=True)
if len(y) < sr * 0.3: return None
f0, _, _ = librosa.pyin(y, fmin=80, fmax=600, sr=sr)
f0 = f0[~np.isnan(f0)]
if len(f0) < 10: return None
mfcc = librosa.feature.mfcc(y=y, sr=sr, n_mfcc=13)
mfcc_d = librosa.feature.delta(mfcc)
feat = np.concatenate([
[np.mean(f0), np.std(f0),
np.percentile(f0, 25), np.percentile(f0, 50), np.percentile(f0, 75)],
np.mean(mfcc, axis=1), np.std(mfcc, axis=1),
np.mean(mfcc_d, axis=1), np.std(mfcc_d, axis=1),
])
return feat.astype(np.float64)
except:
return None
features = []
valid_files = []
for i, wav in enumerate(wav_files):
feat = extract_features(wav)
if feat is not None:
valid_files.append(wav)
features.append(feat)
if (i+1) % 200 == 0:
print(f" {i+1}/{len(wav_files)}")
X = np.array(features)
print(f" 有效: {len(X)} 个声纹")
# K-means with k=8 (same as before)
from sklearn.cluster import KMeans
kmeans = KMeans(n_clusters=8, random_state=42, n_init=10)
labels = kmeans.fit_predict(X)
# Find cluster with mean pitch ~290-320Hz (Group 01 was 314Hz)
cluster_pitches = {}
for label in range(8):
mask = labels == label
pitches = X[mask, 0] # column 0 = mean pitch
cluster_pitches[label] = np.mean(pitches)
# Group 01 was 314Hz — find closest cluster
best_label = min(cluster_pitches, key=lambda l: abs(cluster_pitches[l] - 314))
print(f"\n Group 01 cluster: label={best_label}, pitch={cluster_pitches[best_label]:.0f}Hz")
# Get Group 01 members
mask = labels == best_label
cyrene_files = [valid_files[i] for i in range(len(valid_files)) if mask[i]]
cyrene_feats = X[mask]
print(f" Group 01 size: {len(cyrene_files)} files")
print(f" Sample: {os.path.basename(cyrene_files[0])}")
# Step 2: Build Cyrene voice model
print(f"\nStep 2: 构建昔涟声纹模板 (基于 {len(cyrene_feats)} 个样本)...")
cyrene_center = np.mean(cyrene_feats, axis=0)
print(f" 模板音高: {cyrene_center[0]:.0f}Hz")
def cosine_sim(a, b):
return np.dot(a, b) / (np.linalg.norm(a) * np.linalg.norm(b) + 1e-8)
# Step 3: Search ALL files
print("\nStep 3: 全量搜索...")
all_wavs = []
for root, dirs, files in os.walk(SEARCH_DIR):
for f in files:
if f.endswith('.wav'):
all_wavs.append(os.path.join(root, f))
print(f" 搜索范围: {len(all_wavs)} 个 WAV 文件")
results = []
for i, wav in enumerate(all_wavs):
feat = extract_features(wav)
if feat is not None:
sim = cosine_sim(cyrene_center, feat)
pitch = feat[0]
results.append((sim, pitch, wav))
if (i+1) % 2000 == 0:
print(f" {i+1}/{len(all_wavs)}")
results.sort(key=lambda x: x[0], reverse=True)
# Step 4: Show results by source
print(f"\n=== 昔涟声纹搜索结果 ===")
print(f"Top 50 文件:")
for rank, (sim, pitch, path) in enumerate(results[:50], 1):
fname = os.path.basename(path)
parent = os.path.basename(os.path.dirname(path))
print(f" {rank:2d}. [{sim:.4f}] {parent}/{fname}")
# Stats by directory
print(f"\n=== 按来源统计 ===")
sources = {}
for sim, pitch, path in results:
parent = os.path.basename(os.path.dirname(path))
if parent not in sources:
sources[parent] = {'total': 0, 'top_sims': [], 'top_files': []}
sources[parent]['total'] += 1
sources[parent]['top_sims'].append(sim)
sources[parent]['top_files'].append((sim, os.path.basename(path)))
for src in sorted(sources.keys()):
s = sources[src]
top5_avg = np.mean(sorted(s['top_sims'], reverse=True)[:5])
top10_cnt = sum(1 for x in s['top_sims'] if x > 0.92)
print(f" {src}: total={s['total']}, top5_avg={top5_avg:.4f}, high_match(>0.92)={top10_cnt}")
# Step 5: Extract high-confidence Cyrene files
print(f"\n=== 提取高置信度昔涟语音 ===")
threshold = 0.92
high_conf = [(s, p, w) for s, p, w in results if s > threshold]
print(f" 阈值 >{threshold}: {len(high_conf)} 个文件")
os.makedirs(OUTPUT_DIR, exist_ok=True)
for sim, pitch, path in high_conf:
fname = os.path.basename(path)
dst = os.path.join(OUTPUT_DIR, fname)
if not os.path.exists(dst):
try:
import shutil
shutil.copy2(path, dst)
except:
pass
print(f" 已复制到: {OUTPUT_DIR}")
print(f" 实际文件数: {len(os.listdir(OUTPUT_DIR))}")
# Save results
with open(os.path.join(OUTPUT_DIR, 'search_results.json'), 'w') as f:
json.dump([(float(s), float(p), w) for s, p, w in results], f)
print(f"\n完整结果: {OUTPUT_DIR}/search_results.json")