1"""Speaker embeddings via SpeechBrain ECAPA-TDNN (no HF token needed).
2
3Used to tell the user's voice from others in a session: enroll a voiceprint
4once, then score each transcript segment against it by cosine similarity.
5"""
6import json
7import os
8import subprocess
9import warnings
10
11warnings.filterwarnings("ignore")
12
13import numpy as np
14import torch
15import torchaudio
16
17LIBRARY = os.path.expanduser("~/.clover-whisper/voices.json")
18# Cosine above MATCH_THRESHOLD => a cluster is that library voice. CLUSTER_DIST
19# is the cosine *distance* below which segments merge into one speaker (so your
20# own voice stays a single cluster instead of fragmenting). Tunable.
21MATCH_THRESHOLD = 0.45
22CLUSTER_DIST = 0.55
23# Hard cap on distinct voices the fallback clustering may invent. Without it,
24# short/noisy segments and (in improv) character voices fragment one person into
25# dozens of "speakers". Real sessions here have a handful of people at most.
26MAX_SPEAKERS = 6
27
28_model = None
29
30
31def load_library():
32 if os.path.exists(LIBRARY):
33 return json.load(open(LIBRARY)).get("voices", [])
34 return []
35
36
37def save_library(voices):
38 os.makedirs(os.path.dirname(LIBRARY), exist_ok=True)
39 json.dump({"voices": voices}, open(LIBRARY, "w"))
40
41
42def upsert_voice(name, centroid, count=1):
43 """Add a named voice, or blend into an existing one (running average)."""
44 centroid = np.asarray(centroid, dtype=np.float32)
45 voices = load_library()
46 for v in voices:
47 if v["name"] == name:
48 c, n = np.asarray(v["centroid"], dtype=np.float32), v.get("count", 1)
49 blended = (c * n + centroid * count) / (n + count)
50 blended /= np.linalg.norm(blended)
51 v["centroid"], v["count"] = blended.tolist(), n + count
52 save_library(voices)
53 return
54 voices.append({"name": name, "centroid": centroid.tolist(), "count": count})
55 save_library(voices)
56
57
58def model():
59 global _model
60 if _model is None:
61 from speechbrain.inference.speaker import EncoderClassifier
62
63 _model = EncoderClassifier.from_hparams(
64 source="speechbrain/spkrec-ecapa-voxceleb", run_opts={"device": "cpu"}
65 )
66 return _model
67
68
69def _ffmpeg():
70 user = os.environ.get("USER", "")
71 for c in (
72 f"/etc/profiles/per-user/{user}/bin/ffmpeg",
73 "/run/current-system/sw/bin/ffmpeg",
74 "/opt/homebrew/bin/ffmpeg",
75 "/usr/local/bin/ffmpeg",
76 ):
77 if os.path.exists(c):
78 return c
79 return "ffmpeg"
80
81
82def load_audio(path, target_sr=16000):
83 """Decode any format (m4a/wav/aiff/…) to mono float32 via ffmpeg."""
84 out = subprocess.run(
85 [_ffmpeg(), "-nostdin", "-i", path, "-f", "f32le", "-ac", "1", "-ar", str(target_sr), "-"],
86 capture_output=True,
87 ).stdout
88 return np.frombuffer(out, dtype=np.float32).copy(), target_sr
89
90
91def embed_array(data, sr):
92 sig = torch.from_numpy(np.ascontiguousarray(data)).unsqueeze(0)
93 if sr != 16000:
94 sig = torchaudio.functional.resample(sig, sr, 16000)
95 e = model().encode_batch(sig).squeeze()
96 e = e / e.norm()
97 return e.detach().cpu().numpy()
98
99
100def embed_file(path):
101 data, sr = load_audio(path)
102 return embed_array(data, sr)
103
104
105def embed_segment(data, sr, start, end):
106 a, b = int(start * sr), int(end * sr)
107 seg = data[a:b]
108 if len(seg) < int(0.8 * sr): # too short to be reliable
109 return None
110 return embed_array(seg, sr)
111
112
113def cosine(a, b):
114 return float(np.dot(a, b))