| 1 | """Speaker embeddings via SpeechBrain ECAPA-TDNN (no HF token needed). |
| 2 | |
| 3 | Used to tell the user's voice from others in a session: enroll a voiceprint |
| 4 | once, then score each transcript segment against it by cosine similarity. |
| 5 | """ |
| 6 | import json |
| 7 | import os |
| 8 | import subprocess |
| 9 | import warnings |
| 10 | |
| 11 | warnings.filterwarnings("ignore") |
| 12 | |
| 13 | import numpy as np |
| 14 | import torch |
| 15 | import torchaudio |
| 16 | |
| 17 | LIBRARY = os.path.expanduser("~/.clover-whisper/voices.json") |
| 18 | # Cosine above MATCH_THRESHOLD => a cluster is that library voice. CLUSTER_DIST |
| 19 | # is the cosine *distance* below which segments merge into one speaker (so your |
| 20 | # own voice stays a single cluster instead of fragmenting). Tunable. |
| 21 | MATCH_THRESHOLD = 0.45 |
| 22 | CLUSTER_DIST = 0.55 |
| 23 | # Hard cap on distinct voices the fallback clustering may invent. Without it, |
| 24 | # short/noisy segments and (in improv) character voices fragment one person into |
| 25 | # dozens of "speakers". Real sessions here have a handful of people at most. |
| 26 | MAX_SPEAKERS = 6 |
| 27 | |
| 28 | _model = None |
| 29 | |
| 30 | |
| 31 | def load_library(): |
| 32 | if os.path.exists(LIBRARY): |
| 33 | return json.load(open(LIBRARY)).get("voices", []) |
| 34 | return [] |
| 35 | |
| 36 | |
| 37 | def save_library(voices): |
| 38 | os.makedirs(os.path.dirname(LIBRARY), exist_ok=True) |
| 39 | json.dump({"voices": voices}, open(LIBRARY, "w")) |
| 40 | |
| 41 | |
| 42 | def upsert_voice(name, centroid, count=1): |
| 43 | """Add a named voice, or blend into an existing one (running average).""" |
| 44 | centroid = np.asarray(centroid, dtype=np.float32) |
| 45 | voices = load_library() |
| 46 | for v in voices: |
| 47 | if v["name"] == name: |
| 48 | c, n = np.asarray(v["centroid"], dtype=np.float32), v.get("count", 1) |
| 49 | blended = (c * n + centroid * count) / (n + count) |
| 50 | blended /= np.linalg.norm(blended) |
| 51 | v["centroid"], v["count"] = blended.tolist(), n + count |
| 52 | save_library(voices) |
| 53 | return |
| 54 | voices.append({"name": name, "centroid": centroid.tolist(), "count": count}) |
| 55 | save_library(voices) |
| 56 | |
| 57 | |
| 58 | def model(): |
| 59 | global _model |
| 60 | if _model is None: |
| 61 | from speechbrain.inference.speaker import EncoderClassifier |
| 62 | |
| 63 | _model = EncoderClassifier.from_hparams( |
| 64 | source="speechbrain/spkrec-ecapa-voxceleb", run_opts={"device": "cpu"} |
| 65 | ) |
| 66 | return _model |
| 67 | |
| 68 | |
| 69 | def _ffmpeg(): |
| 70 | user = os.environ.get("USER", "") |
| 71 | for c in ( |
| 72 | f"/etc/profiles/per-user/{user}/bin/ffmpeg", |
| 73 | "/run/current-system/sw/bin/ffmpeg", |
| 74 | "/opt/homebrew/bin/ffmpeg", |
| 75 | "/usr/local/bin/ffmpeg", |
| 76 | ): |
| 77 | if os.path.exists(c): |
| 78 | return c |
| 79 | return "ffmpeg" |
| 80 | |
| 81 | |
| 82 | def load_audio(path, target_sr=16000): |
| 83 | """Decode any format (m4a/wav/aiff/…) to mono float32 via ffmpeg.""" |
| 84 | out = subprocess.run( |
| 85 | [_ffmpeg(), "-nostdin", "-i", path, "-f", "f32le", "-ac", "1", "-ar", str(target_sr), "-"], |
| 86 | capture_output=True, |
| 87 | ).stdout |
| 88 | return np.frombuffer(out, dtype=np.float32).copy(), target_sr |
| 89 | |
| 90 | |
| 91 | def embed_array(data, sr): |
| 92 | sig = torch.from_numpy(np.ascontiguousarray(data)).unsqueeze(0) |
| 93 | if sr != 16000: |
| 94 | sig = torchaudio.functional.resample(sig, sr, 16000) |
| 95 | e = model().encode_batch(sig).squeeze() |
| 96 | e = e / e.norm() |
| 97 | return e.detach().cpu().numpy() |
| 98 | |
| 99 | |
| 100 | def embed_file(path): |
| 101 | data, sr = load_audio(path) |
| 102 | return embed_array(data, sr) |
| 103 | |
| 104 | |
| 105 | def embed_segment(data, sr, start, end): |
| 106 | a, b = int(start * sr), int(end * sr) |
| 107 | seg = data[a:b] |
| 108 | if len(seg) < int(0.8 * sr): # too short to be reliable |
| 109 | return None |
| 110 | return embed_array(seg, sr) |
| 111 | |
| 112 | |
| 113 | def cosine(a, b): |
| 114 | return float(np.dot(a, b)) |