| 1 | #!/usr/bin/env python3 |
| 2 | """Speaker diarization (pyannote) — prints turn boundaries as JSON. |
| 3 | |
| 4 | Runs in the dedicated ~/.clover-diarize venv (pinned deps). session_transcript |
| 5 | calls this to get precise "who spoke when" boundaries; the main venv then names |
| 6 | the speakers (ECAPA) and splits the forced-aligned words at the turn edges. |
| 7 | |
| 8 | python diarize.py <audio> -> [{"start":..,"end":..,"speaker":"SPEAKER_00"}, ...] |
| 9 | """ |
| 10 | import json |
| 11 | import subprocess |
| 12 | import sys |
| 13 | import warnings |
| 14 | |
| 15 | warnings.filterwarnings("ignore") |
| 16 | |
| 17 | import numpy as np |
| 18 | import torch |
| 19 | from pyannote.audio import Pipeline |
| 20 | |
| 21 | audio = sys.argv[1] |
| 22 | pipe = Pipeline.from_pretrained("pyannote/speaker-diarization-3.1") |
| 23 | if torch.backends.mps.is_available(): |
| 24 | pipe.to(torch.device("mps")) |
| 25 | |
| 26 | raw = subprocess.run( |
| 27 | ["ffmpeg", "-nostdin", "-i", audio, "-f", "f32le", "-ac", "1", "-ar", "16000", "-"], |
| 28 | capture_output=True, |
| 29 | ).stdout |
| 30 | wav = torch.from_numpy(np.frombuffer(raw, np.float32).copy()).unsqueeze(0) |
| 31 | |
| 32 | dia = pipe({"waveform": wav, "sample_rate": 16000}) |
| 33 | turns = [ |
| 34 | {"start": float(t.start), "end": float(t.end), "speaker": spk} |
| 35 | for t, _, spk in dia.itertracks(yield_label=True) |
| 36 | ] |
| 37 | print(json.dumps(turns)) |