1#!/usr/bin/env python3
2"""Speaker diarization (pyannote) — prints turn boundaries as JSON.
3
4Runs in the dedicated ~/.clover-diarize venv (pinned deps). session_transcript
5calls this to get precise "who spoke when" boundaries; the main venv then names
6the speakers (ECAPA) and splits the forced-aligned words at the turn edges.
7
8 python diarize.py <audio> -> [{"start":..,"end":..,"speaker":"SPEAKER_00"}, ...]
9"""
10import json
11import subprocess
12import sys
13import warnings
14
15warnings.filterwarnings("ignore")
16
17import numpy as np
18import torch
19from pyannote.audio import Pipeline
20
21audio = sys.argv[1]
22pipe = Pipeline.from_pretrained("pyannote/speaker-diarization-3.1")
23if torch.backends.mps.is_available():
24 pipe.to(torch.device("mps"))
25
26raw = subprocess.run(
27 ["ffmpeg", "-nostdin", "-i", audio, "-f", "f32le", "-ac", "1", "-ar", "16000", "-"],
28 capture_output=True,
29).stdout
30wav = torch.from_numpy(np.frombuffer(raw, np.float32).copy()).unsqueeze(0)
31
32dia = pipe({"waveform": wav, "sample_rate": 16000})
33turns = [
34 {"start": float(t.start), "end": float(t.end), "speaker": spk}
35 for t, _, spk in dia.itertracks(yield_label=True)
36]
37print(json.dumps(turns))