1#!/usr/bin/env bash
2# Optional "precise mode": pyannote speaker diarization for exact turn
3# boundaries (so mid-sentence interjections are attributed correctly).
4#
5# It lives in its OWN venv because pyannote 3.x needs older torch/torchaudio/
6# huggingface_hub than the main transcription venv — isolating it avoids
7# breaking Whisper/forced-alignment. session_transcript.py auto-detects this
8# venv and uses it; without it, it falls back to cluster-then-match labeling.
9#
10# Prerequisites (one-time, free):
11# 1. A HuggingFace token — log in once so it's cached:
12# ~/.clover-diarize/.venv/bin/huggingface-cli login (or set HF_TOKEN)
13# 2. Accept the model terms (click "Agree") at:
14# https://huggingface.co/pyannote/speaker-diarization-3.1
15# https://huggingface.co/pyannote/segmentation-3.0
16#
17# bash ~/devel/creative-toolkit/recorder/setup-diarization.sh
18set -euo pipefail
19
20DIR="$HOME/.clover-diarize"
21echo "==> creating diarization venv at $DIR"
22mkdir -p "$DIR"
23cd "$DIR"
24uv venv
25# Pinned, mutually-compatible set (torch 2.4 keeps numpy 2; torchaudio 2.4 still
26# exposes AudioMetaData; hf_hub 0.25 still has use_auth_token).
27uv pip install \
28 "pyannote.audio==3.3.2" "torch==2.4.1" "torchaudio==2.4.1" \
29 "huggingface_hub==0.25.2" matplotlib
30
31echo "✅ precise mode ready (ensure the token is logged in and model terms accepted)"