1#!/usr/bin/env bash
2# One-time: set up local Whisper for the marker overlay's voice dictation.
3#
4# Creates a venv at ~/.clover-whisper with MLX Whisper (Apple-Silicon optimized)
5# and pre-downloads large-v3-turbo (~1.5 GB). Transcription then runs fully
6# on-device in ~1.5 s per note on this machine.
7#
8# bash ~/devel/creative-toolkit/recorder/setup-dictation.sh
9set -euo pipefail
10
11DIR="$HOME/.clover-whisper"
12echo "==> creating venv at $DIR"
13mkdir -p "$DIR"
14cd "$DIR"
15uv venv
16# mlx-whisper: transcription · speechbrain/torchaudio/scipy: speaker ID + clustering
17uv pip install mlx-whisper speechbrain torchaudio scipy
18
19echo "==> pre-downloading whisper-large-v3-turbo (one-time)"
20say -o /tmp/clover-dictation-warm.aiff "Clover dictation is ready." 2>/dev/null || true
21"$DIR/.venv/bin/python" - /tmp/clover-dictation-warm.aiff <<'PY' || true
22import sys, mlx_whisper
23mlx_whisper.transcribe(sys.argv[1], path_or_hf_repo="mlx-community/whisper-large-v3-turbo")
24PY
25
26echo "✅ dictation ready ($DIR/.venv/bin/python)"