diff --git a/recorder/dictation/asr.py b/recorder/dictation/asr.py new file mode 100644 index 0000000000000000000000000000000000000000..9fd393a424a01a3af7082245f7285936768b1ddd --- /dev/null +++ b/recorder/dictation/asr.py @@ -0,0 +1,51 @@ +#!/usr/bin/env python3 +"""On-device ASR via Qwen3-ASR-1.7B (MLX). Replaces mlx-whisper. + +Why the swap: whisper-large-v3-turbo collapses into infinite word-loops on +multi-hour journals — its condition_on_previous_text snowballs one hallucinated +repeat across hours until the decoder gets stuck ("crazy crazy crazy…" ×220). +Qwen3-ASR is a dedicated ASR head with internal energy-based chunking that +resets context per chunk, so there's nothing to snowball. On a 4.2h journal it +produced zero runaway loops where whisper produced eleven. Runs in the same +~/.clover-whisper/.venv (kept that name so the app's hard-coded paths still work). +""" +from mlx_qwen3_asr import transcribe as _transcribe + +MODEL = "Qwen/Qwen3-ASR-1.7B" + + +def transcribe_text(audio): + """Plain text (marker dictation).""" + return _transcribe(audio, model=MODEL).text.strip() + + +def transcribe_segments(audio): + """Whisper-shaped [{start, end, text}] with coarse timestamps. Qwen emits + very fine segments (often one word); merge_segments() coarsens them so the + per-segment forced-aligner isn't called tens of thousands of times.""" + r = _transcribe(audio, model=MODEL, return_timestamps=True) + segs = [{"start": float(s["start"]), "end": float(s["end"]), "text": s["text"].strip()} + for s in (getattr(r, "segments", None) or []) + if (s.get("text") or "").strip()] + return merge_segments(segs) + + +def merge_segments(segs, max_gap=3.0, max_dur=30.0): + """Glue adjacent segments split by < max_gap of silence, up to max_dur. + + Qwen emits very fine segments (often one word). Two reasons to coarsen: + the per-segment forced-aligner shouldn't run tens of thousands of times, + and render() starts a new paragraph at every gap between segments — so with + raw Qwen segments a gappy gaming monologue shatters into dozens of one-line + paragraphs. Merging across pauses up to max_gap (> render's PAUSE_SPLIT) + makes the surviving boundaries — and thus paragraphs — whisper-sized, while + max_dur still breaks a long unbroken monologue.""" + out = [] + for s in segs: + if (out and s["start"] - out[-1]["end"] <= max_gap + and s["end"] - out[-1]["start"] <= max_dur): + out[-1]["end"] = s["end"] + out[-1]["text"] = (out[-1]["text"] + " " + s["text"]).strip() + else: + out.append(dict(s)) + return out diff --git a/recorder/dictation/backfill_transcripts.py b/recorder/dictation/backfill_transcripts.py new file mode 100644 index 0000000000000000000000000000000000000000..e0c5391dc87a8c96cdde2dca76160ed300521df2 --- /dev/null +++ b/recorder/dictation/backfill_transcripts.py @@ -0,0 +1,138 @@ +#!/usr/bin/env python3 +"""Re-transcribe every archived session with Qwen3-ASR-1.7B, replacing the old +whisper transcripts (which collapsed into word-loops on long recordings). + +Walks the archive for dirs holding both mic.m4a and transcript.md, longest audio +first, and re-runs session_transcript.py in place. The old transcript.md / +transcript.json / speakers.json are preserved once as *.whisper. before the +first overwrite, so nothing is lost. + +Mode (solo vs multi) is inferred from the existing speakers.json: multi if it +recorded more than one speaker or any non-"You" / unknown voice, else solo. The +title is taken from the old transcript's H1, falling back to the folder name. + + python backfill_transcripts.py [--roots DIR ...] [--dry-run] [--limit N] +""" +import argparse +import json +import os +import shutil +import subprocess +import sys +import time + +HERE = os.path.dirname(os.path.abspath(__file__)) +SESSION_SCRIPT = os.path.join(HERE, "session_transcript.py") +PY = sys.executable # this venv's python (has qwen + deps) +# Only the recorder's own output trees — walking all of Archive drags across the +# 2018-2025 media library on the NAS. Journal + Sessions per year is where +# mic.m4a/transcript.md live. +import glob as _glob +DEFAULT_ROOTS = sorted( + _glob.glob("/Volumes/clover/Archive/*/Journal") + + _glob.glob("/Volumes/clover/Archive/*/Sessions")) + + +def audio_duration(path): + try: + out = subprocess.run( + ["ffprobe", "-v", "error", "-show_entries", "format=duration", + "-of", "csv=p=0", path], + capture_output=True, text=True, timeout=60) + return float(out.stdout.strip()) + except Exception: + return 0.0 + + +def find_sessions(roots): + found = [] + for root in roots: + for dirpath, _dirs, files in os.walk(root): + if "mic.m4a" in files and "transcript.md" in files: + found.append(dirpath) + return found + + +# "You" is the solo label; "Clover" is Chloe's own enrolled voice name (self- +# enroll), so it is NOT a guest. Phantom "Speaker N" (unknown=true) is exactly the +# solo-voice fragmentation the cluster fallback produces — not a real person. +SELF_LABELS = {"You", "Clover"} + + +def infer_mode(folder): + """multi ONLY if a genuine, named guest was recorded. Journaling/improv is + solo; the old cluster fallback shattered the solo voice into phantom + "Speaker N", which must not resurrect a multi-speaker run (and, worse, run + pyannote over a multi-hour file).""" + try: + spk = json.load(open(os.path.join(folder, "speakers.json"))).get("speakers", []) + except Exception: + return "solo" + for s in spk: + lab = s.get("label") + if lab and lab not in SELF_LABELS and not lab.startswith("Speaker "): + return "multi" + return "solo" + + +def title_of(folder): + md = os.path.join(folder, "transcript.md") + try: + for line in open(md): + if line.startswith("# "): + return line[2:].strip() + except Exception: + pass + return os.path.basename(folder) + + +def backup_once(folder): + """Preserve the whisper-era outputs the first time we touch this folder.""" + for name in ("transcript.md", "transcript.json", "speakers.json"): + src = os.path.join(folder, name) + base, ext = os.path.splitext(name) + dst = os.path.join(folder, f"{base}.whisper{ext}") + if os.path.exists(src) and not os.path.exists(dst): + shutil.copy2(src, dst) + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--roots", nargs="+", default=DEFAULT_ROOTS) + ap.add_argument("--dry-run", action="store_true") + ap.add_argument("--limit", type=int, default=0) + args = ap.parse_args() + + sessions = find_sessions(args.roots) + ranked = sorted(sessions, key=lambda d: audio_duration(os.path.join(d, "mic.m4a")), + reverse=True) + if args.limit: + ranked = ranked[:args.limit] + + print(f"{len(ranked)} sessions to backfill (longest first)\n") + for i, folder in enumerate(ranked, 1): + mic = os.path.join(folder, "mic.m4a") + dur = audio_duration(mic) / 60 + mode = infer_mode(folder) + title = title_of(folder) + markers = os.path.join(folder, "markers.json") + markers_arg = markers if os.path.exists(markers) else "none" + tag = f"[{i}/{len(ranked)}] {os.path.basename(folder)} {dur:.0f}min {mode}" + if args.dry_run: + print(f"DRY {tag} title={title!r}") + continue + print(f"==> {tag}") + backup_once(folder) + t = time.time() + r = subprocess.run( + [PY, SESSION_SCRIPT, mic, markers_arg, + os.path.join(folder, "transcript.md"), title, mode], + capture_output=True, text=True) + if r.returncode != 0: + print(f" FAILED ({r.returncode}): {r.stderr.strip().splitlines()[-1:]}" ) + else: + print(f" done in {(time.time()-t)/60:.1f} min") + + +if __name__ == "__main__": + main() diff --git a/recorder/dictation/forced_align.py b/recorder/dictation/forced_align.py index 7d62d36a64700e1da0f685234d0a7f29b4e4f026..675a10a9dcd62c22e2c8e1c593a973d35a164b9b 100644 --- a/recorder/dictation/forced_align.py +++ b/recorder/dictation/forced_align.py @@ -34,12 +34,12 @@ def align_segment(data, sr, start, end, text): seg = data[a:b] raw = text.split() if len(seg) < int(0.2 * sr) or not raw: - return [{"word": w, "start": start, "end": end} for w in raw] + return [{"word": w, "start": start, "end": end, "conf": None} for w in raw] norm = [_norm(w) for w in raw] idx = [i for i, n in enumerate(norm) if n] if not idx: - return [{"word": w, "start": start, "end": end} for w in raw] + return [{"word": w, "start": start, "end": end, "conf": None} for w in raw] model, tok, aligner = _load() wav = torch.from_numpy(np.ascontiguousarray(seg)).unsqueeze(0) @@ -48,23 +48,30 @@ def align_segment(data, sr, start, end, text): try: spans = aligner(emit[0], tok([norm[i] for i in idx])) except Exception: - return [{"word": w, "start": start, "end": end} for w in raw] + return [{"word": w, "start": start, "end": end, "conf": None} for w in raw] ratio = wav.shape[1] / emit.shape[1] / sr - times = {} + times, confs = {}, {} for k, i in enumerate(idx): s = spans[k] times[i] = (round(start + s[0].start * ratio, 3), round(start + s[-1].end * ratio, 3)) + # per-word acoustic confidence = mean CTC score over its token spans + # (0..1). A word that doesn't match the audio scores low → flag it. + sc = [float(getattr(tsp, "score", 0.0)) for tsp in s] + confs[i] = round(sum(sc) / len(sc), 3) if sc else 0.0 - out = [{"word": w, "start": None, "end": None} for w in raw] + out = [{"word": w, "start": None, "end": None, "conf": None} for w in raw] for i, t in times.items(): out[i]["start"], out[i]["end"] = t - # Interpolate unaligned words from neighbours. + out[i]["conf"] = confs[i] + # Interpolate unaligned words from neighbours; conf 0.0 marks them as guessed + # (pure numbers/symbols the aligner can't score). last_end = start for i, o in enumerate(out): if o["start"] is None: o["start"] = last_end nxt = next((out[j]["start"] for j in range(i + 1, len(out)) if out[j]["start"]), end) o["end"] = nxt + o["conf"] = 0.0 last_end = o["end"] return out diff --git a/recorder/dictation/merge_sentences.py b/recorder/dictation/merge_sentences.py new file mode 100644 index 0000000000000000000000000000000000000000..f978a611e56fc590aad8784cfd9056b851f8afe6 --- /dev/null +++ b/recorder/dictation/merge_sentences.py @@ -0,0 +1,39 @@ +#!/usr/bin/env python3 +"""Merge adjacent cleaned segments that are obviously ONE sentence split by a +gameplay pause (seg[i] doesn't end in terminal punctuation and seg[i+1] starts +lowercase). The merged segment spans both time windows; re-alignment on the +merged text then re-derives word timings across the pause, so nothing about the +sequencer's per-word data breaks. + + python merge_sentences.py +""" +import json +import sys + +TERMINAL = ".!?…\"')" + + +def should_merge(a, b): + at = a["text"].rstrip() + bt = b["text"].lstrip() + if not at or not bt: + return False + # first segment left hanging (no sentence end) and the next continues lowercase + return at[-1] not in TERMINAL and bt[:1].islower() + + +def main(): + segs = json.load(open(sys.argv[1]))["segments"] + out = [] + for s in segs: + if out and should_merge(out[-1], s): + out[-1]["end"] = s["end"] + out[-1]["text"] = (out[-1]["text"].rstrip() + " " + s["text"].lstrip()).strip() + else: + out.append(dict(s)) + json.dump({"segments": out}, open(sys.argv[2], "w")) + print(f"{len(segs)} -> {len(out)} segments ({len(segs) - len(out)} merged)") + + +if __name__ == "__main__": + main() diff --git a/recorder/dictation/realign.py b/recorder/dictation/realign.py new file mode 100644 index 0000000000000000000000000000000000000000..c78c001657a37e11163de489b5e7fb73b2c56881 --- /dev/null +++ b/recorder/dictation/realign.py @@ -0,0 +1,54 @@ +#!/usr/bin/env python3 +"""Re-run word-level forced alignment on cleaned segment text and write a +sequencer-shaped transcript.json ({segments:[{start,end,text,label,words}]}). + +Used to graft cleaned text (punctuation + corrected proper nouns) back onto real +per-word timings: alignment is re-derived from the CLEANED words against the +audio, so word start/end stay accurate and the sequencer's karaoke captions keep +working. Runs in ~/.clover-whisper/.venv (needs torchaudio). + + python realign.py