1#!/usr/bin/env python3
2"""Apply speaker names to a session: add newly-named voices to the library and
3re-render transcript.md — no re-transcription.
4
5 python relabel.py <session_dir> <mapping.json>
6
7mapping.json maps the auto-labels to names, e.g. {"Speaker 2": "Maya"}.
8"""
9import json
10import os
11import sys
12
13sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
14import speaker_id as sid
15from transcript_render import render
16
17session_dir, mapping_path = sys.argv[1], sys.argv[2]
18mapping = json.load(open(mapping_path))
19
20transcript = json.load(open(os.path.join(session_dir, "transcript.json")))
21spk_path = os.path.join(session_dir, "speakers.json")
22speakers = json.load(open(spk_path)) if os.path.exists(spk_path) else {"speakers": []}
23
24# Teach the library each newly-named unknown voice.
25for sp in speakers.get("speakers", []):
26 new = mapping.get(sp["label"])
27 if new and new != sp["label"] and sp.get("unknown") and sp.get("centroid"):
28 sid.upsert_voice(new, sp["centroid"])
29
30# Re-map labels and re-render.
31markers = []
32mp = os.path.join(session_dir, "markers.json")
33if os.path.exists(mp):
34 for mk in json.load(open(mp)).get("markers", []):
35 markers.append((float(mk.get("offsetSeconds", 0)), mk.get("text")))
36markers.sort(key=lambda x: x[0])
37
38segments = transcript["segments"]
39for s in segments:
40 s["label"] = mapping.get(s.get("label"), s.get("label"))
41render(transcript.get("title", "Session"), segments,
42 [s.get("label") for s in segments], markers,
43 os.path.join(session_dir, "transcript.md"))
44
45json.dump(transcript, open(os.path.join(session_dir, "transcript.json"), "w"))
46for sp in speakers.get("speakers", []):
47 sp["label"] = mapping.get(sp["label"], sp["label"])
48json.dump(speakers, open(spk_path, "w"))
49print("relabeled")