| 1 | """Render a Markdown transcript from segments + speaker labels + markers. |
| 2 | |
| 3 | Shared by session_transcript.py (first pass) and relabel.py (after naming). |
| 4 | """ |
| 5 | PAUSE_SPLIT = 2.0 # seconds of silence that starts a new paragraph |
| 6 | |
| 7 | |
| 8 | def fmt(seconds): |
| 9 | s = int(seconds) |
| 10 | h, m, sec = s // 3600, (s % 3600) // 60, s % 60 |
| 11 | return f"{h}:{m:02d}:{sec:02d}" if h else f"{m:02d}:{sec:02d}" |
| 12 | |
| 13 | |
| 14 | def render(title, segments, labels, markers, out_path): |
| 15 | lines = [f"# {title}", ""] |
| 16 | para, p_start, p_spk, prev_end, mi = [], 0.0, None, 0.0, 0 |
| 17 | # Only show speaker names when there's actually more than one speaker. |
| 18 | show_speakers = len({l for l in labels if l}) > 1 |
| 19 | |
| 20 | def flush(): |
| 21 | if para: |
| 22 | who = f"{p_spk} · " if (show_speakers and p_spk) else "" |
| 23 | lines.append(f"**{who}[{fmt(p_start)}]** " + " ".join(para).strip()) |
| 24 | lines.append("") |
| 25 | para.clear() |
| 26 | |
| 27 | def emit_marker(t, text): |
| 28 | flush() |
| 29 | lines.append(f"**{text.strip()}**" if text and text.strip() else f"**◆ [{fmt(t)}]**") |
| 30 | lines.append("") |
| 31 | |
| 32 | for seg, spk in zip(segments, labels): |
| 33 | start = float(seg["start"]) |
| 34 | while mi < len(markers) and markers[mi][0] <= start: |
| 35 | emit_marker(*markers[mi]) |
| 36 | mi += 1 |
| 37 | if para and (start - prev_end > PAUSE_SPLIT or spk != p_spk): |
| 38 | flush() |
| 39 | if not para: |
| 40 | p_start, p_spk = start, spk |
| 41 | para.append(seg["text"].strip()) |
| 42 | prev_end = float(seg["end"]) |
| 43 | flush() |
| 44 | while mi < len(markers): |
| 45 | emit_marker(*markers[mi]) |
| 46 | mi += 1 |
| 47 | open(out_path, "w").write("\n".join(lines) + "\n") |