From d6189af1de706392cb5ad9a2b978448c02f7e60c Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:17:54 +0400 Subject: [PATCH 001/110] Keep clip segments in sync when a suggestion's range is edited --- src/ui/web-server.ts | 16 +++++++++++-- src/utils/transcript.test.ts | 44 ++++++++++++++++++++++++++++++++++++ src/utils/transcript.ts | 31 +++++++++++++++++++++++++ 3 files changed, 89 insertions(+), 2 deletions(-) create mode 100644 src/utils/transcript.test.ts diff --git a/src/ui/web-server.ts b/src/ui/web-server.ts index 4a107563..cbaca336 100644 --- a/src/ui/web-server.ts +++ b/src/ui/web-server.ts @@ -44,7 +44,13 @@ import { advanceProgress, tagSubmittedClip, tagSubmittedClips } from "../utils/c import { DEMO_ASSETS_DIR } from "./demo-fixtures.js"; import { registerConfigIntegrationRoutes } from "../handlers/integrations.routes.js"; import { childLogger } from "../utils/logger.js"; -import { sliceTranscript, sliceWords, findContentType, findSuggestionSegments } from "../utils/transcript.js"; +import { + sliceTranscript, + sliceWords, + findContentType, + findSuggestionSegments, + reconcileSegmentsForRange, +} from "../utils/transcript.js"; import { errMsg } from "../utils/errors.js"; import { resolveByteRange } from "../utils/http-range.js"; import { @@ -4312,7 +4318,13 @@ app.post("/api/suggestions/modify", (req, res) => { return; } // The energy score was measured over the old range. - if (nextStart !== clip.start_second || nextEnd !== clip.end_second) dropEnergy(clip); + if (nextStart !== clip.start_second || nextEnd !== clip.end_second) { + dropEnergy(clip); + // The old segments array is scoped to the old range; if left stale it + // overrides the new start/end at render time (create_clip reads + // keep_segments ahead of start_second/end_second). + clip.segments = reconcileSegmentsForRange(clip.segments, nextStart, nextEnd); + } if (typeof upd.title === "string") clip.title = upd.title; clip.start_second = nextStart; clip.end_second = nextEnd; diff --git a/src/utils/transcript.test.ts b/src/utils/transcript.test.ts new file mode 100644 index 00000000..1c5ecd4c --- /dev/null +++ b/src/utils/transcript.test.ts @@ -0,0 +1,44 @@ +import { describe, it, expect } from "vitest"; +import { reconcileSegmentsForRange } from "./transcript.js"; + +describe("reconcileSegmentsForRange", () => { + it("passes through when there are no segments", () => { + expect(reconcileSegmentsForRange(undefined, 0, 10)).toBeUndefined(); + expect(reconcileSegmentsForRange([], 0, 10)).toEqual([]); + }); + + it("drops segments entirely outside the new range", () => { + const segments = [ + { start: 0, end: 5 }, + { start: 20, end: 25 }, + ]; + expect(reconcileSegmentsForRange(segments, 20, 25)).toBeUndefined(); + }); + + it("clamps segments straddling a new boundary", () => { + const segments = [ + { start: 0, end: 10 }, + { start: 15, end: 25 }, + ]; + expect(reconcileSegmentsForRange(segments, 5, 20)).toEqual([ + { start: 5, end: 10 }, + { start: 15, end: 20 }, + ]); + }); + + it("clears a single segment that ends up spanning the whole new range", () => { + const segments = [{ start: 0, end: 30 }]; + expect(reconcileSegmentsForRange(segments, 5, 20)).toBeUndefined(); + }); + + it("keeps a multi-segment edit that still carries editorial cuts", () => { + const segments = [ + { start: 2, end: 8 }, + { start: 12, end: 18 }, + ]; + expect(reconcileSegmentsForRange(segments, 0, 20)).toEqual([ + { start: 2, end: 8 }, + { start: 12, end: 18 }, + ]); + }); +}); diff --git a/src/utils/transcript.ts b/src/utils/transcript.ts index 4a10803f..62dd6e9e 100644 --- a/src/utils/transcript.ts +++ b/src/utils/transcript.ts @@ -43,6 +43,37 @@ export function findContentType( return match?.content_type; } +/** + * Keep a clip's keep_segments consistent after its start/end range is edited. + * Without this, a stale segments array (scoped to the old range) overrides + * the new start_second/end_second at render time, since the generator + * derives the actual cut points from keep_segments when present. + * + * Segments outside the new range are dropped, segments straddling a new + * boundary are clamped to it, and a single segment that ends up spanning + * the whole new range is cleared so the generator is free to auto-tighten + * it instead of carrying forward an edit that no longer means anything. + */ +export function reconcileSegmentsForRange( + segments: Array<{ start: number; end: number }> | undefined, + nextStart: number, + nextEnd: number, +): Array<{ start: number; end: number }> | undefined { + if (!segments?.length) return segments; + const clamped = segments + .map((s) => ({ start: Math.max(s.start, nextStart), end: Math.min(s.end, nextEnd) })) + .filter((s) => s.end > s.start); + if (clamped.length === 0) return undefined; + if ( + clamped.length === 1 && + clamped[0].start <= nextStart + 0.01 && + clamped[0].end >= nextEnd - 0.01 + ) { + return undefined; + } + return clamped; +} + export function findSuggestionSegments( suggestions: Array<{ start_second: number; From 4bf3fca51c21cf71df08df8dbebb2303616bdbe9 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:18:44 +0400 Subject: [PATCH 002/110] Fix SRT timestamp rounding so milliseconds never overflow to 1000 --- src/ui/web-server.ts | 12 +++--------- src/utils/srt-time.test.ts | 27 +++++++++++++++++++++++++++ src/utils/srt-time.ts | 23 +++++++++++++++++++++++ 3 files changed, 53 insertions(+), 9 deletions(-) create mode 100644 src/utils/srt-time.test.ts create mode 100644 src/utils/srt-time.ts diff --git a/src/ui/web-server.ts b/src/ui/web-server.ts index cbaca336..9fef0f95 100644 --- a/src/ui/web-server.ts +++ b/src/ui/web-server.ts @@ -51,6 +51,7 @@ import { findSuggestionSegments, reconcileSegmentsForRange, } from "../utils/transcript.js"; +import { formatSrtTime, formatVttTime } from "../utils/srt-time.js"; import { errMsg } from "../utils/errors.js"; import { resolveByteRange } from "../utils/http-range.js"; import { @@ -2155,15 +2156,8 @@ app.get("/api/export-transcript", (_req, res) => { }); } - const fmtSrt = (s: number) => { - const h = Math.floor(s / 3600); - const m = Math.floor((s % 3600) / 60); - const sec = Math.floor(s % 60); - const ms = Math.round((s % 1) * 1000); - return `${String(h).padStart(2, "0")}:${String(m).padStart(2, "0")}:${String(sec).padStart(2, "0")},${String(ms).padStart(3, "0")}`; - }; - - const fmtVtt = (s: number) => fmtSrt(s).replace(",", "."); + const fmtSrt = formatSrtTime; + const fmtVtt = formatVttTime; if (format === "vtt") { let vtt = "WEBVTT\n\n"; diff --git a/src/utils/srt-time.test.ts b/src/utils/srt-time.test.ts new file mode 100644 index 00000000..90995d8f --- /dev/null +++ b/src/utils/srt-time.test.ts @@ -0,0 +1,27 @@ +import { describe, it, expect } from "vitest"; +import { formatSrtTime, formatVttTime } from "./srt-time.js"; + +describe("formatSrtTime", () => { + it("formats a plain value", () => { + expect(formatSrtTime(65.25)).toBe("00:01:05,250"); + }); + + it("carries milliseconds that round up to 1000 into the next second", () => { + // 1.9996 * 1000 = 1999.6 -> rounds to 2000ms, not "01,1000". + expect(formatSrtTime(1.9996)).toBe("00:00:02,000"); + }); + + it("carries a rounded second into the next minute", () => { + expect(formatSrtTime(59.9996)).toBe("00:01:00,000"); + }); + + it("carries through hours", () => { + expect(formatSrtTime(3599.9996)).toBe("01:00:00,000"); + }); +}); + +describe("formatVttTime", () => { + it("uses a period instead of a comma", () => { + expect(formatVttTime(1.9996)).toBe("00:00:02.000"); + }); +}); diff --git a/src/utils/srt-time.ts b/src/utils/srt-time.ts new file mode 100644 index 00000000..b7793b3f --- /dev/null +++ b/src/utils/srt-time.ts @@ -0,0 +1,23 @@ +/** + * Format seconds as an SRT timestamp (HH:MM:SS,mmm). + * + * Rounds to whole milliseconds once, up front, then derives h/m/sec/ms by + * integer division. Rounding the seconds and milliseconds components + * separately lets a value like 1.9996 round seconds down to 1 and ms up to + * 1000, emitting the invalid "01,1000" instead of carrying into "02,000". + */ +export function formatSrtTime(s: number): string { + let totalMs = Math.round(s * 1000); + const h = Math.floor(totalMs / 3_600_000); + totalMs -= h * 3_600_000; + const m = Math.floor(totalMs / 60_000); + totalMs -= m * 60_000; + const sec = Math.floor(totalMs / 1000); + const ms = totalMs - sec * 1000; + return `${String(h).padStart(2, "0")}:${String(m).padStart(2, "0")}:${String(sec).padStart(2, "0")},${String(ms).padStart(3, "0")}`; +} + +/** Format seconds as a WEBVTT timestamp (HH:MM:SS.mmm). */ +export function formatVttTime(s: number): string { + return formatSrtTime(s).replace(",", "."); +} From fe31f12d874215748e2e72dc0c0b9f6a78d22050 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:19:50 +0400 Subject: [PATCH 003/110] Fix NTSC timecode: honor drop-frame and 1001/1000 pulldown rate --- backend/services/multicam.py | 44 ++++++++++++++++++++++++++++++++---- tests/test_multicam.py | 32 ++++++++++++++++++++++++++ 2 files changed, 71 insertions(+), 5 deletions(-) diff --git a/backend/services/multicam.py b/backend/services/multicam.py index 49940338..75b58395 100644 --- a/backend/services/multicam.py +++ b/backend/services/multicam.py @@ -275,16 +275,50 @@ def scan_folder(folder: str) -> list[str]: return found +_TIMECODE_RE = re.compile(r"^(\d{2})([:;])(\d{2})([:;])(\d{2})([:;])(\d{2})$") + + +def _ntsc_frame_duration(fps: float) -> tuple[float, int]: + """Exact frame duration and the nominal (rounded) frame rate for an NTSC-pulldown fps. + + 29.97 is really 30000/1001 fps, so each frame lasts 1001/30000 s, not 1/30 s. The + same pulldown ratio applies to 59.94 and 23.976. Integer rates (25, 24, 30 exactly) + have no pulldown and use a plain 1/fps duration. + """ + rounded = round(fps) + if rounded <= 0: + return 0.0, 0 + if abs(fps - rounded) > 1e-3: + return 1001.0 / (rounded * 1000.0), rounded + return 1.0 / rounded, rounded + + def _timecode_seconds(info: dict, fps: float, sample_rate: int) -> float: - """Embedded start timecode (pro cameras) or BWF time reference (field recorders), in seconds.""" + """Embedded start timecode (pro cameras) or BWF time reference (field recorders), in seconds. + + Honors drop-frame timecode (separator ';' before the frame field): drop-frame + counters skip frame numbers :00 and :01 at the start of every minute except every + tenth, so the raw H:M:S:F reading overstates elapsed time unless those skipped + counts are added back before converting to seconds. + """ tags = [info.get("format", {}).get("tags") or {}] + [s.get("tags") or {} for s in info.get("streams", [])] for t in tags: tc = t.get("timecode") or t.get("TIMECODE") if tc and fps > 0: - parts = re.split(r"[:;.]", tc) - if len(parts) == 4 and all(p.isdigit() for p in parts): - h, m, sec, frames = (int(p) for p in parts) - return h * 3600 + m * 60 + sec + frames / round(fps) + m = _TIMECODE_RE.match(str(tc).strip()) + if m: + h, m1, mi, m2, sec, m3, frames = m.groups() + h, mi, sec, frames = int(h), int(mi), int(sec), int(frames) + drop_frame = m3 == ";" + frame_duration, fps_round = _ntsc_frame_duration(fps) + if fps_round <= 0: + continue + total_frames = fps_round * 3600 * h + fps_round * 60 * mi + fps_round * sec + frames + if drop_frame: + drop_per_min = 2 if fps_round == 30 else (4 if fps_round == 60 else 0) + total_minutes = 60 * h + mi + total_frames -= drop_per_min * (total_minutes - total_minutes // 10) + return total_frames * frame_duration for t in tags: ref = t.get("time_reference") if ref and str(ref).isdigit() and sample_rate > 0: diff --git a/tests/test_multicam.py b/tests/test_multicam.py index 4ea2c34c..a23d3971 100644 --- a/tests/test_multicam.py +++ b/tests/test_multicam.py @@ -952,3 +952,35 @@ def test_a_removal_where_cameras_stop_and_start_is_bridged_by_another_camera(san assert session.cuts[1]["end"] == pytest.approx(13.0, abs=0.04) assert session.removals[0]["start"] == pytest.approx(10.0, abs=0.04) assert session.removals[0]["end"] == pytest.approx(16.0, abs=0.04) + + +@pytest.mark.parametrize( + "fps,tc,expected", + [ + # 25 fps: integer rate, plain frames/fps. + (25.0, "01:00:00:10", 3600 + 10 / 25), + # 29.97 non-drop frame: separator is ':', frame count is not adjusted, + # but each frame is 1001/30000 s rather than 1/30 s. + (29.97, "00:01:00:00", (30 * 60) * (1001 / 30000)), + # 29.97 drop frame: separator is ';' before the frame field. At 1 minute + # in, drop-frame counting has skipped 2 frame numbers versus wall time, + # so :00;00 lands earlier than non-drop ':00:00:00' would. + (29.97, "00:01:00;00", (30 * 60 - 2) * (1001 / 30000)), + # 29.97 drop frame at the tenth minute: drop frame skips 2 counts at + # the start of every minute except every tenth, so by minute 10 the + # cumulative skip is 2 * (10 - 1) = 18 frames, not 0. + (29.97, "00:10:00;00", (30 * 600 - 18) * (1001 / 30000)), + # 59.94 drop frame: 4 frames skipped per non-tenth minute. + (59.94, "00:01:00;00", (60 * 60 - 4) * (1001 / 60000)), + # 23.976 non-drop: nominal 24 fps grid, real frame duration 1001/24000 s. + (23.976, "00:00:10:00", (24 * 10) * (1001 / 24000)), + ], +) +def test_timecode_seconds_handles_ntsc_pulldown_and_drop_frame(fps, tc, expected): + info = {"format": {"tags": {"timecode": tc}}} + assert mc._timecode_seconds(info, fps, 48000) == pytest.approx(expected, abs=1e-6) + + +def test_timecode_seconds_falls_back_to_time_reference_without_embedded_timecode(): + info = {"format": {"tags": {"time_reference": "48000"}}} + assert mc._timecode_seconds(info, 29.97, 48000) == pytest.approx(1.0, abs=1e-6) From 670cff7b5f216d5a77dc118d3f7489fb1f3c7630 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:20:35 +0400 Subject: [PATCH 004/110] Salt the signal cache key with an analyzer version --- backend/services/signal_cache.py | 6 ++++- tests/test_signal_cache.py | 42 ++++++++++++++++++++++++++++++++ 2 files changed, 47 insertions(+), 1 deletion(-) create mode 100644 tests/test_signal_cache.py diff --git a/backend/services/signal_cache.py b/backend/services/signal_cache.py index 791cb27f..831e6b69 100644 --- a/backend/services/signal_cache.py +++ b/backend/services/signal_cache.py @@ -15,9 +15,13 @@ from config.paths import paths from services.transcript_packer import compute_cache_hash +# Bump when the energy or reaction analyzer's math changes, so old cached +# profiles (keyed only on video content) don't get served under a new algorithm. +SIGNAL_CACHE_VERSION = 2 + def _signals_path(video_path: str) -> str: - return os.path.join(paths["cache"], "signals", f"{compute_cache_hash(video_path)}.json") + return os.path.join(paths["cache"], "signals", f"{compute_cache_hash(video_path)}-v{SIGNAL_CACHE_VERSION}.json") def load_signals(video_path: str) -> dict[str, Any]: diff --git a/tests/test_signal_cache.py b/tests/test_signal_cache.py new file mode 100644 index 00000000..10b710ce --- /dev/null +++ b/tests/test_signal_cache.py @@ -0,0 +1,42 @@ +import os +import sys + +import pytest + +ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..")) +BACKEND_ROOT = os.path.join(ROOT, "backend") +if BACKEND_ROOT not in sys.path: + sys.path.insert(0, BACKEND_ROOT) + +from config.paths import paths # noqa: E402 +from services import signal_cache # noqa: E402 + + +@pytest.fixture +def sandbox(tmp_path, monkeypatch): + monkeypatch.setitem(paths, "cache", str(tmp_path / "cache")) + video = tmp_path / "video.mp4" + video.write_bytes(b"fake video bytes") + return str(video) + + +def test_save_and_load_round_trip(sandbox): + signal_cache.save_signals(sandbox, energy_data=[{"t": 1.0, "v": 0.5}]) + assert signal_cache.load_signals(sandbox) == {"energy_data": [{"t": 1.0, "v": 0.5}]} + + +def test_load_misses_cleanly_when_nothing_cached(sandbox): + assert signal_cache.load_signals(sandbox) == {} + + +def test_cache_path_is_salted_with_the_analyzer_version(sandbox, monkeypatch): + signal_cache.save_signals(sandbox, energy_data=[{"t": 1.0}]) + path_now = signal_cache._signals_path(sandbox) + assert os.path.exists(path_now) + + # Simulate an analyzer algorithm change: bumping the version must stop + # serving the old profile and must not collide with its cache file. + monkeypatch.setattr(signal_cache, "SIGNAL_CACHE_VERSION", signal_cache.SIGNAL_CACHE_VERSION + 1) + path_next = signal_cache._signals_path(sandbox) + assert path_next != path_now + assert signal_cache.load_signals(sandbox) == {} From af4a460e5a94bdd4188ae0e1686afd11d9ade075 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:21:55 +0400 Subject: [PATCH 005/110] Fix whisper.cpp English-only decoding and provision medium/large models with pinned revisions - Pass -l auto instead of omitting -l so whisper-cli runs language detection on auto/unset instead of defaulting to English; label output from the detected result.language, not the (possibly auto) param. - --fast now provisions the multilingual tiny model unless --language en is explicit, since tiny.en silently mis-transcribes everything else. - Add tiny, medium and large-v3 pinned downloads so model_size medium/large actually resolve to a real ggml file instead of a 404. - Pin every whisper.cpp/VAD download to a commit revision instead of resolve/main so an upstream re-upload can't swap bytes behind an already-pinned hash. --- backend/services/transcription.py | 9 ++- backend/services/transcription_whispercpp.py | 17 ++++-- cli/internal/provision/provision.go | 51 +++++++++++++--- cli/main.go | 29 +++++++++- cli/main_test.go | 13 ++++- tests/test_transcription_engine.py | 19 ++++++ tests/test_whispercpp_adapter.py | 61 +++++++++++++++++++- 7 files changed, 181 insertions(+), 18 deletions(-) diff --git a/backend/services/transcription.py b/backend/services/transcription.py index e6859e98..93637409 100644 --- a/backend/services/transcription.py +++ b/backend/services/transcription.py @@ -48,9 +48,16 @@ def _whispercpp_cli() -> Optional[str]: return hermetic if os.path.exists(hermetic) else None + +# "large" alone doesn't name a real ggml file (upstream ships v1/v2/v3/v3-turbo +# builds); provisioning always fetches large-v3, so resolve the same way here. +_WHISPERCPP_MODEL_ALIASES = {"large": "large-v3"} + + def _whispercpp_model(model_size: str) -> str: + resolved = _WHISPERCPP_MODEL_ALIASES.get(model_size, model_size) return os.environ.get("PODCLI_WHISPERCPP_MODEL") or os.path.join( - _managed_home(), "models", f"ggml-{model_size}.bin" + _managed_home(), "models", f"ggml-{resolved}.bin" ) diff --git a/backend/services/transcription_whispercpp.py b/backend/services/transcription_whispercpp.py index 6d3548c4..50ad1ffd 100644 --- a/backend/services/transcription_whispercpp.py +++ b/backend/services/transcription_whispercpp.py @@ -164,7 +164,7 @@ def transcribe_file( model_path: str, whisper_cli: str = "whisper-cli", ffmpeg: str = "ffmpeg", - language: Optional[str] = "en", + language: Optional[str] = None, dtw_model: Optional[str] = None, threads: int = 4, vad: bool = False, @@ -196,8 +196,9 @@ def transcribe_file( # systematic early bias (silence-removal remapping). Off by default; # the energy-snap below addresses the same defect without the bias. cmd += ["--vad", "--vad-model", vad_model] - if language: - cmd += ["-l", language] + # An unset language must not fall through to whisper-cli's own "en" + # default; "auto" makes it run language detection instead. + cmd += ["-l", language or "auto"] subprocess.run(cmd, check=True, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE, text=True, timeout=7200) with open(out_base + ".json", encoding="utf-8") as f: @@ -224,7 +225,15 @@ def transcribe_file( "segments": segments, "words": words, "duration": segments[-1]["end"] if segments else 0.0, - "language": (data.get("params") or {}).get("language") or language or "en", + # whisper-cli reports the language it actually detected under + # "result"; "params" only echoes back what we passed in ("auto" + # when unset), so it can't label the output. + "language": ( + (data.get("result") or {}).get("language") + or (data.get("params") or {}).get("language") + or language + or "en" + ), } finally: shutil.rmtree(tmpdir, ignore_errors=True) diff --git a/cli/internal/provision/provision.go b/cli/internal/provision/provision.go index 2af7c718..2e26a85e 100644 --- a/cli/internal/provision/provision.go +++ b/cli/internal/provision/provision.go @@ -35,22 +35,53 @@ type artifactState struct { Files map[string]string `json:"files"` } +// whisperCppRevision pins every ggml-*.bin download to this commit of +// ggerganov/whisper.cpp so an upstream re-upload (or a force-push that +// reshuffles "main") can't silently swap the bytes behind a pinned hash. +const whisperCppRevision = "5359861c739e955e79d9a303bcbc70fb988958b1" + +func whisperCppURL(file string) string { + return "https://huggingface.co/ggerganov/whisper.cpp/resolve/" + whisperCppRevision + "/" + file +} + var models = map[string]model{ "base": { - URL: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.bin", + URL: whisperCppURL("ggml-base.bin"), SHA256: "60ed5bc3dd14eea856493d334349b405782ddcaf0028d4b5df4088345fba2efe", }, + "tiny": { + URL: whisperCppURL("ggml-tiny.bin"), + SHA256: "be07e048e1e599ad46341c8d2a135645097a538221678b7acdd1b1919c6e1b21", + }, "tiny.en": { - URL: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-tiny.en.bin", + URL: whisperCppURL("ggml-tiny.en.bin"), SHA256: "921e4cf8686fdd993dcd081a5da5b6c365bfde1162e72b08d75ac75289920b1f", }, "small": { - URL: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-small.bin", + URL: whisperCppURL("ggml-small.bin"), SHA256: "1be3a9b2063867b937e64e2ec7483364a79917e157fa98c5d94b5c1fffea987b", }, + "medium": { + URL: whisperCppURL("ggml-medium.bin"), + SHA256: "6c14d5adee5f86394037b4e4e8b59f1673b6cee10e3cf0b11bbdbee79c156208", + }, + "large-v3": { + URL: whisperCppURL("ggml-large-v3.bin"), + SHA256: "64d182b440b98d5203c4f9bd541544d84c605196c4f7b845dfa11fb23594d1e2", + }, +} + +// modelAliases maps a schema-level size name to the models key that actually +// has a pinned download. "large" alone is ambiguous upstream (v1/v2/v3); we +// always provision large-v3, the current best-accuracy build. +var modelAliases = map[string]string{ + "large": "large-v3", } -const vadURL = "https://huggingface.co/ggml-org/whisper-vad/resolve/main/ggml-silero-v5.1.2.bin" +// whisperVADRevision pins the Silero VAD download the same way. +const whisperVADRevision = "9ffd54a1e1ee413ddf265af9913beaf518d1639b" + +const vadURL = "https://huggingface.co/ggml-org/whisper-vad/resolve/" + whisperVADRevision + "/ggml-silero-v5.1.2.bin" const vadSHA = "29940d98d42b91fbd05ce489f3ecf7c72f0a42f027e4875919a28fb4c04ea2cf" func ModelPath(size string) string { @@ -69,12 +100,16 @@ func have(p string) bool { } func EnsureModel(size string) (string, error) { - dest := ModelPath(size) - m, ok := models[size] + resolved := size + if alias, ok := modelAliases[size]; ok { + resolved = alias + } + dest := ModelPath(resolved) + m, ok := models[resolved] if !ok { - return "", fmt.Errorf("unknown model size %q (known: base, tiny.en, small)", size) + return "", fmt.Errorf("unknown model size %q (known: base, tiny, tiny.en, small, medium, large)", size) } - if err := download(m.URL, dest, m.SHA256, "ggml-"+size); err != nil { + if err := download(m.URL, dest, m.SHA256, "ggml-"+resolved); err != nil { return "", err } return dest, nil diff --git a/cli/main.go b/cli/main.go index b615dfb4..40244656 100644 --- a/cli/main.go +++ b/cli/main.go @@ -200,12 +200,37 @@ func runEngine(args []string) int { } func transcribeModel(args []string) string { + fast := false for _, arg := range args { if arg == "--fast" { - return "tiny.en" + fast = true + break } } - return "base" + if !fast { + return "base" + } + // tiny.en is English-only; unset language runs auto-detection, which + // needs the multilingual tiny model same as any non-English request. + lang := strings.ToLower(transcribeLanguage(args)) + if lang == "en" || lang == "english" { + return "tiny.en" + } + return "tiny" +} + +// transcribeLanguage extracts --language/--language= the same way +// transcribeEngine extracts --engine. +func transcribeLanguage(args []string) string { + lang := "" + for i, a := range args { + if a == "--language" && i+1 < len(args) { + lang = args[i+1] + } else if strings.HasPrefix(a, "--language=") { + lang = strings.TrimPrefix(a, "--language=") + } + } + return lang } func configCmd(args []string) int { diff --git a/cli/main_test.go b/cli/main_test.go index c98265bc..221b15a2 100644 --- a/cli/main_test.go +++ b/cli/main_test.go @@ -9,8 +9,17 @@ func TestTranscribeModel(t *testing.T) { if got := transcribeModel([]string{"process", "episode.mp4"}); got != "base" { t.Fatalf("default model = %q, want base", got) } - if got := transcribeModel([]string{"process", "episode.mp4", "--fast"}); got != "tiny.en" { - t.Fatalf("fast model = %q, want tiny.en", got) + if got := transcribeModel([]string{"process", "episode.mp4", "--fast"}); got != "tiny" { + t.Fatalf("fast model with unset language = %q, want tiny (multilingual, no language conditioning)", got) + } + if got := transcribeModel([]string{"process", "episode.mp4", "--fast", "--language", "en"}); got != "tiny.en" { + t.Fatalf("fast english model = %q, want tiny.en", got) + } + if got := transcribeModel([]string{"process", "episode.mp4", "--fast", "--language", "ka"}); got != "tiny" { + t.Fatalf("fast non-english model = %q, want tiny", got) + } + if got := transcribeModel([]string{"process", "episode.mp4", "--fast", "--language=ka"}); got != "tiny" { + t.Fatalf("fast non-english model (= form) = %q, want tiny", got) } } diff --git a/tests/test_transcription_engine.py b/tests/test_transcription_engine.py index 13433822..edd565ae 100644 --- a/tests/test_transcription_engine.py +++ b/tests/test_transcription_engine.py @@ -80,5 +80,24 @@ def test_no_fallback_when_whispercpp_unavailable(self): tr.transcribe_file(self._tmp.name, model_size="base", enable_diarization=False) +class WhisperCppModelAliasTests(unittest.TestCase): + """"large" alone doesn't name a real ggml file upstream (v1/v2/v3/v3-turbo + are separate downloads); provisioning always fetches large-v3, so the + model path lookup must resolve the same alias.""" + + def setUp(self): + self._saved = os.environ.pop("PODCLI_WHISPERCPP_MODEL", None) + + def tearDown(self): + if self._saved is not None: + os.environ["PODCLI_WHISPERCPP_MODEL"] = self._saved + + def test_large_resolves_to_large_v3(self): + self.assertTrue(tr._whispercpp_model("large").endswith("ggml-large-v3.bin")) + + def test_medium_is_unaliased(self): + self.assertTrue(tr._whispercpp_model("medium").endswith("ggml-medium.bin")) + + if __name__ == "__main__": unittest.main() diff --git a/tests/test_whispercpp_adapter.py b/tests/test_whispercpp_adapter.py index 78dea57a..0fbfa4bb 100644 --- a/tests/test_whispercpp_adapter.py +++ b/tests/test_whispercpp_adapter.py @@ -1,13 +1,20 @@ +import json import os import sys +import tempfile import unittest +from unittest import mock ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..")) BACKEND_ROOT = os.path.join(ROOT, "backend") if BACKEND_ROOT not in sys.path: sys.path.insert(0, BACKEND_ROOT) -from services.transcription_whispercpp import _dtw_preset_for_model, _tokens_to_words +from services.transcription_whispercpp import ( + _dtw_preset_for_model, + _tokens_to_words, + transcribe_file, +) class WhisperCppAdapterTests(unittest.TestCase): @@ -19,6 +26,58 @@ def test_sentencepiece_marker_is_removed(self): self.assertEqual([w["word"] for w in words], ["hello", "world"]) +class LanguageHandlingTests(unittest.TestCase): + """whisper-cli defaults to English decoding when -l is omitted entirely, so + an unset language must become "-l auto" rather than no flag at all; the + reported language must come from whisper-cli's detected result, not the + (possibly "auto") param we passed in.""" + + def _fake_run(self, result_language, params_language): + def run(cmd, **kwargs): + of_index = cmd.index("-of") + out_base = cmd[of_index + 1] + payload = { + "params": {"language": params_language}, + "result": {"language": result_language}, + "transcription": [], + } + with open(out_base + ".json", "w", encoding="utf-8") as f: + json.dump(payload, f) + return mock.Mock(returncode=0) + + return mock.Mock(side_effect=run) + + def _run_transcribe(self, language, result_language, params_language): + with tempfile.NamedTemporaryFile(suffix=".mp3") as media, \ + tempfile.NamedTemporaryFile(suffix=".bin") as model, \ + tempfile.NamedTemporaryFile(suffix=".wav") as wav: + fake_run = self._fake_run(result_language, params_language) + with mock.patch("services.transcription_whispercpp.subprocess.run", fake_run): + result = transcribe_file( + media.name, + model.name, + language=language, + wav_path=wav.name, + ) + cmd = fake_run.call_args[0][0] + return result, cmd + + def test_unset_language_passes_auto(self): + _, cmd = self._run_transcribe(None, "es", "auto") + self.assertIn("-l", cmd) + self.assertEqual(cmd[cmd.index("-l") + 1], "auto") + + def test_explicit_language_is_passed_through(self): + _, cmd = self._run_transcribe("fr", "fr", "fr") + self.assertEqual(cmd[cmd.index("-l") + 1], "fr") + + def test_label_prefers_detected_result_language_over_param(self): + # Requesting auto-detect but whisper-cli actually detects Spanish: the + # output must be labeled "es", never the "auto" we passed as -l. + result, _ = self._run_transcribe(None, "es", "auto") + self.assertEqual(result["language"], "es") + + class DtwPresetTests(unittest.TestCase): def test_preset_tracks_the_model_file(self): cases = { From bfa00120c4b497195f651518172d497c5b308038 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:22:15 +0400 Subject: [PATCH 006/110] Invalidate session transcript when the working video changes --- src/handlers/batch-clips.handler.ts | 11 +++++ src/handlers/create-clip.handler.ts | 13 ++++++ src/models/index.ts | 2 + src/ui/web-server.ts | 36 +++++++++++++++- src/utils/video-identity.test.ts | 67 +++++++++++++++++++++++++++++ src/utils/video-identity.ts | 53 +++++++++++++++++++++++ 6 files changed, 181 insertions(+), 1 deletion(-) create mode 100644 src/utils/video-identity.test.ts create mode 100644 src/utils/video-identity.ts diff --git a/src/handlers/batch-clips.handler.ts b/src/handlers/batch-clips.handler.ts index 8b178f8d..14947322 100644 --- a/src/handlers/batch-clips.handler.ts +++ b/src/handlers/batch-clips.handler.ts @@ -5,6 +5,7 @@ import { ClipsHistory } from "../services/clips-history.js"; import { paths } from "../config/paths.js"; import { webServerUrl } from "../config/server.js"; import { validateClipRange } from "../utils/clip-validation.js"; +import { transcriptVideoMismatch } from "../utils/video-identity.js"; import { childLogger } from "../utils/logger.js"; import type { BatchClipsInput, @@ -146,6 +147,16 @@ export async function handleBatchClips(input: BatchClipsInput): Promise return JSON.stringify({ error: "video_path is required (no video in session state)" }); } + // When the caller relies on the session transcript (rather than passing + // transcript_words explicitly), refuse to render against a video that was + // swapped in after that transcript was generated. + if (input.transcript_words == null && transcript) { + const mismatch = transcriptVideoMismatch(state?.transcriptVideoIdentity, videoPath); + if (mismatch) { + return JSON.stringify({ error: mismatch }); + } + } + // Auto-resolve transcript words const transcriptWords = input.transcript_words ?? transcript?.words ?? []; diff --git a/src/handlers/create-clip.handler.ts b/src/handlers/create-clip.handler.ts index b68a2c16..57cb0453 100644 --- a/src/handlers/create-clip.handler.ts +++ b/src/handlers/create-clip.handler.ts @@ -6,6 +6,7 @@ import type { ClipResult, CreateClipInput, SuggestedClip, UIState } from "../mod import { childLogger } from "../utils/logger.js"; import { sliceTranscript } from "../utils/transcript.js"; import { validateClipRange } from "../utils/clip-validation.js"; +import { transcriptVideoMismatch } from "../utils/video-identity.js"; const log = childLogger("create-clip"); const executor = new PythonExecutor(); @@ -180,6 +181,18 @@ export async function handleCreateClip(input: CreateClipInput): Promise // Pull multi-cut segments from suggestion (if available) const keepSegments = suggestion?.segments ?? null; + // When the caller relies on the session transcript (rather than passing + // transcript_words explicitly), refuse to render against a video that was + // swapped in after that transcript was generated — set_video clears the + // transcript itself, but older sessions or a stale on-disk state file can + // still carry a mismatched one. + if (input.transcript_words == null && transcript) { + const mismatch = transcriptVideoMismatch(state?.transcriptVideoIdentity, videoPath); + if (mismatch) { + return JSON.stringify({ error: mismatch }); + } + } + // Validate required fields if (!videoPath) { return JSON.stringify({ error: "video_path is required (no video in session state)" }); diff --git a/src/models/index.ts b/src/models/index.ts index e57668a2..f52ffad9 100644 --- a/src/models/index.ts +++ b/src/models/index.ts @@ -117,6 +117,8 @@ export interface UIState { filePath?: string; activeExportJobId?: string | null; transcript?: TranscriptResult | null; + /** Identity (path + size + mtime) of the video the transcript was generated from. */ + transcriptVideoIdentity?: { path: string; size: number; mtimeMs: number } | null; rawTranscriptText?: string; silenceOriginal?: { videoPath: string; transcript: TranscriptResult } | null; silencePlan?: Record | null; diff --git a/src/ui/web-server.ts b/src/ui/web-server.ts index 9fef0f95..78cdaf9b 100644 --- a/src/ui/web-server.ts +++ b/src/ui/web-server.ts @@ -52,6 +52,7 @@ import { reconcileSegmentsForRange, } from "../utils/transcript.js"; import { formatSrtTime, formatVttTime } from "../utils/srt-time.js"; +import { computeVideoIdentity, type VideoIdentity } from "../utils/video-identity.js"; import { errMsg } from "../utils/errors.js"; import { resolveByteRange } from "../utils/http-range.js"; import { @@ -157,6 +158,7 @@ interface UIState { filePath: string; activeExportJobId: string | null; transcript: ServerTranscript | null; + transcriptVideoIdentity: VideoIdentity | null; rawTranscriptText: string; silenceOriginal: SilenceOriginal | null; silencePlan: SilencePlan | null; @@ -195,6 +197,10 @@ function loadPersistedState(): UIState { saved.videoPath = ""; saved.filePath = ""; saved.phase = "idle"; + saved.transcript = null; + saved.transcriptVideoIdentity = null; + saved.suggestions = []; + saved.deselectedIndices = []; } if (saved.silenceOriginal?.videoPath && !existsSync(saved.silenceOriginal.videoPath)) { saved.silenceOriginal = null; @@ -204,6 +210,7 @@ function loadPersistedState(): UIState { filePath: saved.filePath || "", activeExportJobId: null, transcript: saved.transcript || null, + transcriptVideoIdentity: saved.transcriptVideoIdentity || null, rawTranscriptText: saved.rawTranscriptText || "", silenceOriginal: saved.silenceOriginal || null, silencePlan: saved.silencePlan || null, @@ -246,6 +253,7 @@ function loadPersistedState(): UIState { filePath: "", activeExportJobId: null, transcript: null, + transcriptVideoIdentity: null, rawTranscriptText: "", silenceOriginal: null, silencePlan: null, @@ -4149,9 +4157,35 @@ app.post("/api/ui-state", (req, res) => { return; } + // Changing the video without a transcript arriving in the same call means + // the old transcript, suggestions and selections describe a recording + // that's no longer loaded. Carrying them forward lets create_clip burn + // captions and timings from the wrong video. transcribe_podcast and + // import_transcript always send videoPath and transcript together, so + // this only fires for a bare set_video. + const videoChanged = + body.videoPath !== undefined && + body.videoPath !== uiState.videoPath && + body.transcript === undefined; + if (videoChanged) { + uiState.transcript = null; + uiState.transcriptVideoIdentity = null; + uiState.rawTranscriptText = ""; + uiState.suggestions = []; + uiState.deselectedIndices = []; + uiState.energyData = {}; + uiState.silenceOriginal = null; + uiState.silencePlan = null; + } + if (body.videoPath !== undefined) uiState.videoPath = body.videoPath; if (body.filePath !== undefined) uiState.filePath = body.filePath; - if (body.transcript !== undefined) uiState.transcript = body.transcript; + if (body.transcript !== undefined) { + uiState.transcript = body.transcript; + uiState.transcriptVideoIdentity = body.transcript + ? computeVideoIdentity(body.videoPath ?? uiState.videoPath ?? "") + : null; + } if (body.rawTranscriptText !== undefined) uiState.rawTranscriptText = body.rawTranscriptText; if (body.silenceOriginal !== undefined) { diff --git a/src/utils/video-identity.test.ts b/src/utils/video-identity.test.ts new file mode 100644 index 00000000..7a8c2611 --- /dev/null +++ b/src/utils/video-identity.test.ts @@ -0,0 +1,67 @@ +import { describe, it, expect, beforeAll, afterAll } from "vitest"; +import { mkdtempSync, writeFileSync, rmSync, utimesSync } from "fs"; +import { tmpdir } from "os"; +import { join } from "path"; +import { computeVideoIdentity, identitiesMatch, transcriptVideoMismatch } from "./video-identity.js"; + +describe("video-identity", () => { + let dir: string; + let filePath: string; + + beforeAll(() => { + dir = mkdtempSync(join(tmpdir(), "podcli-video-identity-")); + filePath = join(dir, "video.mp4"); + writeFileSync(filePath, "abc"); + }); + + afterAll(() => { + rmSync(dir, { recursive: true, force: true }); + }); + + it("returns null for a missing file", () => { + expect(computeVideoIdentity(join(dir, "missing.mp4"))).toBeNull(); + }); + + it("computes an identity for an existing file", () => { + const identity = computeVideoIdentity(filePath); + expect(identity?.path).toBe(filePath); + expect(identity?.size).toBe(3); + }); + + it("matches two identical identities", () => { + const a = computeVideoIdentity(filePath); + const b = computeVideoIdentity(filePath); + expect(identitiesMatch(a, b)).toBe(true); + }); + + it("does not match once the file is rewritten (size or mtime change)", () => { + const before = computeVideoIdentity(filePath); + writeFileSync(filePath, "a longer replacement body"); + utimesSync(filePath, new Date(Date.now() + 5000), new Date(Date.now() + 5000)); + const after = computeVideoIdentity(filePath); + expect(identitiesMatch(before, after)).toBe(false); + }); + + it("transcriptVideoMismatch is null when there's no recorded identity", () => { + expect(transcriptVideoMismatch(null, filePath)).toBeNull(); + expect(transcriptVideoMismatch(undefined, filePath)).toBeNull(); + }); + + it("transcriptVideoMismatch is null when the current video can't be statted", () => { + const stale = computeVideoIdentity(filePath); + expect(transcriptVideoMismatch(stale, join(dir, "gone.mp4"))).toBeNull(); + }); + + it("transcriptVideoMismatch names the mismatch when the file changed", () => { + const stale = computeVideoIdentity(filePath); + writeFileSync(filePath, "totally different content, different length"); + const msg = transcriptVideoMismatch(stale, filePath); + expect(msg).toMatch(/different video/); + expect(msg).toContain(filePath); + }); + + it("transcriptVideoMismatch is null when the file is unchanged", () => { + const identity = computeVideoIdentity(filePath); + expect(transcriptVideoMismatch(identity, filePath)).toBeNull(); + }); +}); diff --git a/src/utils/video-identity.ts b/src/utils/video-identity.ts new file mode 100644 index 00000000..474bacc5 --- /dev/null +++ b/src/utils/video-identity.ts @@ -0,0 +1,53 @@ +import { statSync } from "fs"; + +/** + * Identifies a specific video file on disk, not just its path, so a + * transcript can be tied to the exact bytes it was generated from. A path + * can be reused for a different recording (re-record, re-encode, swap in a + * trimmed version); size and mtime catch that even though the path matches. + */ +export interface VideoIdentity { + path: string; + size: number; + mtimeMs: number; +} + +/** Stat a video file into a VideoIdentity, or null if it can't be read. */ +export function computeVideoIdentity(path: string): VideoIdentity | null { + try { + const stat = statSync(path); + return { path, size: stat.size, mtimeMs: stat.mtimeMs }; + } catch { + return null; + } +} + +/** True when two identities refer to the same file snapshot. */ +export function identitiesMatch( + a: VideoIdentity | null | undefined, + b: VideoIdentity | null | undefined, +): boolean { + if (!a || !b) return false; + return a.path === b.path && a.size === b.size && a.mtimeMs === b.mtimeMs; +} + +/** + * Checks whether a session's recorded transcript still belongs to the video + * currently loaded at videoPath. Returns an error string naming the mismatch + * when it doesn't, or null when it's safe to proceed (including when there's + * no recorded identity to check, e.g. older sessions before this check existed). + */ +export function transcriptVideoMismatch( + transcriptVideoIdentity: VideoIdentity | null | undefined, + videoPath: string, +): string | null { + if (!transcriptVideoIdentity) return null; + const current = computeVideoIdentity(videoPath); + if (!current) return null; + if (identitiesMatch(transcriptVideoIdentity, current)) return null; + return ( + `Session transcript belongs to a different video than the one currently set ` + + `(${transcriptVideoIdentity.path}). Re-transcribe or re-import a transcript for ` + + `${videoPath} before rendering.` + ); +} From 9b3402919171db9cfbfeeb09b4c68239e9df6460 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:22:49 +0400 Subject: [PATCH 007/110] Detect sentence openers in caseless scripts like Georgian --- backend/services/multicam.py | 25 ++++++++++++++++++++++++- tests/test_multicam.py | 15 +++++++++++++++ 2 files changed, 39 insertions(+), 1 deletion(-) diff --git a/backend/services/multicam.py b/backend/services/multicam.py index 75b58395..6d4e26cc 100644 --- a/backend/services/multicam.py +++ b/backend/services/multicam.py @@ -1522,6 +1522,29 @@ def loudest(start: float, end: float) -> tuple[str, float]: _SENTENCE_END = (".", "?", "!") +_GEORGIAN_RANGES = ((0x10A0, 0x10FF), (0x1C90, 0x1CBF)) + + +def _looks_like_sentence_opener(text: str) -> bool: + """True if `text` could start a new sentence, by case or by having none. + + `str.isupper()` alone misses caseless scripts: CJK, Arabic, Thai, and + Hebrew letters are never upper or lower (`ch.upper() == ch.lower()` + catches those). Georgian is a special case: Unicode still carries a + Mtavruli uppercase mapping for it, so `str.isupper()`/`islower()` report + it as cased, but real Georgian text is written only in the lowercase + Mkhedruli form and never uses that case distinction, so every opener was + silently dropped. Treat Georgian letters as potential openers too. + """ + ch = text[:1] + if not ch.isalpha(): + return False + if ch.isupper() or ch.upper() == ch.lower(): + return True + cp = ord(ch) + return any(lo <= cp <= hi for lo, hi in _GEORGIAN_RANGES) + + def _settle_turn_edges(words: list[dict], margins: list[float], clear: float = 6.0) -> None: """Fix credits that loose word timestamps get wrong around a turn change. @@ -1542,7 +1565,7 @@ def rejoin_strays() -> None: for i in range(1, len(words) - 1): prev, word, nxt = words[i - 1], words[i], words[i + 1] if (margins[i] < clear and word["person"] == prev["person"] != nxt["person"] - and prev["text"].endswith(_SENTENCE_END) and word["text"][:1].isupper()): + and prev["text"].endswith(_SENTENCE_END) and _looks_like_sentence_opener(word["text"])): word["person"] = nxt["person"] # Moving a sentence opener can leave the word after it stranded; rejoin it too. rejoin_strays() diff --git a/tests/test_multicam.py b/tests/test_multicam.py index a23d3971..84f01478 100644 --- a/tests/test_multicam.py +++ b/tests/test_multicam.py @@ -605,6 +605,21 @@ def words(spec): mc._settle_turn_edges(w, [0.0, 15.0, 15.0]) assert [x["person"] for x in w] == ["a", "a", "b"] + # Georgian has no letter case, so a sentence opener there can't pass an + # isupper() check. It must still move to the next speaker. + w = words([("დამიდა.", "g"), ("კარგი,", "g"), + ("ამიტაო", "h")]) + mc._settle_turn_edges(w, [0.0] * len(w)) + assert [x["person"] for x in w] == ["g", "h", "h"] + + +def test_sentence_opener_detection_handles_caseless_scripts(): + assert mc._looks_like_sentence_opener("კარგი") # Georgian, no case + assert mc._looks_like_sentence_opener("Right,") + assert not mc._looks_like_sentence_opener("right,") + assert not mc._looks_like_sentence_opener("123") + assert not mc._looks_like_sentence_opener("") + # --- Remote recordings ------------------------------------------------------------ From d3777cbc68e730ba2a22d7368488e0f1a0b1a9e8 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:23:04 +0400 Subject: [PATCH 008/110] Fix PodStack command wiring and give install-aware studio start hints - auto.md: replace nonexistent transcribe_status with job_status in allowed-tools and in the transcription poll loop - auto.md, bootstrap-knowledge.md: restore the dropped .podcli/knowledge/ path in the blank `` placeholders - auto.md: fallback instruction names both podcli studio and npm run ui instead of only the source-checkout command - generate-titles.md, plan-episode.md, process-transcript.md, produce-shorts.md: add the missing mcp__podcli__knowledge_base to allowed-tools since the body already calls it - server.ts/version.ts: add studioStartCommand(), which picks podcli studio vs npm run ui based on whether PODCLI_VERSION (set only by the Go launcher) is present, and use it everywhere the server tells the agent to start the Web UI - add a vitest contract test that parses every command's frontmatter and body against the server's actually-registered tool names --- .claude/commands/auto.md | 10 ++-- .claude/commands/bootstrap-knowledge.md | 2 +- .claude/commands/generate-titles.md | 2 +- .claude/commands/plan-episode.md | 2 +- .claude/commands/process-transcript.md | 2 +- .claude/commands/produce-shorts.md | 2 +- src/podstack-commands.test.ts | 64 +++++++++++++++++++++++++ src/server.ts | 30 ++++++------ src/version.ts | 12 +++++ 9 files changed, 101 insertions(+), 25 deletions(-) create mode 100644 src/podstack-commands.test.ts diff --git a/.claude/commands/auto.md b/.claude/commands/auto.md index 87aa8f5a..afa6fe37 100644 --- a/.claude/commands/auto.md +++ b/.claude/commands/auto.md @@ -1,6 +1,6 @@ --- description: One-verb pipeline — drop a video, confirm strategy, render clips -allowed-tools: Read, Bash, mcp__podcli__transcribe_podcast, mcp__podcli__transcribe_start, mcp__podcli__transcribe_status, mcp__podcli__get_ui_state, mcp__podcli__set_video, mcp__podcli__suggest_clips, mcp__podcli__batch_create_clips, mcp__podcli__knowledge_base, mcp__podcli__clip_history +allowed-tools: Read, Bash, mcp__podcli__transcribe_podcast, mcp__podcli__transcribe_start, mcp__podcli__job_status, mcp__podcli__get_ui_state, mcp__podcli__set_video, mcp__podcli__suggest_clips, mcp__podcli__batch_create_clips, mcp__podcli__knowledge_base, mcp__podcli__clip_history argument-hint: [video-path-or-episode-slug] [optional: count e.g. "5 clips"] triggers: - auto @@ -21,7 +21,7 @@ This command orchestrates the existing MCP tools on top of the compact packed tr 1. **Read, don't watch.** Reason about clips from the packed markdown view — not raw segments, not frame dumps. 2. **Strategy first, render after.** Propose the cut list and WAIT for user confirmation before calling `batch_create_clips`. -3. **Knowledge base is context, not template.** If `` exists, read it for brand voice and format preferences. If not, infer from the content itself. +3. **Knowledge base is context, not template.** If `.podcli/knowledge/` exists, read it for brand voice and format preferences. If not, infer from the content itself. 4. **Never silently render.** Every clip that ships must appear in the proposal the user approved. 5. **Every clip carries its own context.** A stranger who never heard the episode has to follow it from the first second. If the moment is an answer, the question comes with it. @@ -46,14 +46,14 @@ This command orchestrates the existing MCP tools on top of the compact packed tr - Call `transcribe_start(file_path)` → returns `{job_id, cached, estimate}` immediately. - If `cached: true`, skip to step 3. - Otherwise emit a short status to the user: _"Transcription started — estimated {estimate}. I'll check progress every 30s."_ - - Loop: call `transcribe_status(job_id, wait_seconds: 30)`. Between calls, emit ONE terse line to the user like `"Progress: 47% — pyannote diarization"`. Keep it to one line per poll — no repeat prose. Exit the loop when `done: true`. + - Loop: call `job_status(job_id, wait_seconds: 30)`. Between calls, emit ONE terse line to the user like `"Progress: 47% — pyannote diarization"`. Keep it to one line per poll — no repeat prose. Exit the loop when `done: true`. - If `status: "error"`, stop and report the error. 3. Read the packed transcript: `get_ui_state(include_transcript: true)`. This returns a compact phrase-grouped view with speakers, silence gaps, and energy peaks. - **If the header says speakers: 0**, stop and tell the user before going further. Without speaker labels you cannot tell a question from an answer, so the whole question-with-the-answer rule below is inert and the picks will be worse. Offer to re-transcribe with `transcribe_start(file_path, enable_diarization: true)`. Only continue without it if the user says to. -4. If `` exists, read `01-brand-identity.md`, `02-voice-and-tone.md`, and `04-shorts-creation-guide.md` for show context. Skip silently if missing — `/auto` works on any content. +4. If `.podcli/knowledge/` exists, read `01-brand-identity.md`, `02-voice-and-tone.md`, and `04-shorts-creation-guide.md` for show context. Skip silently if missing — `/auto` works on any content. 5. Call `clip_history` to see what's already been shipped for this episode. Avoid duplicates in the proposal. -**Fallback**: if `transcribe_start` returns an error about the Web UI not running, tell the user and offer either (a) run `npm run ui` in another terminal then retry, or (b) fall back to the synchronous `transcribe_podcast` (no live progress, works silently). +**Fallback**: if `transcribe_start` returns an error about the Web UI not running, tell the user and offer either (a) start the Web UI in another terminal then retry — `podcli studio` for a launcher install, `npm run ui` in a source checkout — or (b) fall back to the synchronous `transcribe_podcast` (no live progress, works silently). ### Phase 2 — Topic Map (silent) diff --git a/.claude/commands/bootstrap-knowledge.md b/.claude/commands/bootstrap-knowledge.md index caad6f39..a1a2695b 100644 --- a/.claude/commands/bootstrap-knowledge.md +++ b/.claude/commands/bootstrap-knowledge.md @@ -18,7 +18,7 @@ triggers: ## Before starting -1. If `` has no files, run `podcli knowledge init` first so all 14 templates exist. +1. If `.podcli/knowledge/` has no files, run `podcli knowledge init` first so all 14 templates exist. 2. Ask for whichever of these the user has not provided: - Channel or podcast URL (YouTube channel, Spotify show, RSS feed) - Or a few sentences about the show if nothing is published yet diff --git a/.claude/commands/generate-titles.md b/.claude/commands/generate-titles.md index 39224bc5..e91f2d3d 100644 --- a/.claude/commands/generate-titles.md +++ b/.claude/commands/generate-titles.md @@ -1,6 +1,6 @@ --- description: Generate 8 verified title options for a clip, moment, or episode -allowed-tools: Read +allowed-tools: Read, mcp__podcli__knowledge_base argument-hint: [clip-transcript-or-moment-brief] triggers: - titles for diff --git a/.claude/commands/plan-episode.md b/.claude/commands/plan-episode.md index b90f3e12..77d030c7 100644 --- a/.claude/commands/plan-episode.md +++ b/.claude/commands/plan-episode.md @@ -1,6 +1,6 @@ --- description: Design questions, story arc, and moment map BEFORE recording an episode -allowed-tools: Read, Write +allowed-tools: Read, Write, mcp__podcli__knowledge_base argument-hint: [guest-name-and-company] triggers: - plan episode diff --git a/.claude/commands/process-transcript.md b/.claude/commands/process-transcript.md index bf609fb9..9074a9d0 100644 --- a/.claude/commands/process-transcript.md +++ b/.claude/commands/process-transcript.md @@ -1,6 +1,6 @@ --- description: Extract, score, and classify the best moments from a raw podcast transcript -allowed-tools: Read, Write +allowed-tools: Read, Write, mcp__podcli__knowledge_base argument-hint: [transcript-file-or-paste] triggers: - transcript diff --git a/.claude/commands/produce-shorts.md b/.claude/commands/produce-shorts.md index cb1f1aa5..cbd07f92 100644 --- a/.claude/commands/produce-shorts.md +++ b/.claude/commands/produce-shorts.md @@ -1,6 +1,6 @@ --- description: Full pipeline from transcript to publish-ready content package -allowed-tools: Read, Write, Edit, Task +allowed-tools: Read, Write, Edit, Task, mcp__podcli__knowledge_base argument-hint: [transcript-file-or-episode-number] triggers: - process episode diff --git a/src/podstack-commands.test.ts b/src/podstack-commands.test.ts new file mode 100644 index 00000000..bde383cd --- /dev/null +++ b/src/podstack-commands.test.ts @@ -0,0 +1,64 @@ +import { describe, it, expect } from "vitest"; +import { readFileSync, readdirSync } from "fs"; +import { dirname, join, resolve } from "path"; +import { fileURLToPath } from "url"; +import { createServer } from "./server.js"; + +// Catches the class of bug fixed alongside this test: a PodStack command +// frontmatter or body references an MCP tool name that doesn't match what +// the server actually registers (e.g. a renamed tool, or a typo). +const repoRoot = resolve(dirname(fileURLToPath(import.meta.url)), ".."); +const commandsDir = join(repoRoot, ".claude", "commands"); + +function registeredToolNames(): Set { + const server = createServer() as unknown as { _registeredTools: Record }; + return new Set(Object.keys(server._registeredTools)); +} + +function parseFrontmatter(source: string): { allowedTools: string[]; body: string } { + const match = source.match(/^---\n([\s\S]*?)\n---\n([\s\S]*)$/); + if (!match) return { allowedTools: [], body: source }; + const [, frontmatter, body] = match; + const line = frontmatter.split("\n").find((l) => l.startsWith("allowed-tools:")); + const allowedTools = line + ? line + .slice("allowed-tools:".length) + .split(",") + .map((t) => t.trim()) + .filter(Boolean) + : []; + return { allowedTools, body }; +} + +const commandFiles = readdirSync(commandsDir).filter((f) => f.endsWith(".md")); + +describe("PodStack command frontmatter matches registered MCP tools", () => { + const toolNames = registeredToolNames(); + + it("the server registers at least one tool", () => { + expect(toolNames.size).toBeGreaterThan(0); + }); + + for (const file of commandFiles) { + it(`${file}: every mcp__podcli__* in allowed-tools is a registered tool`, () => { + const source = readFileSync(join(commandsDir, file), "utf-8"); + const { allowedTools } = parseFrontmatter(source); + const mcpTools = allowedTools.filter((t) => t.startsWith("mcp__podcli__")); + const unknown = mcpTools + .map((t) => t.slice("mcp__podcli__".length)) + .filter((name) => !toolNames.has(name)); + expect(unknown, `${file} allows unknown tools: ${unknown.join(", ")}`).toEqual([]); + }); + + it(`${file}: every backtick tool-call reference in the body names a registered tool`, () => { + const source = readFileSync(join(commandsDir, file), "utf-8"); + const { body } = parseFrontmatter(source); + // Only call-shaped references, e.g. `set_video(file_path)` or + // `job_status(job_id, wait_seconds: 30)` — bare backticked words like + // `payoff` or `async_mode` are field names, not tool calls. + const calls = Array.from(body.matchAll(/`([a-z][a-z0-9_]*)\(/g)).map((m) => m[1]); + const unknown = calls.filter((name) => !toolNames.has(name)); + expect(unknown, `${file} calls unknown tool(s): ${unknown.join(", ")}`).toEqual([]); + }); + } +}); diff --git a/src/server.ts b/src/server.ts index 2419bd83..d2392602 100644 --- a/src/server.ts +++ b/src/server.ts @@ -33,7 +33,7 @@ import { paths } from "./config/paths.js"; import { webServerUrl } from "./config/server.js"; import { childLogger } from "./utils/logger.js"; import { mcpError } from "./utils/errors.js"; -import { podcliVersion } from "./version.js"; +import { podcliVersion, studioStartCommand } from "./version.js"; import type { Format, SuggestedClip, UIState, WordTimestamp } from "./models/index.js"; const log = childLogger("server"); @@ -190,7 +190,7 @@ async function getWorkflowGuidance(): Promise { "4. Suggest clips: analyze the transcript yourself, then call suggest_clips with your picks.\n" + " Each pick needs a payoff and a standalone check, and an answer needs its question.\n" + "5. Export: use batch_create_clips(export_selected: true) or create_clip(clip_number: N)\n\n" + - "Note: The Web UI is not running. Start it with: npm run ui" + `Note: The Web UI is not running. Start it with: ${studioStartCommand()}` ); } @@ -1458,7 +1458,7 @@ export function createServer(): McpServer { content: [ { type: "text" as const, - text: `Web UI is not running. Start with: npm run ui\n\n${guidance}`, + text: `Web UI is not running. Start with: ${studioStartCommand()}\n\n${guidance}`, }, ], }; @@ -1587,7 +1587,7 @@ export function createServer(): McpServer { content: [ { type: "text" as const, - text: "Web UI is not running. Start with: npm run ui", + text: `Web UI is not running. Start with: ${studioStartCommand()}`, }, ], }; @@ -1670,7 +1670,7 @@ export function createServer(): McpServer { content: [ { type: "text" as const, - text: "Web UI is not running. Start with: npm run ui", + text: `Web UI is not running. Start with: ${studioStartCommand()}`, }, ], }; @@ -1762,7 +1762,7 @@ export function createServer(): McpServer { content: [ { type: "text" as const, - text: "Web UI is not running. Start with: npm run ui", + text: `Web UI is not running. Start with: ${studioStartCommand()}`, }, ], }; @@ -1816,7 +1816,7 @@ export function createServer(): McpServer { content: [ { type: "text" as const, - text: "Web UI is not running. Start with: npm run ui", + text: `Web UI is not running. Start with: ${studioStartCommand()}`, }, ], }; @@ -1953,7 +1953,7 @@ export function createServer(): McpServer { content: [ { type: "text" as const, - text: "Web UI is not running. Start with: npm run ui", + text: `Web UI is not running. Start with: ${studioStartCommand()}`, }, ], }; @@ -2005,7 +2005,7 @@ export function createServer(): McpServer { } catch (err: unknown) { const msg = err instanceof Error ? err.message : String(err); if (msg.includes("ECONNREFUSED") || msg.includes("fetch failed")) { - return { content: [{ type: "text" as const, text: "Web UI is not running. Start with: npm run ui" }] }; + return { content: [{ type: "text" as const, text: `Web UI is not running. Start with: ${studioStartCommand()}` }] }; } return mcpError(msg); } @@ -2105,7 +2105,7 @@ export function createServer(): McpServer { content: [ { type: "text" as const, - text: "Web UI is not running. Start with: npm run ui", + text: `Web UI is not running. Start with: ${studioStartCommand()}`, }, ], }; @@ -2207,7 +2207,7 @@ export function createServer(): McpServer { content: [ { type: "text" as const, - text: "Web UI is not running. Start with: npm run ui", + text: `Web UI is not running. Start with: ${studioStartCommand()}`, }, ], }; @@ -2326,7 +2326,7 @@ export function createServer(): McpServer { const msg = err instanceof Error ? err.message : String(err); if (msg.includes("ECONNREFUSED") || msg.includes("fetch failed")) { return { - content: [{ type: "text" as const, text: "Web UI is not running. Start with: npm run ui" }], + content: [{ type: "text" as const, text: `Web UI is not running. Start with: ${studioStartCommand()}` }], }; } return mcpError(err); @@ -2386,7 +2386,7 @@ export function createServer(): McpServer { content: [ { type: "text" as const, - text: "Web UI is not running. Start with: npm run ui", + text: `Web UI is not running. Start with: ${studioStartCommand()}`, }, ], }; @@ -2487,7 +2487,7 @@ export function createServer(): McpServer { content: [ { type: "text" as const, - text: "Web UI is not running. Start with: npm run ui", + text: `Web UI is not running. Start with: ${studioStartCommand()}`, }, ], }; @@ -2575,7 +2575,7 @@ export function createServer(): McpServer { content: [ { type: "text" as const, - text: "Web UI is not running. Start with: npm run ui", + text: `Web UI is not running. Start with: ${studioStartCommand()}`, }, ], }; diff --git a/src/version.ts b/src/version.ts index 42c19126..eed9051c 100644 --- a/src/version.ts +++ b/src/version.ts @@ -15,3 +15,15 @@ export function podcliVersion(): string { return "0.0.0-dev"; } + +// PODCLI_VERSION is only set when the Go launcher spawns this process (see +// cli/internal/engine/engine.go nodeEnv). A source checkout running `npm run +// ui` directly never has it, so its presence tells the two install types apart. +export function isLauncherInstall(): boolean { + return !!process.env.PODCLI_VERSION?.trim(); +} + +/** The right command to start the Web UI for whichever install type is running. */ +export function studioStartCommand(): string { + return isLauncherInstall() ? "podcli studio" : "npm run ui"; +} From d1d31d47ccca630d3ea9b7711eb37537c0e60545 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:24:55 +0400 Subject: [PATCH 009/110] Key preview stills by sync basis so a nudge invalidates them --- backend/services/multicam.py | 20 +++++++++++++++++--- tests/test_multicam.py | 27 +++++++++++++++++++++++++++ 2 files changed, 44 insertions(+), 3 deletions(-) diff --git a/backend/services/multicam.py b/backend/services/multicam.py index 6d4e26cc..bb77cf5f 100644 --- a/backend/services/multicam.py +++ b/backend/services/multicam.py @@ -1384,6 +1384,19 @@ def _still(session: MulticamSession, cam: Source, tl: float, out: Path, look: st return out +def _sync_basis(s: Source) -> str: + """Short fingerprint of a source's offset and speed. + + Stills are cached to disk by filename. The source's mapping from timeline + time to source time (`source_time`) depends on offset and speed, so a + cache key that only captures the timeline moment `at` goes stale the + instant a nudge or re-sync changes that mapping: the filename looks the + same but would now decode a different source frame. + """ + raw = f"{s.offset}:{s.speed}" + return hashlib.sha1(raw.encode()).hexdigest()[:8] + + def previews(session: MulticamSession, *, looks: bool = False, at: Optional[float] = None) -> dict: """One still per camera, and optionally one still per look from the wide camera.""" work = _work_dir(session.session_id) @@ -1398,7 +1411,7 @@ def moment(s: Source) -> float: if s.kind != "video": continue t = moment(s) - out = work / f"frame-{s.id}-{int(t * 10)}.jpg" + out = work / f"frame-{s.id}-{int(t * 10)}-{_sync_basis(s)}.jpg" if not out.exists(): _still(session, s, t, out) frames["cameras"][s.id] = str(out) @@ -1406,10 +1419,11 @@ def moment(s: Source) -> float: cams = session.cameras() or [s for s in session.sources if s.kind == "video"] if cams: cam = next((c for c in cams if c.person == "wide"), cams[0]) + t = moment(cam) for name in LOOKS: - out = work / f"look-{cam.id}-{name}.jpg" + out = work / f"look-{cam.id}-{name}-{int(t * 10)}-{_sync_basis(cam)}.jpg" if not out.exists(): - _still(session, cam, moment(cam), out, look=name, width=640) + _still(session, cam, t, out, look=name, width=640) frames["looks"][name] = str(out) return frames diff --git a/tests/test_multicam.py b/tests/test_multicam.py index 84f01478..21a41fa2 100644 --- a/tests/test_multicam.py +++ b/tests/test_multicam.py @@ -526,6 +526,33 @@ def test_cli_json_mode_applies_a_cut_and_builds_the_preview(episode, monkeypatch assert "error" in json.loads(capsys.readouterr().out) +def test_preview_stills_regenerate_after_a_nudge_instead_of_serving_a_stale_frame(episode): + session = mc.new_session(folder=str(episode), people=["Nika", "Ana"]) + mc.update_mapping(session, {"sources": [ + {"id": s.id, "role": "camera", "person": "nika" if "one" in os.path.basename(s.path) else "ana"} + for s in session.sources if s.kind == "video" + ]}) + session = mc.sync_session(session) + cam = next(s for s in session.sources if os.path.basename(s.path) == "cam_one.mp4") + + frames = mc.previews(session, looks=True, at=cam.timeline_start() + 1.0) + before = frames["cameras"][cam.id] + look_before = frames["looks"][next(iter(frames["looks"]))] + assert os.path.exists(before) and os.path.exists(look_before) + + # Nudging changes the offset->source mapping for the same timeline moment, + # so the cache key must change and a fresh still must be rendered: reusing + # the old file would show the pre-nudge frame. + mc.update_mapping(session, {"sources": [{"id": cam.id, "nudge": 2.0}]}) + session = mc.MulticamSession.load(session.session_id) + cam = session.source(cam.id) + frames = mc.previews(session, looks=True, at=cam.timeline_start() + 1.0) + after = frames["cameras"][cam.id] + look_after = frames["looks"][next(iter(frames["looks"]))] + assert after != before and os.path.exists(after) + assert look_after != look_before and os.path.exists(look_after) + + def test_last_person_is_the_guest_and_roles_survive_renames(episode): session = mc.new_session(folder=str(episode), people=["Nihal", "Cameron", "Ana"]) assert [(p.name, p.role) for p in session.people] == [("Nihal", "host"), ("Cameron", "host"), ("Ana", "guest")] From d950e2423c1729889224d7a4aa5849fb09a4f5bd Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:26:08 +0400 Subject: [PATCH 010/110] Make get_ui_state work when the Web UI is not running readUIState() now falls back to reading paths.uiState straight off disk when the Web UI's HTTP API is unreachable, instead of reporting state as unavailable. get_ui_state's handler now goes through readUIState() too (it previously duplicated the fetch and only reported the Web UI as down), so it only falls back to start-from-scratch guidance when there is truly no state anywhere. --- src/server.ts | 50 +++++++++++++-------- src/ui-state-offline.test.ts | 86 ++++++++++++++++++++++++++++++++++++ 2 files changed, 118 insertions(+), 18 deletions(-) create mode 100644 src/ui-state-offline.test.ts diff --git a/src/server.ts b/src/server.ts index d2392602..13ba1767 100644 --- a/src/server.ts +++ b/src/server.ts @@ -1,6 +1,6 @@ import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; import { z } from "zod"; -import { readFileSync, writeFileSync } from "fs"; +import { existsSync, readFileSync, writeFileSync } from "fs"; import { transcribeToolDef, @@ -165,16 +165,39 @@ interface ImportTranscriptResult extends ApiError { }; } +/** + * Read session state straight off disk (paths.uiState) instead of the Web + * UI's HTTP API. Used when the Web UI isn't running — the state file is the + * same JSON the UI persists on every change, so the agent isn't blind just + * because nothing is listening on webServerUrl. + */ +function readUIStateFromDisk(): ServerUIState | null { + try { + if (!existsSync(paths.uiState)) return null; + const raw = JSON.parse(readFileSync(paths.uiState, "utf-8")) as UIState; + const words = raw.transcript?.words; + return { + ...raw, + transcriptWordCount: Array.isArray(words) ? words.length : 0, + }; + } catch (err) { + log.debug("readUIStateFromDisk failed", { + err: err instanceof Error ? err.message : String(err), + }); + return null; + } +} + async function readUIState(): Promise { try { const res = await fetch(`${webServerUrl}/api/ui-state`); - if (!res.ok) return null; + if (!res.ok) return readUIStateFromDisk(); return (await res.json()) as ServerUIState; } catch (err) { - log.debug("readUIState failed (UI likely not running)", { + log.debug("readUIState via Web UI failed, falling back to disk", { err: err instanceof Error ? err.message : String(err), }); - return null; + return readUIStateFromDisk(); } } @@ -1358,9 +1381,11 @@ export function createServer(): McpServer { }, async ({ include_transcript }) => { try { - const res = await fetch(`${webServerUrl}/api/ui-state`); - if (!res.ok) throw new Error(`HTTP ${res.status}`); - const state = (await res.json()) as ServerUIState; + const state = await readUIState(); + if (!state) { + const guidance = await getWorkflowGuidance(); + return { content: [{ type: "text" as const, text: guidance }] }; + } const lines: string[] = []; lines.push(`Phase: ${state.phase}`); @@ -1452,17 +1477,6 @@ export function createServer(): McpServer { return { content: [{ type: "text" as const, text: lines.join("\n") }] }; } catch (err: unknown) { const msg = err instanceof Error ? err.message : String(err); - if (msg.includes("ECONNREFUSED") || msg.includes("fetch failed")) { - const guidance = await getWorkflowGuidance(); - return { - content: [ - { - type: "text" as const, - text: `Web UI is not running. Start with: ${studioStartCommand()}\n\n${guidance}`, - }, - ], - }; - } return { content: [ { type: "text" as const, text: `Error reading UI state: ${msg}` }, diff --git a/src/ui-state-offline.test.ts b/src/ui-state-offline.test.ts new file mode 100644 index 00000000..5ee62fbc --- /dev/null +++ b/src/ui-state-offline.test.ts @@ -0,0 +1,86 @@ +import { describe, it, expect, beforeAll, afterAll, afterEach } from "vitest"; +import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "fs"; +import { tmpdir } from "os"; +import { join } from "path"; + +// get_ui_state used to report "Web UI is not running" whenever the Web UI +// process was down, even though the same state it would have served lives on +// disk at paths.uiState. These tests pin the fallback: offline reads should +// come from the state file, and the "no session yet" message should only +// appear when there is truly nothing to read. + +const tmp = mkdtempSync(join(tmpdir(), "podcli-ui-state-offline-")); +const savedHome = process.env.PODCLI_HOME; +const savedData = process.env.PODCLI_DATA; +const savedFetch = globalThis.fetch; + +let createServer: typeof import("./server.js").createServer; +let uiStatePath: string; + +beforeAll(async () => { + process.env.PODCLI_HOME = join(tmp, "home"); + process.env.PODCLI_DATA = join(tmp, "data"); + const { paths } = await import("./config/paths.js"); + uiStatePath = paths.uiState; + mkdirSync(paths.home, { recursive: true }); + ({ createServer } = await import("./server.js")); +}); + +afterAll(() => { + if (savedHome === undefined) delete process.env.PODCLI_HOME; + else process.env.PODCLI_HOME = savedHome; + if (savedData === undefined) delete process.env.PODCLI_DATA; + else process.env.PODCLI_DATA = savedData; + rmSync(tmp, { recursive: true, force: true }); +}); + +afterEach(() => { + globalThis.fetch = savedFetch; + rmSync(uiStatePath, { force: true }); +}); + +function getUiStateHandler() { + const server = createServer() as unknown as { + _registeredTools: Record Promise<{ content: { text: string }[] }> }>; + }; + return server._registeredTools["get_ui_state"].handler; +} + +function stubWebUiDown() { + globalThis.fetch = (() => Promise.reject(new Error("fetch failed"))) as typeof fetch; +} + +describe("get_ui_state offline fallback", () => { + it("reads session state from disk when the Web UI is not running", async () => { + writeFileSync( + uiStatePath, + JSON.stringify({ + videoPath: "/videos/ep42.mp4", + phase: "suggesting", + settings: { captionStyle: "hormozi", cropStrategy: "speaker" }, + suggestions: [], + deselectedIndices: [], + transcript: { words: ["a", "b", "c"] }, + }), + ); + stubWebUiDown(); + + const handler = getUiStateHandler(); + const result = await handler({ include_transcript: false }, {}); + const text = result.content[0].text; + + expect(text).toContain("/videos/ep42.mp4"); + expect(text).toContain("Transcript: 3 words"); + expect(text).not.toContain("Web UI is not running"); + }); + + it("falls back to start-from-scratch guidance when there is no state anywhere", async () => { + stubWebUiDown(); + + const handler = getUiStateHandler(); + const result = await handler({ include_transcript: false }, {}); + const text = result.content[0].text; + + expect(text).toContain("Start from scratch"); + }); +}); From 3b9ab0ca4f154a7c325adebe067d8909ec9c52f7 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:27:07 +0400 Subject: [PATCH 011/110] Use medians, not means, for performance-learnings buckets A single viral or flop clip was swinging a small bucket's mean retention/CTR/ views far past what's typical. Switch _agg to statistics.median, and drop buckets under 4 clips entirely instead of reporting a number nobody should trust yet. --- .../integrations/youtube/learnings.py | 17 ++++- tests/test_learnings.py | 71 +++++++++++++++++++ 2 files changed, 85 insertions(+), 3 deletions(-) create mode 100644 tests/test_learnings.py diff --git a/backend/services/integrations/youtube/learnings.py b/backend/services/integrations/youtube/learnings.py index 049746eb..fba0d9f5 100644 --- a/backend/services/integrations/youtube/learnings.py +++ b/backend/services/integrations/youtube/learnings.py @@ -8,10 +8,15 @@ from __future__ import annotations import os +import statistics from collections import defaultdict from datetime import datetime, timezone from typing import Any, Callable, Optional +# Below this, one viral or one flop clip swings the bucket's number enough +# that it reads as a trend. Hide it until there's enough data to trust. +MIN_BUCKET_SIZE = 4 + from config.paths import paths from services.clips_history import load_clips_history @@ -34,18 +39,24 @@ def _has_perf(c: dict) -> bool: def _agg(clips: list[dict], key: Callable[[dict], Any]) -> list[dict]: + """Group clips by `key` and report the median (not mean) of each metric, + since one viral or one flop clip would otherwise swing a small bucket's + average far past what's typical. Buckets under MIN_BUCKET_SIZE are + dropped — too little data to call it a trend.""" groups: dict[str, list[dict]] = defaultdict(list) for c in clips: groups[key(c) or "—"].append(c) out = [] for k, cs in groups.items(): + if len(cs) < MIN_BUCKET_SIZE: + continue vals = lambda f: [c["metrics"][f] for c in cs if (c.get("metrics") or {}).get(f) is not None] ret, ctr, views = vals("retention"), vals("ctr"), vals("views") out.append({ "key": k, "n": len(cs), - "ret": round(sum(ret) / len(ret), 1) if ret else None, - "ctr": round(sum(ctr) / len(ctr), 1) if ctr else None, - "views": int(sum(views) / len(views)) if views else None, + "ret": round(statistics.median(ret), 1) if ret else None, + "ctr": round(statistics.median(ctr), 1) if ctr else None, + "views": int(statistics.median(views)) if views else None, }) return out diff --git a/tests/test_learnings.py b/tests/test_learnings.py new file mode 100644 index 00000000..ea226ddd --- /dev/null +++ b/tests/test_learnings.py @@ -0,0 +1,71 @@ +"""Tests for the performance-learnings digest: median aggregation and the +small-bucket suppression that keeps a single viral or flop clip from reading +as a trend.""" + +import os +import sys +import tempfile +import unittest +from unittest import mock + +ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..")) +BACKEND_ROOT = os.path.join(ROOT, "backend") +if BACKEND_ROOT not in sys.path: + sys.path.insert(0, BACKEND_ROOT) + +from services.integrations.youtube import learnings + + +def _clip(content_type, retention, views=1000, ctr=5.0, caption_style="hormozi", duration=30): + return { + "title": f"clip-{content_type}-{retention}", + "content_type": content_type, + "caption_style": caption_style, + "duration": duration, + "metrics": {"retention": retention, "ctr": ctr, "views": views}, + } + + +class AggTests(unittest.TestCase): + def test_uses_median_not_mean(self): + # One outlier (90) would drag a mean way up; the median should stay + # anchored to the typical clip. + clips = [_clip("story", r) for r in [20, 22, 21, 90]] + out = learnings._agg(clips, lambda c: c["content_type"]) + self.assertEqual(len(out), 1) + self.assertEqual(out[0]["ret"], 21.5) + self.assertEqual(out[0]["n"], 4) + + def test_suppresses_buckets_under_minimum_size(self): + clips = [_clip("story", 20), _clip("story", 22), _clip("story", 24)] + out = learnings._agg(clips, lambda c: c["content_type"]) + self.assertEqual(out, []) + + def test_keeps_bucket_at_minimum_size(self): + clips = [_clip("story", r) for r in [10, 20, 30, 40]] + out = learnings._agg(clips, lambda c: c["content_type"]) + self.assertEqual(len(out), 1) + self.assertEqual(out[0]["n"], 4) + + +class WriteLearningsTests(unittest.TestCase): + def test_small_bucket_is_dropped_from_output(self): + # "video" has only 3 clips (below MIN_BUCKET_SIZE) and must not show + # up in the rendered "By moment type" section. + clips = [_clip("story", r) for r in [10, 20, 30, 40]] + [ + _clip("video", r) for r in [60, 70, 80] + ] + with tempfile.TemporaryDirectory() as tmp: + with mock.patch.object(learnings, "load_clips_history", return_value=clips), mock.patch.object( + learnings, "paths", {"knowledge": tmp} + ): + path = learnings.write_learnings(min_clips=3) + self.assertIsNotNone(path) + with open(path, encoding="utf-8") as f: + content = f.read() + self.assertIn("story", content) + self.assertNotIn("video:", content) + + +if __name__ == "__main__": + unittest.main() From a9245e50496893faa9e83024531e954bd3fe05e0 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:28:03 +0400 Subject: [PATCH 012/110] Strip ASS override tags on SRT/VTT import and escape control characters in rendered captions --- backend/services/caption_renderer.py | 28 +++++++++++-- backend/services/transcript_parser.py | 25 +++++++----- tests/test_caption_renderer.py | 59 +++++++++++++++++++++++++++ tests/test_transcript_parser.py | 55 +++++++++++++++++++++++++ 4 files changed, 153 insertions(+), 14 deletions(-) diff --git a/backend/services/caption_renderer.py b/backend/services/caption_renderer.py index 8f44d3a9..60141c54 100644 --- a/backend/services/caption_renderer.py +++ b/backend/services/caption_renderer.py @@ -14,6 +14,26 @@ from utils.timing_utils import seconds_to_ass +def _sanitize_ass_text(text: str) -> str: + """Neutralize characters that mean something to the ASS/libass parser + when they show up in transcribed word text instead of our own markup. + + A literal backslash isn't escapable in the plain-text part of a + Dialogue line: libass still reads \\N, \\n and \\h out of it (that's how + an uppercase transform can turn a stray "\\n" into a forced \\N + linebreak), and a literal "{" opens a new override block early, letting + anything after it (including a following "}") be read as ASS tags + instead of rendered as text. There's no escape sequence for any of + these in plain text, so swap in a full-width look-alike glyph that + renders the same character shape without being special to the parser. + """ + return ( + text.replace("\\", "\") + .replace("{", "{") + .replace("}", "}") + ) + + def generate_ass_header(style: dict, play_res_x: int = 1080, play_res_y: int = 1920) -> str: """Generate the ASS file header with style definitions.""" bold_val = -1 if style["bold"] else 0 @@ -207,7 +227,7 @@ def _render_hormozi(words: list[dict], style: dict, offset: float) -> str: parts = [] for w in chunk: duration_cs = int((w["end"] - w["start"]) * 100) - text = w["word"].upper() if uppercase else w["word"] + text = _sanitize_ass_text(w["word"].upper() if uppercase else w["word"]) parts.append(f"{{\\kf{duration_cs}}}{text}") # \c = active (filled) color, \2c = inactive (unfilled) color @@ -246,7 +266,7 @@ def _render_karaoke(words: list[dict], style: dict, offset: float) -> str: parts = [] for w in sentence: duration_cs = int((w["end"] - w["start"]) * 100) - text = w["word"] + text = _sanitize_ass_text(w["word"]) parts.append(f"{{\\kf{duration_cs}}}{text}") line_text = " ".join(parts) @@ -279,7 +299,7 @@ def _render_subtle(words: list[dict], style: dict, offset: float) -> str: line_start = max(0, line_words[0]["start"] - offset) line_end = _hold_through_gap(lines, idx, max(0, line_words[-1]["end"] - offset), offset) - line_text = " ".join(w["word"] for w in line_words) + line_text = " ".join(_sanitize_ass_text(w["word"]) for w in line_words) start_ts = seconds_to_ass(line_start) end_ts = seconds_to_ass(line_end) @@ -544,7 +564,7 @@ def _render_branded(words: list[dict], style: dict, offset: float) -> str: # Normalize casing normalized = [] for j, w in enumerate(chunk): - text = _normalize_case(w["word"]) + text = _sanitize_ass_text(_normalize_case(w["word"])) if j == 0: text = text[0].upper() + text[1:] if len(text) > 1 else text.upper() normalized.append(text) diff --git a/backend/services/transcript_parser.py b/backend/services/transcript_parser.py index e34fa4c7..f4bccd01 100644 --- a/backend/services/transcript_parser.py +++ b/backend/services/transcript_parser.py @@ -14,6 +14,16 @@ from typing import List, Dict, Any, Optional +def _clean_caption_text(text: str) -> str: + """Strip markup that isn't spoken text: HTML-like tags and ASS/SSA + override blocks such as {\\an8} that some SRT/VTT exports carry over. + Left in place, they get imported as literal words. + """ + text = re.sub(r'<[^>]+>', '', text) + text = re.sub(r'\{[^}]*\}', '', text) + return re.sub(r'\s+', ' ', text).strip() + + def parse_srt_timestamp(ts: str) -> float: """Convert SRT timestamp HH:MM:SS,mmm to seconds.""" ts = ts.strip().replace(",", ".") @@ -71,8 +81,7 @@ def parse_srt(text: str, total_duration: Optional[float] = None, time_adjust: fl # End of block text_content = " ".join(current_text_lines).strip() # Strip HTML-like tags - text_content = re.sub(r'<[^>]+>', '', text_content) - text_content = re.sub(r'\s+', ' ', text_content).strip() + text_content = _clean_caption_text(text_content) if text_content: blocks.append({ "start": current_start, @@ -86,8 +95,7 @@ def parse_srt(text: str, total_duration: Optional[float] = None, time_adjust: fl # Handle last block if file doesn't end with blank line if state == "text" and current_text_lines: text_content = " ".join(current_text_lines).strip() - text_content = re.sub(r'<[^>]+>', '', text_content) - text_content = re.sub(r'\s+', ' ', text_content).strip() + text_content = _clean_caption_text(text_content) if text_content: blocks.append({ "start": current_start, @@ -138,8 +146,7 @@ def parse_vtt(text: str, total_duration: Optional[float] = None, time_adjust: fl # Save previous block if in_text and current_text_lines: text_content = " ".join(current_text_lines).strip() - text_content = re.sub(r'<[^>]+>', '', text_content) - text_content = re.sub(r'\s+', ' ', text_content).strip() + text_content = _clean_caption_text(text_content) if text_content: blocks.append({ "start": current_start, @@ -156,8 +163,7 @@ def parse_vtt(text: str, total_duration: Optional[float] = None, time_adjust: fl if in_text: if line_stripped == "": text_content = " ".join(current_text_lines).strip() - text_content = re.sub(r'<[^>]+>', '', text_content) - text_content = re.sub(r'\s+', ' ', text_content).strip() + text_content = _clean_caption_text(text_content) if text_content: blocks.append({ "start": current_start, @@ -173,8 +179,7 @@ def parse_vtt(text: str, total_duration: Optional[float] = None, time_adjust: fl # Handle last block if in_text and current_text_lines: text_content = " ".join(current_text_lines).strip() - text_content = re.sub(r'<[^>]+>', '', text_content) - text_content = re.sub(r'\s+', ' ', text_content).strip() + text_content = _clean_caption_text(text_content) if text_content: blocks.append({ "start": current_start, diff --git a/tests/test_caption_renderer.py b/tests/test_caption_renderer.py index fa130c4a..4d0a9e41 100644 --- a/tests/test_caption_renderer.py +++ b/tests/test_caption_renderer.py @@ -98,5 +98,64 @@ def test_text_events_span_the_whole_chunk(self): self.assertEqual(len(ends), 1) +class AssTextSanitizationTests(unittest.TestCase): + """Words containing backslash/brace characters must not corrupt the ASS + override-block syntax or get read as \\N/\\n/\\h control codes.""" + + def _residual_text(self, content: str) -> str: + # Strip the override blocks we generate ourselves; anything left + # over came from word text and must not contain raw \, { or }. + import re + stripped = re.sub(r"\{\\[^}]*\}", "", content) + return stripped + + def test_sanitize_replaces_control_characters(self): + self.assertEqual(cr._sanitize_ass_text("C:\\path"), "C:\path") + self.assertEqual(cr._sanitize_ass_text("{weird}"), "{weird}") + self.assertEqual(cr._sanitize_ass_text(r"a\nb"), "a\nb") + + def test_render_hormozi_escapes_backslash_and_braces(self): + style = get_style("hormozi") + words = _flowing_words(["a\\nb", "{curly}"]) + content = cr._render_hormozi(words, style, 0.0) + residual = self._residual_text(content) + self.assertNotIn("\\n", residual) + self.assertNotIn("{curly}", content) + self.assertIn("\", content) + # hormozi uppercases word text; the braces are swapped regardless of case. + self.assertIn("{CURLY}" if style["uppercase"] else "{curly}", content) + + def test_render_karaoke_escapes_backslash_and_braces(self): + style = get_style("karaoke") + words = _flowing_words(["a\\nb", "{curly}"]) + content = cr._render_karaoke(words, style, 0.0) + self.assertIn("\", content) + self.assertIn("{curly}", content) + + def test_render_subtle_escapes_backslash_and_braces(self): + style = get_style("subtle") + words = _flowing_words(["a\\nb", "{curly}"]) + content = cr._render_subtle(words, style, 0.0) + self.assertIn("\", content) + self.assertIn("{curly}", content) + + def test_render_branded_escapes_backslash_and_braces(self): + style = get_style("branded") + words = _flowing_words(["a\\nb", "{curly}"]) + content = cr._render_branded(words, style, 0.0) + self.assertIn("\", content) + self.assertIn("{curly}", content) + + def test_uppercase_does_not_turn_escaped_backslash_into_linebreak(self): + # Before the fix, a literal "\n" surviving into hormozi's uppercase + # transform became the literal two characters "\" + "N", which + # libass reads as a forced line break control code. + style = get_style("hormozi") + words = _flowing_words(["a\\nb"]) + content = cr._render_hormozi(words, style, 0.0) + self.assertNotIn("\\N", content) + self.assertNotIn("\\n", content) + + if __name__ == "__main__": unittest.main() diff --git a/tests/test_transcript_parser.py b/tests/test_transcript_parser.py index 6c9f3878..31bfa7be 100644 --- a/tests/test_transcript_parser.py +++ b/tests/test_transcript_parser.py @@ -136,5 +136,60 @@ def test_hh_mm_ss_headers(self): self.assertGreaterEqual(result["words"][0]["start"], 3600.0 - 1.0) +class CaptionMarkupStrippingTests(unittest.TestCase): + """SRT/VTT override tags like {\\an8} aren't spoken text and must not + end up imported as literal words in the transcript.""" + + def test_parse_srt_strips_ass_override_blocks(self): + raw = ( + "1\n" + "00:00:01,000 --> 00:00:04,000\n" + "{\\an8}Hello this is text\n" + ) + result = tp.parse_srt(raw) + self.assertNotIn("error", result) + self.assertEqual(result["segments"][0]["text"], "Hello this is text") + self.assertNotIn("an8", result["transcript"]) + self.assertNotIn("{", result["transcript"]) + + def test_parse_srt_strips_html_tags(self): + raw = ( + "1\n" + "00:00:01,000 --> 00:00:04,000\n" + "Hello world\n" + ) + result = tp.parse_srt(raw) + self.assertEqual(result["segments"][0]["text"], "Hello world") + + def test_parse_srt_strips_override_block_in_trailing_block(self): + # No trailing blank line, so this exercises the "last block" path. + raw = ( + "1\n" + "00:00:01,000 --> 00:00:04,000\n" + "{\\an8}No trailing blank line" + ) + result = tp.parse_srt(raw) + self.assertEqual(result["segments"][0]["text"], "No trailing blank line") + + def test_parse_vtt_strips_ass_override_blocks(self): + raw = ( + "WEBVTT\n\n" + "00:00:01.000 --> 00:00:04.000\n" + "{\\an8}Hello this is text\n" + ) + result = tp.parse_vtt(raw) + self.assertNotIn("error", result) + self.assertEqual(result["segments"][0]["text"], "Hello this is text") + + def test_parse_vtt_strips_override_block_in_trailing_block(self): + raw = ( + "WEBVTT\n\n" + "00:00:01.000 --> 00:00:04.000\n" + "{\\an8}No trailing blank line" + ) + result = tp.parse_vtt(raw) + self.assertEqual(result["segments"][0]["text"], "No trailing blank line") + + if __name__ == "__main__": unittest.main() From 5f2f78d4ff4d878726bb8f4cc8374cbd06a83be5 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:29:31 +0400 Subject: [PATCH 013/110] Merge multi-word corrections across consecutive caption words --- backend/services/corrections.py | 74 ++++++++++++++++++++++++++++++++- tests/test_corrections.py | 69 ++++++++++++++++++++++++++++++ 2 files changed, 142 insertions(+), 1 deletion(-) diff --git a/backend/services/corrections.py b/backend/services/corrections.py index 959c2fc1..2dda8f43 100644 --- a/backend/services/corrections.py +++ b/backend/services/corrections.py @@ -73,6 +73,68 @@ def _replace_match(match: re.Match, corrections: dict[str, str]) -> str: return replacement if replacement is not None else matched +def _strip_for_match(text: str) -> str: + return text.strip(".,!?;:\"'()-") + + +def _merge_multiword_corrections(words: list[dict], corrections: dict[str, str]) -> list[dict]: + """ + Merge consecutive words matching a multi-word correction key (e.g. + "open AI" -> "OpenAI") into the corrected word(s). + + Segment text gets multi-word corrections for free from the regex pass + below, but captions are burned from individual words — left split, + "open" and "AI" render as two separate caption words instead of the + fix. The merged word(s) span from the start of the first matched word + to the end of the last one, so caption timing stays continuous. + """ + multiword = {k: v for k, v in corrections.items() if len(k.split()) > 1} + if not multiword or not words: + return words + + # Longest key first so a 3-word phrase wins over a 2-word prefix of it. + key_tokens = sorted( + ((key.split(), value) for key, value in multiword.items()), + key=lambda kv: len(kv[0]), + reverse=True, + ) + + merged: list[dict] = [] + i = 0 + n = len(words) + while i < n: + match = None + for tokens, replacement in key_tokens: + span = len(tokens) + if i + span > n: + continue + candidate = words[i : i + span] + candidate_norm = [_strip_for_match(w.get("word", "")).lower() for w in candidate] + if candidate_norm == [t.lower() for t in tokens]: + match = (candidate, replacement) + break + if match is None: + merged.append(words[i]) + i += 1 + continue + + candidate, replacement = match + first, last = candidate[0], candidate[-1] + repl_words = replacement.split() or [replacement] + span_start = first["start"] + span_end = last["end"] + span_dur = max(0.0, span_end - span_start) + per = span_dur / len(repl_words) + speaker = first.get("speaker") + for idx, rw in enumerate(repl_words): + w_start = span_start + per * idx + w_end = span_end if idx == len(repl_words) - 1 else span_start + per * (idx + 1) + merged.append({"word": rw, "start": w_start, "end": w_end, "speaker": speaker}) + i += len(candidate) + + return merged + + def apply_corrections( words: list[dict], segments: list[dict], @@ -81,7 +143,11 @@ def apply_corrections( Apply corrections to transcript words and segments in-place. Modifies the 'word' field in each word dict and the 'text' field - in each segment dict. Returns the same lists (mutated). + in each segment dict. Multi-word corrections can change the number of + words (several words merge into the correction's word(s)), so the + `words` list itself is replaced in-place via slice assignment — + callers that hold a reference to the original list still see the + update. Returns the same lists (mutated). """ corrections = _load_corrections() if not corrections: @@ -93,6 +159,12 @@ def apply_corrections( replacer = lambda m: _replace_match(m, corrections) + # Merge multi-word corrections first so captions (built from words) read + # the fix the same way the segment text already does. + merged_words = _merge_multiword_corrections(words, corrections) + if merged_words is not words: + words[:] = merged_words + # Fix individual words (strip punctuation for matching, preserve it in output) for w in words: word_text = w.get("word", "") diff --git a/tests/test_corrections.py b/tests/test_corrections.py index 75600381..d0d73519 100644 --- a/tests/test_corrections.py +++ b/tests/test_corrections.py @@ -95,6 +95,75 @@ def test_apply_corrections_word_boundary(self): _, s = corrections.apply_corrections([], segments) self.assertEqual(s[0]["text"], "we maintain A.I. systems") + def test_apply_corrections_merges_multiword_word_run(self): + # "open AI" split across two words must merge into one caption word, + # not just fix the segment text and leave "open"/"AI" as-is. + self._set({"open AI": "OpenAI"}) + words = [ + {"word": "open", "start": 1.0, "end": 1.4, "speaker": "SPEAKER_00"}, + {"word": "AI", "start": 1.4, "end": 1.8, "speaker": "SPEAKER_00"}, + ] + segments = [{"text": "I love open AI"}] + w, s = corrections.apply_corrections(words, segments) + self.assertEqual([x["word"] for x in w], ["OpenAI"]) + self.assertAlmostEqual(w[0]["start"], 1.0) + self.assertAlmostEqual(w[0]["end"], 1.8) + self.assertEqual(w[0]["speaker"], "SPEAKER_00") + self.assertEqual(s[0]["text"], "I love OpenAI") + + def test_apply_corrections_multiword_merge_is_case_insensitive(self): + self._set({"open AI": "OpenAI"}) + words = [ + {"word": "Open", "start": 0.0, "end": 0.5}, + {"word": "ai", "start": 0.5, "end": 1.0}, + ] + w, _ = corrections.apply_corrections(words, []) + self.assertEqual([x["word"] for x in w], ["OpenAI"]) + + def test_apply_corrections_multiword_merge_ignores_punctuation(self): + self._set({"open AI": "OpenAI"}) + words = [ + {"word": "open", "start": 0.0, "end": 0.5}, + {"word": "AI.", "start": 0.5, "end": 1.0}, + ] + w, _ = corrections.apply_corrections(words, []) + self.assertEqual([x["word"] for x in w], ["OpenAI"]) + + def test_apply_corrections_multiword_replacement_splits_time_across_words(self): + # A multi-word replacement distributes the merged span evenly across + # its own word count, rather than collapsing into a single word. + self._set({"open ai": "Open AI"}) + words = [ + {"word": "open", "start": 0.0, "end": 1.0}, + {"word": "ai", "start": 1.0, "end": 2.0}, + ] + w, _ = corrections.apply_corrections(words, []) + self.assertEqual([x["word"] for x in w], ["Open", "AI"]) + self.assertAlmostEqual(w[0]["start"], 0.0) + self.assertAlmostEqual(w[0]["end"], 1.0) + self.assertAlmostEqual(w[1]["start"], 1.0) + self.assertAlmostEqual(w[1]["end"], 2.0) + + def test_apply_corrections_multiword_longest_match_wins_in_words(self): + self._set({"open AI": "OpenAI", "AI": "A.I."}) + words = [ + {"word": "open", "start": 0.0, "end": 0.5}, + {"word": "AI", "start": 0.5, "end": 1.0}, + ] + w, _ = corrections.apply_corrections(words, []) + self.assertEqual([x["word"] for x in w], ["OpenAI"]) + + def test_apply_corrections_leaves_non_matching_words_alone(self): + self._set({"open AI": "OpenAI"}) + words = [ + {"word": "I", "start": 0.0, "end": 0.2}, + {"word": "love", "start": 0.2, "end": 0.5}, + {"word": "open", "start": 0.5, "end": 0.8}, + {"word": "AI", "start": 0.8, "end": 1.0}, + ] + w, _ = corrections.apply_corrections(words, []) + self.assertEqual([x["word"] for x in w], ["I", "love", "OpenAI"]) + def test_apply_corrections_returns_same_list_objects(self): self._set({"Foo": "Bar"}) words = [{"word": "Foo"}] From 7f40ce6c330fff2199e20ccc7858ea56c4497ce8 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:30:44 +0400 Subject: [PATCH 014/110] Fix transcript cache misses from unresolved engine and widen the cache key - Resolve the engine before reading the cache (new resolve_transcribe_engine backend call) instead of reading with the raw unset/requested engine: an unset request that falls back to whisper.cpp was written under 'whispercpp' but read under the bare key, so it never hit. - web-server.ts's cache.set after a background transcribe job also wrote under the request-time engine instead of what transcribe_file actually ran with; it now uses the resolved engine from the result. - Widen the raw JSON cache key to {engine, model, language}; base model and auto/empty language contribute no suffix so existing cache entries keep reading the same way they always have. - Make cache writes atomic (write to a temp file, then rename). --- backend/main.py | 13 +++++++ backend/services/transcription.py | 25 +++++++++++++ src/handlers/transcribe.handler.ts | 24 +++++++++---- src/models/index.ts | 2 +- src/server.ts | 11 +++++- src/services/engine-resolve.test.ts | 37 +++++++++++++++++++ src/services/engine-resolve.ts | 30 ++++++++++++++++ src/services/transcript-cache.test.ts | 41 +++++++++++++++++++++ src/services/transcript-cache.ts | 52 +++++++++++++++++++++++---- src/ui/web-server.ts | 21 +++++++++-- tests/test_transcription_engine.py | 48 +++++++++++++++++++++++++ 11 files changed, 287 insertions(+), 17 deletions(-) create mode 100644 src/services/engine-resolve.test.ts create mode 100644 src/services/engine-resolve.ts diff --git a/backend/main.py b/backend/main.py index a124f3b3..ce3e5749 100644 --- a/backend/main.py +++ b/backend/main.py @@ -67,6 +67,18 @@ def handle_ping(task_id: str, params: dict): emit_result(task_id, "success", data={"message": "pong", "version": VERSION}) +def handle_resolve_transcribe_engine(task_id: str, params: dict): + """Predict transcribe_file's engine resolution without transcribing, so + callers can build the right cache key before deciding whether to run it.""" + from services.transcription import resolve_engine_info + + emit_result( + task_id, + "success", + data=resolve_engine_info(params.get("engine"), params.get("model_size", "base")), + ) + + def handle_transcribe(task_id: str, params: dict): """Transcribe a podcast video/audio file with speaker detection.""" from services.transcription import transcribe_file @@ -1098,6 +1110,7 @@ def handle_run_integration_tool(task_id: str, params: dict): TASK_HANDLERS = { "ping": handle_ping, + "resolve_transcribe_engine": handle_resolve_transcribe_engine, "transcribe": handle_transcribe, "parse_transcript": handle_parse_transcript, "create_clip": handle_create_clip, diff --git a/backend/services/transcription.py b/backend/services/transcription.py index 93637409..3bc48d70 100644 --- a/backend/services/transcription.py +++ b/backend/services/transcription.py @@ -463,6 +463,31 @@ def _attach_speakers_and_faces( return base +def resolve_engine_info(requested: Optional[str], model_size: str = "base") -> dict: + """Predict which engine transcribe_file will actually use, without doing + any transcription work. Callers cache transcripts by engine; the cache + key has to match what transcribe_file resolves to, not the raw request, + or an unset engine writes under one key and reads under another. + + This mirrors the fallback in transcribe_file but checks only `import + whisper` rather than whisper.load_model(), so it can't catch the rarer + case where the import succeeds but loading the model weights fails. That + case still falls back correctly inside transcribe_file itself; it just + means a cache lookup for it can miss once before the result is written + under the engine it actually ran with. + """ + requested = requested if requested is not None else os.environ.get("PODCLI_ENGINE", "") + engine = normalize_engine(requested) + if engine != "whisper-py": + return {"engine": engine, "model_size": model_size} + try: + import whisper # noqa: F401 + except Exception: + if not requested and _whispercpp_ready(model_size): + return {"engine": "whispercpp", "model_size": model_size} + return {"engine": "whisper-py", "model_size": model_size} + + def transcribe_file( file_path: str, model_size: str = "base", diff --git a/src/handlers/transcribe.handler.ts b/src/handlers/transcribe.handler.ts index 4ec89d28..e0eb3529 100644 --- a/src/handlers/transcribe.handler.ts +++ b/src/handlers/transcribe.handler.ts @@ -1,6 +1,7 @@ import { basename } from "path"; import { PythonExecutor } from "../services/python-executor.js"; import { TranscriptCache, hasSpeakerLabels } from "../services/transcript-cache.js"; +import { resolveTranscribeEngine } from "../services/engine-resolve.js"; import { webServerUrl } from "../config/server.js"; import type { TranscriptResult } from "../models/index.js"; @@ -81,13 +82,19 @@ export async function handleTranscribe(input: TranscribeInput): Promise const enableDiarization = input.enable_diarization !== false; const numSpeakers = input.num_speakers; + // Resolve before reading the cache: an unset engine is written under + // whatever transcribe_file actually ran (e.g. "whispercpp" on a native + // install), so reading with the raw unset request always misses. + const resolvedEngine = await resolveTranscribeEngine(executor, engine, modelSize); + const cacheKey = { engine: resolvedEngine, model: modelSize, language }; + // Check cache first. A cached transcript without speakers cannot answer a // request for them, so serving it makes re-transcribing look like a no-op. - const cachedRaw = await cache.get(filePath, engine); + const cachedRaw = await cache.get(filePath, cacheKey); const cached = cachedRaw && enableDiarization && !hasSpeakerLabels(cachedRaw) ? null : cachedRaw; if (cached) { - const packedEngine = cached.engine ?? engine; + const packedEngine = cached.engine ?? resolvedEngine; // Backfill packed view if this cache predates auto-packing. let packed = await cache.getPackedMarkdown(filePath, packedEngine); if (!packed) { @@ -121,11 +128,13 @@ export async function handleTranscribe(input: TranscribeInput): Promise throw new Error("Transcription returned no data"); } const data = result.data; - const resolvedEngine = data.engine; + const actualEngine = data.engine ?? resolvedEngine; - // Cache the raw result - await cache.set(filePath, data, resolvedEngine); - const packed = await cache.getPackedMarkdown(filePath, resolvedEngine); + // Cache the raw result under what it actually ran with, not the prediction + // above — resolveTranscribeEngine can't see a model-load failure that only + // shows up once transcribe_file tries it for real. + await cache.set(filePath, data, { engine: actualEngine, model: modelSize, language }); + const packed = await cache.getPackedMarkdown(filePath, actualEngine); return JSON.stringify({ cached: false, packed_ready: !!packed, ...formatResult(data) }); } @@ -140,6 +149,9 @@ function formatResult(data: TranscriptResult) { return { duration: data.duration, language: data.language, + // Exposed so callers that need to re-read the cache (e.g. the UI state + // push in server.ts) key it the same way this handler just wrote it. + engine: data.engine, word_count: (data.words ?? []).length, segment_count: (data.segments ?? []).length, speakers: data.speakers ?? { num_speakers: 0, speakers: {} }, diff --git a/src/models/index.ts b/src/models/index.ts index e57668a2..9418896e 100644 --- a/src/models/index.ts +++ b/src/models/index.ts @@ -2,7 +2,7 @@ export interface TaskRequest { task_id: string; - task_type: "transcribe" | "parse_transcript" | "create_clip" | "batch_clips" | "analyze_energy" | "detect_highlights" | "manage_reel" | "pack_transcript" | "detect_encoder" | "presets" | "ping" | "suggest_clips" | "find_moment" | "generate_content" | "generate_custom" | "corrections" | "manage_integrations" | "run_integration_tool" | "manage_config" | "manage_env" | "ai_cli_status" | "ai_provider_status" | "analyze_silence" | "render_silence_removed" | "manage_multicam"; + task_type: "transcribe" | "resolve_transcribe_engine" | "parse_transcript" | "create_clip" | "batch_clips" | "analyze_energy" | "detect_highlights" | "manage_reel" | "pack_transcript" | "detect_encoder" | "presets" | "ping" | "suggest_clips" | "find_moment" | "generate_content" | "generate_custom" | "corrections" | "manage_integrations" | "run_integration_tool" | "manage_config" | "manage_env" | "ai_cli_status" | "ai_provider_status" | "analyze_silence" | "render_silence_removed" | "manage_multicam"; params: Record; } diff --git a/src/server.ts b/src/server.ts index 2419bd83..ccb72760 100644 --- a/src/server.ts +++ b/src/server.ts @@ -315,7 +315,16 @@ export function createServer(): McpServer { // on-disk cache — NOT the trimmed MCP response. Without words in UI // state, downstream batch_create_clips can't burn captions. try { - const cached = await transcriptCache.get(file_path, engine); + // handleTranscribe's result carries the engine it actually resolved + // to and cached under; the request-time `engine` can be unset while + // the write landed under "whispercpp", so reading with it misses. + const resolvedEngine = + (JSON.parse(result) as { engine?: string }).engine ?? engine; + const cached = await transcriptCache.get(file_path, { + engine: resolvedEngine, + model: model_size, + language, + }); if (cached) { await uiPing({ videoPath: file_path, diff --git a/src/services/engine-resolve.test.ts b/src/services/engine-resolve.test.ts new file mode 100644 index 00000000..b52fd26e --- /dev/null +++ b/src/services/engine-resolve.test.ts @@ -0,0 +1,37 @@ +import { describe, it, expect, vi } from "vitest"; +import type { PythonExecutor } from "./python-executor.js"; +import { resolveTranscribeEngine } from "./engine-resolve.js"; + +function fakeExecutor(response: unknown): PythonExecutor { + return { execute: vi.fn().mockResolvedValue(response) } as unknown as PythonExecutor; +} + +describe("resolveTranscribeEngine", () => { + it("returns what the backend predicts it will resolve to", async () => { + const executor = fakeExecutor({ data: { engine: "whispercpp" } }); + const engine = await resolveTranscribeEngine(executor, undefined, "base"); + expect(engine).toBe("whispercpp"); + expect(executor.execute).toHaveBeenCalledWith("resolve_transcribe_engine", { + engine: undefined, + model_size: "base", + }); + }); + + it("passes an explicit engine through for the backend to normalize", async () => { + const executor = fakeExecutor({ data: { engine: "assemblyai" } }); + const engine = await resolveTranscribeEngine(executor, "assemblyai", "base"); + expect(engine).toBe("assemblyai"); + }); + + it("falls back to the raw request if resolution itself fails", async () => { + const executor = { execute: vi.fn().mockRejectedValue(new Error("boom")) } as unknown as PythonExecutor; + const engine = await resolveTranscribeEngine(executor, "whisper-py", "base"); + expect(engine).toBe("whisper-py"); + }); + + it("falls back to whisper-py when both the request and the resolution are empty", async () => { + const executor = { execute: vi.fn().mockRejectedValue(new Error("boom")) } as unknown as PythonExecutor; + const engine = await resolveTranscribeEngine(executor, undefined, "base"); + expect(engine).toBe("whisper-py"); + }); +}); diff --git a/src/services/engine-resolve.ts b/src/services/engine-resolve.ts new file mode 100644 index 00000000..4d8bd916 --- /dev/null +++ b/src/services/engine-resolve.ts @@ -0,0 +1,30 @@ +import { PythonExecutor } from "./python-executor.js"; + +/** + * Predicts which engine backend/services/transcription.py's transcribe_file + * will actually use for an unset/auto request, without transcribing anything. + * + * The transcript cache is keyed by engine. Reading the cache with the raw + * request (e.g. undefined, meaning "whisper-py unless this install can't run + * it") instead of what transcribe_file resolves to always misses on a native + * install, because the write afterward lands under "whispercpp" — the key + * the read never looked at. + */ +export async function resolveTranscribeEngine( + executor: PythonExecutor, + engine: string | undefined, + modelSize: string +): Promise { + try { + const result = await executor.execute<{ engine: string }>("resolve_transcribe_engine", { + engine, + model_size: modelSize, + }); + return result.data?.engine ?? engine ?? "whisper-py"; + } catch { + // Resolution is a best-effort cache-key optimization; a failure here + // should fall through to a normal (possibly-missed) cache lookup rather + // than block transcription. + return engine ?? "whisper-py"; + } +} diff --git a/src/services/transcript-cache.test.ts b/src/services/transcript-cache.test.ts index 990983b6..1970caee 100644 --- a/src/services/transcript-cache.test.ts +++ b/src/services/transcript-cache.test.ts @@ -84,6 +84,47 @@ describe("TranscriptCache", () => { writeFileSync(join(tmp, "cache", "transcripts", `${hash}.json`), "this is not json"); expect(await cache.get(file)).toBeNull(); }); + + it("keys the raw cache by engine, model and language, not engine alone", async () => { + const file = makeFakeVideo("multi-key.mp4", "same media, different requests"); + await cache.set(file, { ...fakeTranscript, transcript: "base-auto" }, { + engine: "whispercpp", + }); + await cache.set(file, { ...fakeTranscript, transcript: "small-ka" }, { + engine: "whispercpp", + model: "small", + language: "ka", + }); + expect((await cache.get(file, { engine: "whispercpp" }))?.transcript).toBe("base-auto"); + expect( + (await cache.get(file, { engine: "whispercpp", model: "small", language: "ka" })) + ?.transcript, + ).toBe("small-ka"); + // A request for a third, never-written combo must miss, not fall back to + // either of the above. + expect( + await cache.get(file, { engine: "whispercpp", model: "medium", language: "fr" }), + ).toBeNull(); + }); + + it("treats base model and auto language as no suffix, for backward compatibility", async () => { + const file = makeFakeVideo("default-key.mp4", "legacy cache shape"); + // A plain string engine (the pre-existing call shape) must land on the + // same key as the equivalent object form with default model/language. + await cache.set(file, { ...fakeTranscript, transcript: "legacy" }, "whispercpp"); + expect( + (await cache.get(file, { engine: "whispercpp", model: "base", language: "auto" })) + ?.transcript, + ).toBe("legacy"); + }); + + it("writes atomically — no temp file left behind, and no partial reads", async () => { + const file = makeFakeVideo("atomic.mp4", "atomic write check"); + await cache.set(file, fakeTranscript); + const { readdirSync } = await import("fs"); + const files = readdirSync(join(tmp, "cache", "transcripts")); + expect(files.some((f) => f.endsWith(".tmp"))).toBe(false); + }); }); diff --git a/src/services/transcript-cache.ts b/src/services/transcript-cache.ts index c9637988..cc5c5a2c 100644 --- a/src/services/transcript-cache.ts +++ b/src/services/transcript-cache.ts @@ -1,10 +1,19 @@ import { createHash } from "crypto"; -import { readFile, writeFile, mkdir } from "fs/promises"; +import { readFile, writeFile, mkdir, rename } from "fs/promises"; import { existsSync } from "fs"; import { join } from "path"; import { paths } from "../config/paths.js"; import type { TranscriptResult } from "../models/index.js"; +/** Cache key parts beyond the file hash. A bare string is shorthand for + * { engine: string }, the call shape every existing caller already used. */ +export interface CacheKeyParts { + engine?: string; + model?: string; + language?: string; +} +export type CacheKey = string | CacheKeyParts; + /** * Caches transcripts by file hash so we don't re-transcribe * the same podcast when creating multiple clips. @@ -86,6 +95,13 @@ export class TranscriptCache { }); } + /** + * Engine-only suffix, matching backend/services/transcript_packer.py's + * engine_cache_suffix(). The packed markdown is written there without + * model/language in its key, so looking it up has to use the same narrower + * key as the write, not the richer {engine, model, language} key the raw + * JSON cache (get/set below) uses. + */ async getFileHashForEngine(filePath: string, engine?: string): Promise { return `${await this.getFileHash(filePath)}${this.engineSuffix(engine)}`; } @@ -101,10 +117,30 @@ export class TranscriptCache { return ""; } - async get(filePath: string, engine?: string): Promise { + /** + * Full cache key suffix: engine + model + language. A plain string argument + * (the pre-existing call shape) is treated as engine-only. + * + * base model and auto/empty language contribute no suffix, so a cache + * written before model/language were tracked — always base, whisper-py (or + * whatever engine was passed), auto-detected language — still reads back + * under the same key. Any other model or language gets its own key instead + * of silently colliding with that implicit default. + */ + private keySuffix(key?: CacheKey): string { + const parts = typeof key === "string" ? { engine: key } : key ?? {}; + const engineSuffix = this.engineSuffix(parts.engine); + const model = (parts.model ?? "").trim().toLowerCase(); + const modelSuffix = model && model !== "base" ? `-m${model}` : ""; + const language = (parts.language ?? "").trim().toLowerCase(); + const languageSuffix = language && language !== "auto" ? `-l${language}` : ""; + return `${engineSuffix}${modelSuffix}${languageSuffix}`; + } + + async get(filePath: string, key?: CacheKey): Promise { try { const hash = await this.getFileHash(filePath); - const cachePath = join(this.cacheDir, `${hash}${this.engineSuffix(engine)}.json`); + const cachePath = join(this.cacheDir, `${hash}${this.keySuffix(key)}.json`); if (!existsSync(cachePath)) return null; @@ -115,11 +151,15 @@ export class TranscriptCache { } } - async set(filePath: string, transcript: TranscriptResult, engine?: string): Promise { + async set(filePath: string, transcript: TranscriptResult, key?: CacheKey): Promise { await this.ensureDir(); const hash = await this.getFileHash(filePath); - const cachePath = join(this.cacheDir, `${hash}${this.engineSuffix(engine)}.json`); - await writeFile(cachePath, JSON.stringify(transcript), "utf-8"); + const cachePath = join(this.cacheDir, `${hash}${this.keySuffix(key)}.json`); + // Write-then-rename: a reader never observes a half-written cache file, + // and a crash mid-write leaves only an orphaned .tmp, not a corrupt entry. + const tmpPath = `${cachePath}.${process.pid}.${Date.now()}.tmp`; + await writeFile(tmpPath, JSON.stringify(transcript), "utf-8"); + await rename(tmpPath, cachePath); } /** diff --git a/src/ui/web-server.ts b/src/ui/web-server.ts index 4a107563..87aa526f 100644 --- a/src/ui/web-server.ts +++ b/src/ui/web-server.ts @@ -32,6 +32,7 @@ import { v4 as uuidv4 } from "uuid"; import { PythonExecutor, terminateProcessTree } from "../services/python-executor.js"; import { hasSpeakerLabels, TranscriptCache } from "../services/transcript-cache.js"; +import { resolveTranscribeEngine } from "../services/engine-resolve.js"; import { FileManager } from "../services/file-manager.js"; import { AssetManager, inferType, safeName } from "../services/asset-manager.js"; import { ClipsHistory } from "../services/clips-history.js"; @@ -1046,9 +1047,15 @@ app.post("/api/transcribe", async (req, res) => { res.status(400).json({ error: "File not found" }); return; } + // Resolve before reading the cache: an unset engine is written under + // whatever transcribe_file actually ran (e.g. "whispercpp" on a native + // install), so reading with the raw unset request always misses. + const resolvedEngine = await resolveTranscribeEngine(executor, engine, model_size); + const cacheKey = { engine: resolvedEngine, model: model_size, language }; + // Check cache first. A cached transcript without speakers cannot answer a // request for them, so serving it makes re-transcribing look like a no-op. - const cachedRaw = await cache.get(file_path, engine); + const cachedRaw = await cache.get(file_path, cacheKey); const cached = cachedRaw && enable_diarization && !hasSpeakerLabels(cachedRaw) ? null : cachedRaw; if (cached) { @@ -1113,9 +1120,17 @@ app.post("/api/transcribe", async (req, res) => { // forever: the job id lived only in the tab that started it, and nothing // else announces that the transcript landed. broadcastSSE("state-sync", uiState); - // Cache it + // Cache it under the engine it actually ran with — a fresh resolution + // (or the pre-transcribe request) can differ from what transcribe_file + // fell back to once it tried loading the model for real. try { - await cache.set(file_path, result.data as unknown as TranscriptResult, engine); + const actualEngine = + (result.data as unknown as TranscriptResult | undefined)?.engine ?? resolvedEngine; + await cache.set(file_path, result.data as unknown as TranscriptResult, { + engine: actualEngine, + model: model_size, + language, + }); } catch (err) { log.warn("Failed to cache transcript", { file_path, err: errMsg(err) }); } diff --git a/tests/test_transcription_engine.py b/tests/test_transcription_engine.py index edd565ae..d882e5bc 100644 --- a/tests/test_transcription_engine.py +++ b/tests/test_transcription_engine.py @@ -80,6 +80,54 @@ def test_no_fallback_when_whispercpp_unavailable(self): tr.transcribe_file(self._tmp.name, model_size="base", enable_diarization=False) +class ResolveEngineInfoTests(unittest.TestCase): + """resolve_engine_info predicts transcribe_file's engine choice so a + caller can build a matching cache key before deciding to transcribe.""" + + def setUp(self): + self._had_whisper = sys.modules.get("whisper", "__absent__") + self._orig_ready = tr._whispercpp_ready + self._saved_engine = os.environ.pop("PODCLI_ENGINE", None) + + def tearDown(self): + if self._had_whisper == "__absent__": + sys.modules.pop("whisper", None) + else: + sys.modules["whisper"] = self._had_whisper + tr._whispercpp_ready = self._orig_ready + if self._saved_engine is None: + os.environ.pop("PODCLI_ENGINE", None) + else: + os.environ["PODCLI_ENGINE"] = self._saved_engine + + def test_explicit_whispercpp_resolves_as_is(self): + self.assertEqual(tr.resolve_engine_info("whispercpp")["engine"], "whispercpp") + + def test_explicit_assemblyai_resolves_as_is(self): + self.assertEqual(tr.resolve_engine_info("assemblyai")["engine"], "assemblyai") + + def test_unset_falls_back_to_whispercpp_on_native_install(self): + sys.modules["whisper"] = None # simulate "import whisper" failing + tr._whispercpp_ready = lambda size: True + self.assertEqual(tr.resolve_engine_info(None)["engine"], "whispercpp") + + def test_unset_stays_whisper_py_when_whisper_importable(self): + sys.modules.pop("whisper", None) + try: + import whisper # noqa: F401 + except Exception: + self.skipTest("openai-whisper not installed in this environment") + self.assertEqual(tr.resolve_engine_info(None)["engine"], "whisper-py") + + def test_explicit_whisper_py_request_never_falls_back(self): + # Fallback only applies to an unset request; an explicit whisper-py + # ask should resolve as whisper-py even on a native install, matching + # transcribe_file which raises instead of silently substituting. + sys.modules["whisper"] = None + tr._whispercpp_ready = lambda size: True + self.assertEqual(tr.resolve_engine_info("whisper-py")["engine"], "whisper-py") + + class WhisperCppModelAliasTests(unittest.TestCase): """"large" alone doesn't name a real ggml file upstream (v1/v2/v3/v3-turbo are separate downloads); provisioning always fetches large-v3, so the From f8487c011813c6ee65daa131ea3345560a0ebab0 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:31:35 +0400 Subject: [PATCH 015/110] Add a review sync status for fits that look clean but may be wrong --- backend/services/multicam.py | 35 +++++++++++++++++++++++++++-- backend/services/multicam_signal.py | 33 ++++++++++++++++++--------- src/ui/client/multicam-sync.tsx | 7 ++++++ src/ui/client/multicam-types.ts | 4 +++- tests/test_multicam.py | 30 +++++++++++++++++++++++++ tests/test_multicam_signal.py | 22 ++++++++++++++++++ 6 files changed, 117 insertions(+), 14 deletions(-) diff --git a/backend/services/multicam.py b/backend/services/multicam.py index bb77cf5f..d85b6b0f 100644 --- a/backend/services/multicam.py +++ b/backend/services/multicam.py @@ -1015,13 +1015,44 @@ def _match(ref: np.ndarray, ref_env: np.ndarray, src: np.ndarray) -> tuple[Optio "status": "rough", "score": round(match.score, 1), "message": "Synced to within 10 ms; fine alignment found no clear speech.", } - return fit, { - "status": "ok", + overlap_seconds = min(len(ref), len(src)) / SYNC_RATE + reasons = _sync_review_reasons(fit, match, overlap_seconds) + report = { + "status": "review" if reasons else "ok", "score": round(match.score, 1), "checkpoints": fit.checkpoints, "residual_ms": round(fit.residual_ms, 1), + "residual_all_ms": round(fit.residual_all_ms, 1), "drift_ppm": round((fit.speed - 1.0) * 1e6, 1), } + if reasons: + report["reasons"] = reasons + report["message"] = "Sync may be off: " + "; ".join(reasons) + ". Check it before rendering." + return fit, report + + +def _sync_review_reasons(fit: sig.ClockFit, match: sig.CoarseMatch, overlap_seconds: float) -> list[str]: + """Why a fit that otherwise looks like a clean sync should get a second look. + + A fit can report a tidy inlier residual while still being wrong: outliers + get dropped before the residual is measured, drift can be forced back to + speed 1 when it looked implausible, or there just weren't enough + checkpoints to trust a long overlap's drift estimate. + """ + reasons = [] + worst_residual_ms = max(fit.residual_ms, fit.residual_all_ms) + if worst_residual_ms > 30.0: + reasons.append(f"residual {worst_residual_ms:.0f} ms") + if fit.speed_fallback: + reasons.append("drift fit was implausible, so speed was forced back to 1.0") + dropped = fit.total_checkpoints - fit.checkpoints + if fit.total_checkpoints and dropped / fit.total_checkpoints > 0.4: + reasons.append(f"dropped {dropped} of {fit.total_checkpoints} checkpoints as outliers") + if overlap_seconds > 600.0 and fit.checkpoints < 3: + reasons.append("fewer than 3 checkpoints over a 10+ minute overlap") + if match.peak_ratio > 0.5: + reasons.append("weak correlation peak") + return reasons def sync_session( diff --git a/backend/services/multicam_signal.py b/backend/services/multicam_signal.py index 22a47f9b..c9db9a73 100644 --- a/backend/services/multicam_signal.py +++ b/backend/services/multicam_signal.py @@ -116,8 +116,11 @@ def gcc_phat(ref: np.ndarray, src: np.ndarray, sample_rate: int, max_shift_secon class ClockFit: offset: float speed: float - residual_ms: float - checkpoints: int + residual_ms: float # worst residual among the surviving (inlier) checkpoints + checkpoints: int # surviving checkpoint count + total_checkpoints: int = 0 # checkpoints offered to the fit, before outliers were dropped + residual_all_ms: float = 0.0 # worst residual against every checkpoint, outliers included + speed_fallback: bool = False # drift looked implausible, so speed was forced back to 1.0 def fit_clock(points: list[tuple[float, float, float]], *, max_residual: float = 0.02) -> Optional[ClockFit]: @@ -130,7 +133,8 @@ def fit_clock(points: list[tuple[float, float, float]], *, max_residual: float = """ if not points: return None - pts = np.array(points, dtype=np.float64) + all_pts = np.array(points, dtype=np.float64) + total = len(all_pts) def solve(p: np.ndarray) -> tuple[float, float]: if len(p) < 2 or np.ptp(p[:, 0]) < 1.0: @@ -142,25 +146,32 @@ def solve(p: np.ndarray) -> tuple[float, float]: # Theil-Sen seeds the outlier test; least squares alone lets one bad # checkpoint drag the line far enough that good points look bad too. slopes = [ - (pts[j, 1] - pts[i, 1]) / (pts[j, 0] - pts[i, 0]) - for i in range(len(pts)) for j in range(i + 1, len(pts)) - if pts[j, 0] - pts[i, 0] >= 1.0 + (all_pts[j, 1] - all_pts[i, 1]) / (all_pts[j, 0] - all_pts[i, 0]) + for i in range(total) for j in range(i + 1, total) + if all_pts[j, 0] - all_pts[i, 0] >= 1.0 ] speed = float(np.median(slopes)) if slopes else 1.0 - offset = float(np.median(pts[:, 1] - speed * pts[:, 0])) - keep = np.abs(pts[:, 1] - (offset + speed * pts[:, 0])) <= max_residual - if keep.any(): - pts = pts[keep] + offset = float(np.median(all_pts[:, 1] - speed * all_pts[:, 0])) + keep = np.abs(all_pts[:, 1] - (offset + speed * all_pts[:, 0])) <= max_residual + pts = all_pts[keep] if keep.any() else all_pts offset, speed = solve(pts) residual = np.abs(pts[:, 1] - (offset + speed * pts[:, 0])) - if abs(speed - 1.0) > 1e-3: + speed_fallback = abs(speed - 1.0) > 1e-3 + if speed_fallback: offset, speed = float(np.median(pts[:, 1] - pts[:, 0])), 1.0 residual = np.abs(pts[:, 1] - (offset + pts[:, 0])) + # Residual over every checkpoint offered to the fit, outliers included: a + # fit that dropped its way to a clean-looking inlier residual can still be + # wrong if most of the checkpoints it threw out disagreed with the line. + residual_all = np.abs(all_pts[:, 1] - (offset + speed * all_pts[:, 0])) return ClockFit( offset=offset, speed=speed, residual_ms=float(residual.max()) * 1000.0, checkpoints=len(pts), + total_checkpoints=total, + residual_all_ms=float(residual_all.max()) * 1000.0, + speed_fallback=speed_fallback, ) diff --git a/src/ui/client/multicam-sync.tsx b/src/ui/client/multicam-sync.tsx index 957f151c..73db9b38 100644 --- a/src/ui/client/multicam-sync.tsx +++ b/src/ui/client/multicam-sync.tsx @@ -6,6 +6,7 @@ import { whoLabel } from "./multicam-types"; const STATUS: Record = { reference: { label: "Reference", color: "var(--green)", pill: "pill-green" }, ok: { label: "Synced", color: "var(--green)", pill: "pill-green" }, + review: { label: "Check sync", color: "var(--amber)", pill: "pill-amber" }, rough: { label: "Rough sync", color: "var(--amber)", pill: "pill-amber" }, manual: { label: "Set by hand", color: "var(--blue)", pill: "pill-sky" }, assumed: { label: "Starts with the others", color: "var(--amber)", pill: "pill-amber" }, @@ -100,6 +101,12 @@ function SyncRow({ source, people, onEdit, disabled }: { source: McSource; peopl {source.sync.message || "Sync this file, or type where it starts."} )} {source.sync.status === "assumed" && {source.sync.message}} + {source.sync.status === "review" && ( + + Residual {Math.round(Math.max(source.sync.residual_ms ?? 0, source.sync.residual_all_ms ?? 0))} ms + {source.sync.reasons?.length ? ` (${source.sync.reasons.join(", ")})` : ""} + + )}
{source.offset === null ? ( diff --git a/src/ui/client/multicam-types.ts b/src/ui/client/multicam-types.ts index cfaa5e1f..442d5722 100644 --- a/src/ui/client/multicam-types.ts +++ b/src/ui/client/multicam-types.ts @@ -9,15 +9,17 @@ export interface McPerson { } export type McRole = "camera" | "mic" | "ignore"; -export type McSyncStatus = "reference" | "ok" | "rough" | "failed" | "manual" | "assumed"; +export type McSyncStatus = "reference" | "ok" | "review" | "rough" | "failed" | "manual" | "assumed"; export interface McSourceSync { status?: McSyncStatus; score?: number; checkpoints?: number; residual_ms?: number; + residual_all_ms?: number; drift_ppm?: number; message?: string; + reasons?: string[]; } export interface McSource { diff --git a/tests/test_multicam.py b/tests/test_multicam.py index 21a41fa2..2c2c07e3 100644 --- a/tests/test_multicam.py +++ b/tests/test_multicam.py @@ -271,6 +271,36 @@ def test_sync_measures_clock_drift(sandbox): assert abs(moved.offset - 4.0) < 0.005 +def test_sync_review_reasons_flag_a_fit_that_looks_clean_but_isnt(): + import services.multicam_signal as sig + + clean = sig.ClockFit(offset=0.0, speed=1.0, residual_ms=5.0, checkpoints=5, total_checkpoints=5, + residual_all_ms=5.0, speed_fallback=False) + good_match = sig.CoarseMatch(lag_seconds=0.0, score=30.0, peak_ratio=0.1) + assert mc._sync_review_reasons(clean, good_match, overlap_seconds=1200.0) == [] + + # Speed was forced back to 1.0, and the full-checkpoint residual (500 ms) + # is nothing like the tidy inlier residual (5 ms) the caller would see if + # it only looked at residual_ms. + bad_fallback = sig.ClockFit(offset=0.0, speed=1.0, residual_ms=5.0, checkpoints=2, total_checkpoints=2, + residual_all_ms=500.0, speed_fallback=True) + reasons = mc._sync_review_reasons(bad_fallback, good_match, overlap_seconds=60.0) + assert any("residual" in r for r in reasons) + assert any("implausible" in r for r in reasons) + + # Too few checkpoints over a long overlap, and most checkpoints dropped. + sparse = sig.ClockFit(offset=0.0, speed=1.0, residual_ms=1.0, checkpoints=1, total_checkpoints=6, + residual_all_ms=1.0, speed_fallback=False) + reasons = mc._sync_review_reasons(sparse, good_match, overlap_seconds=900.0) + assert any("checkpoints" in r and "outliers" in r for r in reasons) + assert any("10+ minute overlap" in r for r in reasons) + + # A weak correlation peak alone should trigger review even if the fit is tidy. + weak_match = sig.CoarseMatch(lag_seconds=0.0, score=30.0, peak_ratio=0.7) + reasons = mc._sync_review_reasons(clean, weak_match, overlap_seconds=60.0) + assert any("correlation" in r for r in reasons) + + @pytest.mark.skipif(not shutil.which("ffmpeg"), reason="ffmpeg not installed") def test_shared_audio_falls_back_to_diarization(sandbox, monkeypatch): from services import speaker_detection diff --git a/tests/test_multicam_signal.py b/tests/test_multicam_signal.py index 300a5bd0..027f74c4 100644 --- a/tests/test_multicam_signal.py +++ b/tests/test_multicam_signal.py @@ -2,6 +2,7 @@ import sys import numpy as np +import pytest ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..")) BACKEND_ROOT = os.path.join(ROOT, "backend") @@ -95,6 +96,27 @@ def test_fit_clock_recovers_drift_and_drops_an_outlier(): def test_fit_clock_rejects_implausible_speed(): fit = fit_clock([(0, 5.0, 1.0), (100, 106.0, 1.0)]) assert fit is not None and fit.speed == 1.0 + # Falling back to speed 1.0 does not make the fit trustworthy: the 1 s + # drift between these two points still shows up as a large residual, and + # callers need that signal to flag the sync for review. + assert fit.speed_fallback is True + assert fit.residual_ms == pytest.approx(500.0, abs=1.0) + assert fit.residual_all_ms == pytest.approx(500.0, abs=1.0) + + +def test_fit_clock_reports_total_checkpoints_and_outlier_residual(): + speed = 1.0 + 80e-6 + pts = [(s, 12.5 + s * speed, 1.0) for s in (60, 1200, 2400, 3600, 4800)] + pts.append((3000, 12.5 + 3000 * speed + 0.4, 1.0)) + fit = fit_clock(pts) + assert fit is not None + assert fit.total_checkpoints == 6 + assert fit.checkpoints == 5 + # The inlier residual is tiny, but the dropped outlier was 0.4 s off: the + # all-checkpoints residual must still surface that. + assert fit.residual_ms < 0.01 + assert fit.residual_all_ms == pytest.approx(400.0, abs=1.0) + assert fit.speed_fallback is False def _two_person_levels(): From 442b74010a36aee446db2100416431642c984ed2 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:33:11 +0400 Subject: [PATCH 016/110] CI: install ffmpeg for Python tests, exclude dist from vitest, add release-hygiene check - ci.yml python job now installs ffmpeg on all three OSes (same per-OS steps nightly.yml already uses). Without it, every test that shells out to ffmpeg/ffprobe was silently skipped, about 22 tests on every run. - test_ai_fallback.py's shell-lookup-fallback test now redirects HOME to a temp dir before calling _find_cli: without it, a real ~/.local/bin/claude on the machine running the test wins before the shell-lookup mock is ever exercised, so the test could pass for the wrong reason. - vitest.config.ts excludes dist/, since `npm run build` compiles src/**/*.test.ts into dist/**/*.test.js and vitest would otherwise run every test twice against stale output. - add scripts/release-hygiene.mjs: fails on tracked audio/video files, tracked files over 1MB not on a short allowlist (the three that are already legitimately tracked), tracked .env files, and a few common secret patterns. Wired into CI as its own job. --- .github/workflows/ci.yml | 23 ++++++ scripts/release-hygiene.mjs | 124 +++++++++++++++++++++++++++++++ scripts/release-hygiene.test.mjs | 38 ++++++++++ tests/test_ai_fallback.py | 14 +++- vitest.config.ts | 10 +++ 5 files changed, 205 insertions(+), 4 deletions(-) create mode 100644 scripts/release-hygiene.mjs create mode 100644 scripts/release-hygiene.test.mjs create mode 100644 vitest.config.ts diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index aa821459..0b2f06ea 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -87,6 +87,18 @@ jobs: - run: node scripts/gen-docs-manifest.mjs - run: node scripts/check-docs-drift.mjs + release-hygiene: + name: Release hygiene check + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: "20" + - run: node scripts/release-hygiene.mjs + python: name: Python tests runs-on: ${{ matrix.os }} @@ -102,6 +114,17 @@ jobs: with: python-version: "3.12" cache: "pip" + - name: Install ffmpeg + # Without this, every test that shells out to ffmpeg/ffprobe skips — + # about 22 tests were silently not running on any OS. + shell: bash + run: | + case "${{ runner.os }}" in + Linux) sudo apt-get update -qq && sudo apt-get install -y -qq ffmpeg ;; + macOS) brew install --quiet ffmpeg ;; + Windows) choco install ffmpeg -y --no-progress ;; + esac + ffmpeg -version | head -1 - name: Install minimal test deps # Skip whisper/torch — tests don't exercise them and they balloon CI time. run: | diff --git a/scripts/release-hygiene.mjs b/scripts/release-hygiene.mjs new file mode 100644 index 00000000..a0658e25 --- /dev/null +++ b/scripts/release-hygiene.mjs @@ -0,0 +1,124 @@ +// Catches what a build step can't: things that should never reach a commit. +// Runs over `git ls-files` (tracked files only) so untracked local scratch +// (data/, podcli-clips/, .env) never trips it. +import { execSync } from "child_process"; +import { readFileSync, statSync } from "fs"; +import { join, dirname, extname } from "path"; +import { fileURLToPath, pathToFileURL } from "url"; + +const root = join(dirname(fileURLToPath(import.meta.url)), ".."); + +// Real source recordings/renders have no business in git history — they +// belong in data/ or podcli-clips/, both gitignored. Icons, logos, and other +// still-image UI assets are handled by the size check below instead, since +// those are legitimately tracked. +const MEDIA_EXTENSIONS = new Set([ + ".mp4", ".mov", ".mkv", ".avi", ".webm", ".wmv", ".flv", + ".mp3", ".wav", ".flac", ".m4a", ".aac", ".ogg", +]); + +// Files over 1MB that are legitimately tracked. Keep this list short — add +// to it only when the file truly belongs in history (a model weight, a brand +// asset), not as a workaround for a one-off mistake. +const LARGE_FILE_ALLOWLIST = new Set([ + "backend/models/yamnet.onnx", // audio-energy model analyze_energy depends on + "public/promo.gif", // README hero asset + "remotion/public/style/detail-kraft.png", // riso pack texture +]); + +const MAX_SIZE_BYTES = 1024 * 1024; + +const SECRET_PATTERNS = [ + { name: "AWS access key", re: /\bAKIA[0-9A-Z]{16}\b/ }, + { name: "private key block", re: /-----BEGIN (?:RSA |EC |OPENSSH |DSA )?PRIVATE KEY-----/ }, + { name: "GitHub token", re: /\bghp_[A-Za-z0-9]{36}\b/ }, + { name: "Slack token", re: /\bxox[baprs]-[A-Za-z0-9-]{10,}\b/ }, + { name: "OpenAI-style secret key", re: /\bsk-[A-Za-z0-9]{32,}\b/ }, +]; + +// .env.example documents the shape of the file without real values — keep it. +const ENV_FILE_RE = /(^|\/)\.env(\..+)?$/; +const ENV_ALLOWLIST = new Set([".env.example"]); + +function trackedFiles() { + return execSync("git ls-files", { cwd: root, encoding: "utf8" }) + .split("\n") + .filter(Boolean); +} + +export function checkMediaFiles(files) { + return files + .filter((f) => MEDIA_EXTENSIONS.has(extname(f).toLowerCase())) + .map((f) => `tracked media file: ${f}`); +} + +function checkLargeFiles(files) { + const errors = []; + for (const f of files) { + if (LARGE_FILE_ALLOWLIST.has(f)) continue; + let size; + try { + size = statSync(join(root, f)).size; + } catch { + continue; // deleted-but-staged path, or a submodule gitlink; not our concern here + } + if (size > MAX_SIZE_BYTES) { + errors.push(`file over 1MB, not on the allowlist: ${f} (${(size / 1024 / 1024).toFixed(2)}MB)`); + } + } + return errors; +} + +export function checkEnvFiles(files) { + return files + .filter((f) => ENV_FILE_RE.test(f) && !ENV_ALLOWLIST.has(f)) + .map((f) => `tracked .env file: ${f}`); +} + +// Binary files (images, the onnx model) throw on readFileSync(..., "utf8") +// often enough that it's simpler to just skip anything non-text-shaped. +const TEXT_EXTENSIONS = new Set([ + ".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".py", ".go", ".sh", + ".md", ".json", ".yml", ".yaml", ".toml", ".txt", ".html", ".css", +]); + +export function matchSecretPatterns(content) { + return SECRET_PATTERNS.filter(({ re }) => re.test(content)).map(({ name }) => name); +} + +function checkSecretPatterns(files) { + const errors = []; + for (const f of files) { + if (!TEXT_EXTENSIONS.has(extname(f).toLowerCase())) continue; + let content; + try { + content = readFileSync(join(root, f), "utf8"); + } catch { + continue; + } + for (const name of matchSecretPatterns(content)) { + errors.push(`possible ${name} in ${f}`); + } + } + return errors; +} + +// Guarded so src/release-hygiene.test.ts can import the pure check functions +// above without running the whole CLI (and without a stray process.exit). +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { + const files = trackedFiles(); + const errors = [ + ...checkMediaFiles(files), + ...checkLargeFiles(files), + ...checkEnvFiles(files), + ...checkSecretPatterns(files), + ]; + + if (errors.length) { + console.error("Release hygiene check failed:"); + for (const e of errors) console.error(" -", e); + process.exit(1); + } + + console.log(`Release hygiene check passed: ${files.length} tracked files checked.`); +} diff --git a/scripts/release-hygiene.test.mjs b/scripts/release-hygiene.test.mjs new file mode 100644 index 00000000..1bff52e7 --- /dev/null +++ b/scripts/release-hygiene.test.mjs @@ -0,0 +1,38 @@ +import { describe, it, expect } from "vitest"; +import { checkMediaFiles, checkEnvFiles, matchSecretPatterns } from "./release-hygiene.mjs"; + +describe("release-hygiene: tracked media files", () => { + it("flags audio/video extensions", () => { + expect(checkMediaFiles(["episode.mp4", "raw.wav"])).toHaveLength(2); + }); + + it("leaves still-image assets alone (handled by the size check instead)", () => { + expect(checkMediaFiles(["logo.png", "icon.svg"])).toEqual([]); + }); +}); + +describe("release-hygiene: tracked .env files", () => { + it("flags .env and dotted variants", () => { + expect(checkEnvFiles([".env", ".env.local", "backend/.env"])).toHaveLength(3); + }); + + it("allows .env.example", () => { + expect(checkEnvFiles([".env.example"])).toEqual([]); + }); +}); + +describe("release-hygiene: secret patterns", () => { + it("flags an AWS access key", () => { + expect(matchSecretPatterns("key = AKIAABCDEFGHIJKLMNOP")).toContain("AWS access key"); + }); + + it("flags a private key block", () => { + expect(matchSecretPatterns("-----BEGIN RSA PRIVATE KEY-----\nabc\n-----END RSA PRIVATE KEY-----")).toContain( + "private key block", + ); + }); + + it("leaves ordinary source text alone", () => { + expect(matchSecretPatterns("const apiKey = process.env.API_KEY;")).toEqual([]); + }); +}); diff --git a/tests/test_ai_fallback.py b/tests/test_ai_fallback.py index 9c9bf02a..332e1caf 100644 --- a/tests/test_ai_fallback.py +++ b/tests/test_ai_fallback.py @@ -412,13 +412,19 @@ def test_configured_path_reads_from_env_file(self): self.assertEqual(found, cli) def test_find_cli_falls_back_to_shell_lookup(self): - with tempfile.TemporaryDirectory() as tmp: + with tempfile.TemporaryDirectory() as tmp, tempfile.TemporaryDirectory() as home: cli = os.path.join(tmp, "claude") with open(cli, "w", encoding="utf-8") as fh: fh.write("#!/bin/sh\n") - with mock.patch.object(ai, "_shell_lookup", return_value=cli): - with mock.patch("shutil.which", return_value=None): - found = ai._find_cli("claude", []) + # Without redirecting HOME, _find_cli's fixed lookup dirs (e.g. + # ~/.local/bin) hit the real filesystem — on a machine with an + # actual claude CLI installed there, it wins before shell lookup + # ever runs, and this test passes for the wrong reason. + with mock.patch.dict(os.environ, {"HOME": home, "PATH": ""}, clear=False): + with mock.patch("os.path.expanduser", side_effect=lambda p: p.replace("~", home)): + with mock.patch.object(ai, "_shell_lookup", return_value=cli): + with mock.patch("shutil.which", return_value=None): + found = ai._find_cli("claude", []) self.assertEqual(found, cli) def test_get_ai_cli_status_reports_candidates(self): diff --git a/vitest.config.ts b/vitest.config.ts new file mode 100644 index 00000000..d1c48a45 --- /dev/null +++ b/vitest.config.ts @@ -0,0 +1,10 @@ +import { defineConfig } from "vitest/config"; + +export default defineConfig({ + test: { + // `npm run build` compiles src/**/*.test.ts into dist/**/*.test.js. Without + // this, a stale build directory makes vitest run every test twice — once + // against the TS source, once against whatever JS it compiled last time. + exclude: ["**/node_modules/**", "**/dist/**"], + }, +}); From 62738694288e05c60704d25087a7af2281b6c61b Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:33:22 +0400 Subject: [PATCH 017/110] Fix silence removal dropping zero-duration words and stitching across cuts - _map_range required overlap_end > overlap_start, which a zero-duration word (start == end) can never satisfy even when it sits inside a kept range; map it as a point instead of dropping it, and keep the 10ms-floor filter from also swallowing it. - A word that straddles a removed cut (part in one kept range, part in the gap or the next one) is now assigned wholly to whichever side holds its midpoint instead of being stitched across the cut, which silently absorbed the cut into the word's own duration. Segment-level stitching across a genuine gap (sentences spanning real pauses) is unchanged. --- backend/services/silence_removal.py | 65 ++++++++++++++++++++++++++--- tests/test_silence_removal.py | 48 +++++++++++++++++++++ 2 files changed, 107 insertions(+), 6 deletions(-) diff --git a/backend/services/silence_removal.py b/backend/services/silence_removal.py index 9fcc111c..fc6c075c 100644 --- a/backend/services/silence_removal.py +++ b/backend/services/silence_removal.py @@ -327,7 +327,46 @@ def analyze_silence( return plan -def _map_range(start: float, end: float, keep_segments: list[dict]) -> Optional[tuple[float, float]]: +def _map_point(t: float, keep_segments: list[dict]) -> Optional[tuple[float, float]]: + """Map a single instant through keep_segments. None if t fell in a + removed range; otherwise its output position as a zero-length range.""" + cursor = 0.0 + for segment in keep_segments: + if segment["start"] <= t <= segment["end"]: + mapped = cursor + (t - segment["start"]) + return mapped, mapped + cursor += segment["end"] - segment["start"] + return None + + +def _map_range( + start: float, + end: float, + keep_segments: list[dict], + *, + assign_by_midpoint: bool = False, +) -> Optional[tuple[float, float]]: + # A zero (or inverted) duration item has no overlap for the interval test + # below to find — overlap_end > overlap_start never holds when they're + # equal — so it always mapped to None and got dropped. Map it as a point + # instead: keep it if it falls inside a kept range, drop it if not. + if end <= start: + return _map_point(start, keep_segments) + if assign_by_midpoint: + # A word that straddles a cut (part of it sits in the removed gap + # between two kept ranges) belongs wholly to whichever side holds its + # midpoint, not to a stitched span across the cut — stitching would + # silently absorb the cut into the word's own duration. + midpoint = (start + end) / 2.0 + cursor = 0.0 + for segment in keep_segments: + seg_start, seg_end = segment["start"], segment["end"] + if seg_start <= midpoint <= seg_end: + overlap_start = max(start, seg_start) + overlap_end = min(end, seg_end) + return cursor + overlap_start - seg_start, cursor + overlap_end - seg_start + cursor += seg_end - seg_start + return None output_cursor = 0.0 mapped_parts: list[tuple[float, float]] = [] for segment in keep_segments: @@ -344,7 +383,12 @@ def _map_range(start: float, end: float, keep_segments: list[dict]) -> Optional[ return mapped_parts[0][0], mapped_parts[-1][1] -def remap_timed_items(items: list[dict], keep_segments: list[dict]) -> list[dict]: +def remap_timed_items( + items: list[dict], + keep_segments: list[dict], + *, + assign_by_midpoint: bool = False, +) -> list[dict]: remapped: list[dict] = [] for item in items: try: @@ -352,16 +396,25 @@ def remap_timed_items(items: list[dict], keep_segments: list[dict]) -> list[dict end = float(item["end"]) except (KeyError, TypeError, ValueError): continue - mapped = _map_range(start, end, keep_segments) - if not mapped or mapped[1] - mapped[0] < 0.01: + mapped = _map_range(start, end, keep_segments, assign_by_midpoint=assign_by_midpoint) + if mapped is None: + continue + mapped_start, mapped_end = mapped + # The 10ms floor only makes sense for a real interval that got + # clipped down to near-nothing; a genuinely zero-duration source item + # (mapped_start == mapped_end by construction, from _map_point) is + # meant to be kept as a point marker. + if end > start and mapped_end - mapped_start < 0.01: continue - remapped.append({**item, "start": round(mapped[0], 3), "end": round(mapped[1], 3)}) + remapped.append({**item, "start": round(mapped_start, 3), "end": round(mapped_end, 3)}) return remapped def remap_transcript(transcript: dict, keep_segments: list[dict]) -> dict: remapped = dict(transcript or {}) - remapped["words"] = remap_timed_items(list(remapped.get("words") or []), keep_segments) + remapped["words"] = remap_timed_items( + list(remapped.get("words") or []), keep_segments, assign_by_midpoint=True + ) remapped["segments"] = remap_timed_items(list(remapped.get("segments") or []), keep_segments) remapped["duration"] = round(sum(s["end"] - s["start"] for s in keep_segments), 3) remapped["silence_removed"] = True diff --git a/tests/test_silence_removal.py b/tests/test_silence_removal.py index bc3af06e..a777bf44 100644 --- a/tests/test_silence_removal.py +++ b/tests/test_silence_removal.py @@ -1,14 +1,18 @@ import os import sys +import pytest + ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..")) BACKEND_ROOT = os.path.join(ROOT, "backend") if BACKEND_ROOT not in sys.path: sys.path.insert(0, BACKEND_ROOT) from services.silence_removal import ( + _map_range, plan_silence_removal, probabilities_to_speech_segments, + remap_timed_items, remap_transcript, ) @@ -64,6 +68,50 @@ def test_remap_transcript_closes_removed_gaps(): assert remapped["duration"] == 3.0 +def test_zero_duration_word_inside_a_kept_range_is_kept_as_a_point(): + keep_segments = [{"start": 0.5, "end": 2.0}, {"start": 4.5, "end": 6.0}] + words = [{"word": "", "start": 1.2, "end": 1.2}] + + remapped = remap_timed_items(words, keep_segments, assign_by_midpoint=True) + + assert len(remapped) == 1 + assert remapped[0]["start"] == remapped[0]["end"] == 0.7 + + +def test_zero_duration_word_in_a_removed_range_is_dropped(): + keep_segments = [{"start": 0.5, "end": 2.0}, {"start": 4.5, "end": 6.0}] + words = [{"word": "", "start": 3.0, "end": 3.0}] + + assert remap_timed_items(words, keep_segments, assign_by_midpoint=True) == [] + + +def test_word_straddling_a_removed_gap_with_no_owning_side_is_dropped(): + # Word runs 1.8-2.2 across a removed gap from 1.9-4.0. Its midpoint (2.0) + # falls inside that gap, so neither side owns it; it's dropped rather + # than stitched across the cut. + keep_segments = [{"start": 0.0, "end": 1.9}, {"start": 4.0, "end": 6.0}] + mapped = _map_range(1.8, 2.2, keep_segments, assign_by_midpoint=True) + assert mapped is None + + +def test_word_straddling_a_cut_is_not_stitched_across_it(): + keep_segments = [{"start": 0.0, "end": 2.0}, {"start": 2.0, "end": 4.0}] + # Midpoint 2.05 belongs to the second segment; the old stitching behavior + # would have spanned from the first segment's overlap through the + # second's, inflating the word's output duration across the cut. + mapped = _map_range(1.9, 2.2, keep_segments, assign_by_midpoint=True) + assert mapped == (2.0, 2.2) + + +def test_segments_still_stitch_across_a_removed_gap(): + # Sentence-level segments intentionally keep the old stitching behavior + # (see test_remap_transcript_closes_removed_gaps) — only words are + # reassigned by midpoint. + keep_segments = [{"start": 0.5, "end": 2.0}, {"start": 4.5, "end": 6.0}] + mapped = _map_range(1.0, 5.4, keep_segments) + assert mapped == pytest.approx((0.5, 2.4)) + + def test_probability_hysteresis_ignores_short_noise(): probabilities = [0.0] * 5 + [0.8] * 12 + [0.0] * 8 + [0.9] * 2 + [0.0] * 8 speech = probabilities_to_speech_segments( From 32a7506a1682d87d8255a279b461275b9c5cd882 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:33:59 +0400 Subject: [PATCH 018/110] Add docs/limitations.md stating what each stage can and cannot do Covers multicam sync, auto-cut to speaker, transcription engines and diarization, caption styles and font coverage, thumbnail generation, and export targets, each with a plain can-do/cannot-do statement. --- docs/limitations.md | 76 +++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 76 insertions(+) create mode 100644 docs/limitations.md diff --git a/docs/limitations.md b/docs/limitations.md new file mode 100644 index 00000000..682fe6ec --- /dev/null +++ b/docs/limitations.md @@ -0,0 +1,76 @@ +# Limitations + +What each stage can and cannot do today. This is a plain statement of current +behavior, not a roadmap. + +## Multicam sync + +- Sync is audio-only: coarse cross-correlation of onset envelopes, then a + fine GCC-PHAT match with linear clock-drift correction + (`backend/services/multicam_signal.py`). There is no timecode-based sync; + embedded timecode is read as session metadata only + (`backend/services/multicam.py:278-282`). +- A camera with no audio track cannot be synced automatically. Its offset + must be set by hand (`backend/services/multicam.py:1082`). +- Remote call recordings (Zoom, Riverside, one file per person) are assumed + to start together, since there's no shared audio to cross-correlate + against. + +## Auto-cut to speaker + +- Speaker assignment comes from per-person microphone levels, or from + imported diarization turns. There is no video-based speaker detection + (face tracking, lip sync). +- Short interjections are suppressed so a quick "mm-hm" doesn't steal the + cut, which means genuine quick back-and-forth exchanges can read as + slower-paced than they were. +- Call-recording layouts (one file per person as a split screen, one + gallery recording split into a tile per camera) render to MP4 fine but + cannot export to Premiere XML or FCPXML yet (`src/server.ts`, + `manage_multicam` tool description). + +## Transcription + +- Three engines: `whisper-py` (OpenAI Whisper, default), `whispercpp` + (local, no network), and `assemblyai` (`backend/services/engines.py`). +- Speaker diarization runs on the `whisper-py` and `assemblyai` paths. + `whispercpp` never produces speaker labels. Every word comes back with + `speaker: null` (`backend/services/transcription_whispercpp.py:67,215`). +- Diarization needs `HF_TOKEN` set (see `docs/configuration.md`). Without + it, or if diarization fails for any other reason, transcription still + completes. It just comes back with no speaker labels, degrading + silently rather than failing the job. +- Language support follows whichever engine is in use; podcli does not add + or restrict language coverage beyond what Whisper, whisper.cpp, or + AssemblyAI support natively. + +## Captions + +- Four built-in styles: `hormozi`, `karaoke`, `subtle`, `branded` + (`backend/services/caption_renderer.py`). +- Burn-in (rendered into the video via ASS/libass) is the primary path. + Soft `.srt` export also exists, and `.srt`/`.vtt` transcripts can be + imported directly. +- Non-Latin scripts (Georgian, Russian, etc.) render through font-fallback + stacks, not through the primary brand font. The brand font only draws + Latin glyphs. Expect a visibly different typeface on non-Latin captions + until a matching family is bundled for the style in use. + +## Thumbnails + +- Layout and copy are generated together: an LLM writes the HTML/CSS, and + Playwright renders it (`backend/services/thumbnail_ai.py`). +- Split-screen/multicam source frames are detected and cropped to the half + containing the speaking face. There's no true multi-person composited + layout (two subjects placed and sized independently in one frame). +- Headline copy is written to match the clip's theme, not quoted verbatim + from the transcript. There's no automated check that the headline's + claim matches what's actually said in the clip. + +## Exports + +- Supported targets: DaVinci Resolve FCPXML, Premiere XML/FCPXML (via + `manage_multicam`), and the podcli cloud editor. +- The cloud editor requires a podcli account and an active Pro plan. +- Call-recording-derived layouts (split screen, gallery tiles) cannot + export to Premiere or FCPXML. See Auto-cut above. From b12cf6623e8226f0e7b2be8c033597ddd376f551 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:34:18 +0400 Subject: [PATCH 019/110] Bound clip ranges to source duration and verify renders before reporting success --- backend/services/clip_generator.py | 76 +++++++++++++++++++++++++++++- backend/services/video_cut.py | 32 +++++++++++++ tests/test_clip_generator.py | 44 +++++++++++++++++ tests/test_video_cut.py | 62 ++++++++++++++++++++++++ 4 files changed, 213 insertions(+), 1 deletion(-) diff --git a/backend/services/clip_generator.py b/backend/services/clip_generator.py index 314f390d..34d8327e 100644 --- a/backend/services/clip_generator.py +++ b/backend/services/clip_generator.py @@ -32,6 +32,7 @@ normalize_audio, concat_outro, ) +from services.video_cut import probe_has_audio_stream, verify_full_decode from config.caption_styles import get_style from services.formats import get_format @@ -131,6 +132,35 @@ def _get_media_duration(path: str) -> float: return 0.0 +def _bound_range_to_source( + start_second: float, + end_second: float, + keep_segments: Optional[list[dict]], + source_duration: float, +) -> tuple[float, Optional[list[dict]]]: + """Clamp the requested range (and any keep_segments) to the source's + actual duration, so a too-long end_second can't silently truncate the + render instead of failing. source_duration <= 0 means ffprobe couldn't + read it; skip bounding rather than clamp against an unknown length. + """ + if source_duration <= 0: + return end_second, keep_segments + if start_second >= source_duration: + raise ValueError( + f"start_second ({start_second}s) is at or past the end of the " + f"source video ({source_duration:.2f}s)." + ) + end_second = min(end_second, source_duration) + if keep_segments: + bounded_segments = [] + for seg in keep_segments: + if seg["start"] >= source_duration: + continue + bounded_segments.append({**seg, "end": min(seg["end"], source_duration)}) + keep_segments = bounded_segments or None + return end_second, keep_segments + + def _detect_scene_cuts(path: str, threshold: float = 0.22, max_cuts: int = 32) -> list[float]: """ Detect hard visual changes likely to feel like jump cuts. @@ -912,6 +942,15 @@ def generate_clip( if end_second <= start_second: raise ValueError("end_second must be greater than start_second") + # ffmpeg's -ss/-t cut silently stops at EOF when end_second runs past the + # source, so without this the render "succeeds" with a shorter clip than + # requested while duration/end_second in the result still say the planned + # (too-long) window. Bound every range to what the source actually has. + source_duration = _get_media_duration(video_path) + end_second, keep_segments = _bound_range_to_source( + start_second, end_second, keep_segments, source_duration + ) + spec = get_format(format) # An episode that was never filmed takes the other road entirely. Branching @@ -1310,14 +1349,17 @@ def generate_clip( final_video_path = with_intro_path intro_offset = max(0.0, _get_media_duration(with_intro_path) - clip_duration) + outro_offset = 0.0 if outro_path and os.path.exists(outro_path): if progress_callback: progress_callback(85, f"Adding outro ({total_steps}/{total_steps})") + pre_outro_duration = _get_media_duration(final_video_path) with_outro_path = os.path.join(work_dir, "with_outro.mp4") concat_outro(final_video_path, outro_path, with_outro_path, crossfade_duration=bookend_fade) final_video_path = with_outro_path + outro_offset = max(0.0, _get_media_duration(with_outro_path) - pre_outro_duration) # Step 6: Move to output if progress_callback: @@ -1352,6 +1394,35 @@ def generate_clip( designed=_designed_cuts(cards, offset=intro_offset), ) + # A 0-exit ffmpeg run can still have written a short, silent, or + # corrupt file (source ran out under the requested range, a crop/ + # caption/normalize pass dropped the audio track, a concat mismatch). + # Catch that here instead of returning a "successful" result that + # describes a different clip than what landed on disk. + expected_duration = duration + intro_offset + outro_offset + actual_duration = _get_media_duration(final_path) + duration_tolerance = max(0.75, 0.05 * expected_duration) + if actual_duration <= 0 or abs(actual_duration - expected_duration) > duration_tolerance: + os.remove(final_path) + raise RuntimeError( + f"Render produced a {actual_duration:.2f}s file but expected " + f"~{expected_duration:.2f}s (body {duration:.2f}s" + f"{f' + intro {intro_offset:.2f}s' if intro_offset else ''}" + f"{f' + outro {outro_offset:.2f}s' if outro_offset else ''}). " + f"The source video likely ended before the requested range, or " + f"the render failed partway through." + ) + if not probe_has_audio_stream(final_path): + os.remove(final_path) + raise RuntimeError( + "Render completed but the output has no audio stream. " + "Re-check the source video and caption/crop pipeline." + ) + decode_error = verify_full_decode(final_path) + if decode_error: + os.remove(final_path) + raise RuntimeError(f"Render produced an undecodable output: {decode_error}") + # Get file size file_size = os.path.getsize(final_path) file_size_mb = round(file_size / (1024 * 1024), 2) @@ -1361,7 +1432,10 @@ def generate_clip( out = { "output_path": final_path, - "duration": round(duration, 2), + # The probed duration of the file actually on disk, not the + # planned window — they can differ if intro/outro were added, or + # (now caught above instead) if the source ran out early. + "duration": round(actual_duration, 2), "file_size_mb": file_size_mb, "title": title, "start_second": start_second, diff --git a/backend/services/video_cut.py b/backend/services/video_cut.py index cbe72d91..ed08b1c0 100644 --- a/backend/services/video_cut.py +++ b/backend/services/video_cut.py @@ -101,3 +101,35 @@ def cut_multi_segment( os.remove(p) if os.path.exists(concat_file): os.remove(concat_file) + + +def probe_has_audio_stream(path: str) -> bool: + """True if ffprobe finds at least one audio stream in the file.""" + cmd = [ + "ffprobe", "-v", "error", + "-select_streams", "a", + "-show_entries", "stream=codec_type", + "-of", "csv=p=0", + path, + ] + result = proc_run(cmd, timeout=FFMPEG_TIMEOUT, check=False) + if result.returncode != 0: + return False + return "audio" in (result.stdout or "") + + +def verify_full_decode(path: str) -> str | None: + """Decode the whole file and return ffmpeg's error output, or None if clean. + + An ffmpeg render that exits 0 can still have written a truncated or + corrupt file — a moov atom cut short, a partial frame at the tail, a + stream copy/concat mismatch. Those only surface on a full decode, which + is what this runs: the same check as `ffmpeg -v error -i x -f null -` + from the command line. + """ + cmd = ["ffmpeg", "-v", "error", "-i", path, "-f", "null", "-"] + result = proc_run(cmd, timeout=FFMPEG_TIMEOUT, check=False) + stderr = (result.stderr or "").strip() + if result.returncode != 0 or stderr: + return stderr[-1000:] if stderr else f"ffmpeg exited {result.returncode} decoding the output" + return None diff --git a/tests/test_clip_generator.py b/tests/test_clip_generator.py index 1f212f8e..0c7a74a3 100644 --- a/tests/test_clip_generator.py +++ b/tests/test_clip_generator.py @@ -519,5 +519,49 @@ def test_sidecar_paths_inherit_the_unique_stem(self): self.assertTrue(base.endswith("same_title_short-2")) +class BoundRangeToSourceTests(unittest.TestCase): + """A requested range past the end of the source must be clamped before + it reaches ffmpeg's -ss/-t cut, which would otherwise silently stop at + EOF and under-deliver instead of failing.""" + + def test_unknown_source_duration_leaves_range_untouched(self): + end, segs = cg._bound_range_to_source(10.0, 999.0, None, source_duration=0.0) + self.assertEqual(end, 999.0) + self.assertIsNone(segs) + + def test_end_second_clamped_to_source_duration(self): + end, segs = cg._bound_range_to_source(5.0, 120.0, None, source_duration=42.0) + self.assertEqual(end, 42.0) + self.assertIsNone(segs) + + def test_end_second_within_source_is_unchanged(self): + end, _ = cg._bound_range_to_source(5.0, 20.0, None, source_duration=42.0) + self.assertEqual(end, 20.0) + + def test_start_at_or_past_source_duration_raises(self): + with self.assertRaises(ValueError): + cg._bound_range_to_source(42.0, 50.0, None, source_duration=42.0) + with self.assertRaises(ValueError): + cg._bound_range_to_source(50.0, 60.0, None, source_duration=42.0) + + def test_keep_segments_clamped_and_out_of_range_ones_dropped(self): + segs_in = [ + {"start": 5.0, "end": 10.0}, + {"start": 35.0, "end": 60.0}, # end runs past source + {"start": 50.0, "end": 55.0}, # starts past source entirely + ] + end, segs = cg._bound_range_to_source(5.0, 60.0, segs_in, source_duration=42.0) + self.assertEqual(end, 42.0) + self.assertEqual(segs, [ + {"start": 5.0, "end": 10.0}, + {"start": 35.0, "end": 42.0}, + ]) + + def test_all_keep_segments_dropped_falls_back_to_none(self): + segs_in = [{"start": 50.0, "end": 55.0}] + _, segs = cg._bound_range_to_source(5.0, 60.0, segs_in, source_duration=42.0) + self.assertIsNone(segs) + + if __name__ == "__main__": unittest.main() diff --git a/tests/test_video_cut.py b/tests/test_video_cut.py index d9ea0072..12d11e16 100644 --- a/tests/test_video_cut.py +++ b/tests/test_video_cut.py @@ -213,5 +213,67 @@ def test_cut_on_a_keyframe_starts_at_the_requested_second(self): self._assert_frame_is(out, "cyan") +@unittest.skipUnless( + shutil.which("ffmpeg") and shutil.which("ffprobe"), "ffmpeg/ffprobe not installed" +) +class OutputVerificationTests(unittest.TestCase): + """probe_has_audio_stream and verify_full_decode are the post-render + checks that catch a 0-exit ffmpeg run that still wrote a bad file.""" + + @classmethod + def setUpClass(cls): + cls.tmpdir = tempfile.mkdtemp(prefix="podcli-verify-test-") + + cls.with_audio = os.path.join(cls.tmpdir, "with_audio.mp4") + subprocess.run( + [ + "ffmpeg", "-y", "-loglevel", "error", + "-f", "lavfi", "-i", "color=c=red:s=64x64:r=25:d=1", + "-f", "lavfi", "-i", "sine=frequency=440:duration=1", + "-c:v", "libx264", "-pix_fmt", "yuv420p", "-c:a", "aac", + "-shortest", cls.with_audio, + ], + check=True, capture_output=True, + ) + + cls.no_audio = os.path.join(cls.tmpdir, "no_audio.mp4") + subprocess.run( + [ + "ffmpeg", "-y", "-loglevel", "error", + "-f", "lavfi", "-i", "color=c=blue:s=64x64:r=25:d=1", + "-c:v", "libx264", "-pix_fmt", "yuv420p", + cls.no_audio, + ], + check=True, capture_output=True, + ) + + # A truncated file: valid header, but chopped mid-stream so a full + # decode (not just ffprobe's duration read) turns up an error. + cls.truncated = os.path.join(cls.tmpdir, "truncated.mp4") + with open(cls.with_audio, "rb") as f: + data = f.read() + with open(cls.truncated, "wb") as f: + f.write(data[: len(data) // 2]) + + @classmethod + def tearDownClass(cls): + shutil.rmtree(cls.tmpdir, ignore_errors=True) + + def test_probe_has_audio_stream_true_when_present(self): + self.assertTrue(video_cut.probe_has_audio_stream(self.with_audio)) + + def test_probe_has_audio_stream_false_when_missing(self): + self.assertFalse(video_cut.probe_has_audio_stream(self.no_audio)) + + def test_probe_has_audio_stream_false_for_missing_file(self): + self.assertFalse(video_cut.probe_has_audio_stream("/no/such/file.mp4")) + + def test_verify_full_decode_clean_file_returns_none(self): + self.assertIsNone(video_cut.verify_full_decode(self.with_audio)) + + def test_verify_full_decode_flags_truncated_file(self): + self.assertIsNotNone(video_cut.verify_full_decode(self.truncated)) + + if __name__ == "__main__": unittest.main() From 66284311c40cedd4adca4b3018477346c80006c3 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:35:49 +0400 Subject: [PATCH 020/110] Detect sources that changed on disk and force a re-sync --- backend/services/multicam.py | 72 ++++++++++++++++++++++++++++++++++-- tests/test_multicam.py | 34 +++++++++++++++++ 2 files changed, 103 insertions(+), 3 deletions(-) diff --git a/backend/services/multicam.py b/backend/services/multicam.py index d85b6b0f..0d50597c 100644 --- a/backend/services/multicam.py +++ b/backend/services/multicam.py @@ -112,6 +112,8 @@ class Source: height: int = 0 fps: float = 0.0 timecode: float = 0.0 # embedded start timecode in seconds; editors address media from here + file_size: int = 0 # size at probe time; a mismatch on reopen means the file changed underneath us + file_mtime_ns: int = 0 # mtime at probe time, nanosecond resolution role: str = "ignore" # "camera" | "mic" | "ignore" # camera: a person id or "wide"; mic: a person id, or "" for a shared room mic person: str = "" @@ -350,6 +352,7 @@ def probe_source(path: str) -> Source: if not 1 <= fps <= 240: fps = 30.0 sample_rate = int(audio.get("sample_rate") or 0) if audio else 0 + stat = os.stat(path) return Source( # Built from name and size so the same recording gets the same id # on another machine and a saved edit or cut list still points at it. @@ -363,9 +366,63 @@ def probe_source(path: str) -> Source: height=int(video.get("height") or 0) if video else 0, fps=round(fps, 3), timecode=round(_timecode_seconds(info, fps, sample_rate), 6), + file_size=stat.st_size, + file_mtime_ns=stat.st_mtime_ns, ) +def _file_identity(s: Source) -> str: + """Short fingerprint of the file state a source was last probed against. + + Used to key derived caches (extracted sync audio, activity) so that a + file silently changing underneath its source (replaced, re-exported, + re-encoded) can't serve stale cached work keyed only on the source id, + which is built from basename and size and so does not change. + """ + return hashlib.sha1(f"{s.file_size}:{s.file_mtime_ns}".encode()).hexdigest()[:8] + + +def _source_changed_on_disk(s: Source) -> bool: + """True if the file at `s.path` no longer matches what was probed.""" + try: + stat = os.stat(s.path) + except OSError: + return False + return stat.st_size != s.file_size or stat.st_mtime_ns != s.file_mtime_ns + + +def refresh_stale_sources(session: "MulticamSession") -> bool: + """Re-probe any source whose file changed since it was last probed. + + A saved session reopens by matching the set of file paths alone, so a + camera file swapped out for a re-export with the same name silently kept + its old sync, offset and caches. Re-probe it, reset everything derived + from its old bytes, and let the normal flows (sync, activity, previews) + regenerate against the new file. + """ + affected_in_use = False + for s in session.sources: + if s.virtual or not s.path or not _source_changed_on_disk(s): + continue + try: + fresh = probe_source(s.path) + except Exception: + continue + s.duration, s.has_audio, s.audio_channels = fresh.duration, fresh.has_audio, fresh.audio_channels + s.width, s.height, s.fps, s.timecode = fresh.width, fresh.height, fresh.fps, fresh.timecode + s.file_size, s.file_mtime_ns = fresh.file_size, fresh.file_mtime_ns + s.offset, s.speed, s.sync = None, 1.0, { + "status": "failed", + "message": "This file changed on disk since it was last synced. Sync again.", + } + affected_in_use = affected_in_use or _in_use(session, s) + if affected_in_use: + session.activity_key = "" + session.cuts = [] + session.range_start = session.range_end = None + return affected_in_use + + _WIDE = re.compile(r"(^|[^a-z])(wide|master|main|both|all|group|ws|two.?shot|2.?shot|overview)([^a-z]|$)") _MIX = re.compile(r"(^|[^a-z])(mix|lr|stereo|master.?mix|program)([^a-z]|$)") _TRACK = re.compile(r"(?:tr|track|ch|channel|in|input|mic)[ _-]?0*(\d{1,2})(?!\d)") @@ -670,7 +727,12 @@ def new_session( # caller makes after, under the same lock as any other. existing = find_session(found) if existing: - _emit(progress_callback, 100, f"Reopened the edit for these {len(found)} files") + if refresh_stale_sources(existing): + existing.save() + _emit(progress_callback, 100, f"Reopened the edit for these {len(found)} files; " + "one or more changed on disk and need syncing again") + else: + _emit(progress_callback, 100, f"Reopened the edit for these {len(found)} files") return existing named = [(str(p.get("name", "")).strip(), p.get("role")) if isinstance(p, dict) else (str(p or "").strip(), None) for p in (people or [])] @@ -929,7 +991,11 @@ def _cut_inputs(session: MulticamSession) -> str: # --------------------------------------------------------------------------- def _extract(source: Source, out: Path, channel: int = -1, rate: int = SYNC_RATE) -> Path: - if out.exists() and out.stat().st_mtime >= os.path.getmtime(source.path): + # Keyed on the probed file identity, not a live mtime check: a replaced + # file can land within the same mtime second, and the id in `out`'s name + # (basename:size) doesn't change either, so neither alone is reliable. + out = out.with_name(f"{out.stem}-{_file_identity(source)}{out.suffix}") + if out.exists(): return out tmp = out.with_suffix(".tmp.wav") pick = "pan=mono|c0=c0" if source.audio_channels < 2 else "pan=mono|c0=0.5*c0+0.5*c1" @@ -1165,7 +1231,7 @@ def wav(s: Source) -> np.ndarray: # --------------------------------------------------------------------------- def _activity_key(session: MulticamSession) -> str: - feeds = [(s.id, ch, pid, s.offset, s.speed) for s, ch, pid in session.person_mics()] + feeds = [(s.id, _file_identity(s), ch, pid, s.offset, s.speed) for s, ch, pid in session.person_mics()] blob = json.dumps([feeds, session.speaker_map, session.person_ids()], sort_keys=True, default=str) return hashlib.sha1(blob.encode()).hexdigest()[:16] diff --git a/tests/test_multicam.py b/tests/test_multicam.py index 2c2c07e3..99127f0f 100644 --- a/tests/test_multicam.py +++ b/tests/test_multicam.py @@ -583,6 +583,40 @@ def test_preview_stills_regenerate_after_a_nudge_instead_of_serving_a_stale_fram assert look_after != look_before and os.path.exists(look_after) +def test_reopening_a_session_with_an_unchanged_camera_keeps_its_sync(episode): + session = mc.new_session(folder=str(episode), people=["Nika", "Ana"]) + session = mc.sync_session(session) + cam = next(s for s in session.sources if os.path.basename(s.path) == "cam_one.mp4") + assert cam.synced + + reopened = mc.new_session(folder=str(episode), people=["Nika", "Ana"]) + assert reopened.session_id == session.session_id + assert reopened.source(cam.id).synced + assert reopened.source(cam.id).offset == cam.offset + + +def test_reopening_a_session_detects_a_camera_replaced_on_disk(episode): + session = mc.new_session(folder=str(episode), people=["Nika", "Ana"]) + session = mc.sync_session(session) + cam_path = episode / "cam_one.mp4" + cam = next(s for s in session.sources if os.path.basename(s.path) == "cam_one.mp4") + assert cam.synced + old_identity = (cam.file_size, cam.file_mtime_ns) + + # Re-export the same file path with different content: same name, same id + # (basename:size can coincide), but the bytes underneath changed. + _ffmpeg("-f", "lavfi", "-i", "color=c=yellow:s=160x90:r=30:d=40", "-i", str(episode.parent / "one.wav"), + "-shortest", "-c:v", "libx264", "-pix_fmt", "yuv420p", "-c:a", "aac", "-y", str(cam_path)) + + reopened = mc.new_session(folder=str(episode), people=["Nika", "Ana"]) + assert reopened.session_id == session.session_id + fresh = reopened.source(cam.id) + assert (fresh.file_size, fresh.file_mtime_ns) != old_identity + assert not fresh.synced + assert fresh.sync["status"] == "failed" + assert reopened.cuts == [] + + def test_last_person_is_the_guest_and_roles_survive_renames(episode): session = mc.new_session(folder=str(episode), people=["Nihal", "Cameron", "Ana"]) assert [(p.name, p.role) for p in session.people] == [("Nihal", "host"), ("Cameron", "host"), ("Ana", "guest")] From ee87544b35a33dfb1d837caf1c2148efc2272a06 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:36:22 +0400 Subject: [PATCH 021/110] Thread a real language through parsed transcripts and fix speaker_segments drift - transcript_parser.py hardcoded language: en for every format (SRT, VTT, JSON, speaker-labeled). None of these formats carry language info of their own, so default to 'und' (undetermined) and accept an optional language override threaded through parse_transcript end to end (MCP schema, web-server route, Python handler). - speaker_segments used each block's raw start/end while words and segments applied time_adjust, so a non-zero adjust put speaker boundaries out of sync with the words and segments inside them. Apply time_adjust there too. --- backend/main.py | 5 +- backend/services/transcript_parser.py | 70 +++++++++++++++++++++------ src/server.ts | 10 +++- src/ui/web-server.ts | 3 +- tests/test_transcript_parser.py | 30 ++++++++++++ 5 files changed, 101 insertions(+), 17 deletions(-) diff --git a/backend/main.py b/backend/main.py index ce3e5749..567a83c2 100644 --- a/backend/main.py +++ b/backend/main.py @@ -343,13 +343,16 @@ def handle_parse_transcript(task_id: str, params: dict): raw_text = params.get("raw_text", "") total_duration = params.get("total_duration") time_adjust = params.get("time_adjust", 0.0) + language = params.get("language") if not raw_text: emit_result(task_id, "error", error="raw_text is required") return emit_progress(task_id, "parsing", 50, "Parsing transcript...") - result = detect_and_parse(raw_text, total_duration=total_duration, time_adjust=time_adjust) + result = detect_and_parse( + raw_text, total_duration=total_duration, time_adjust=time_adjust, language=language + ) if "error" in result: emit_result(task_id, "error", error=result["error"]) diff --git a/backend/services/transcript_parser.py b/backend/services/transcript_parser.py index e34fa4c7..d3e12680 100644 --- a/backend/services/transcript_parser.py +++ b/backend/services/transcript_parser.py @@ -34,7 +34,12 @@ def parse_vtt_timestamp(ts: str) -> float: return 0.0 -def parse_srt(text: str, total_duration: Optional[float] = None, time_adjust: float = 0.0) -> Dict[str, Any]: +def parse_srt( + text: str, + total_duration: Optional[float] = None, + time_adjust: float = 0.0, + language: Optional[str] = None, +) -> Dict[str, Any]: """ Parse an SRT subtitle file into structured data with word-level timestamps. @@ -98,10 +103,15 @@ def parse_srt(text: str, total_duration: Optional[float] = None, time_adjust: fl if not blocks: return {"error": "No subtitle blocks found in SRT"} - return _blocks_to_result(blocks, total_duration, time_adjust, fmt="srt") + return _blocks_to_result(blocks, total_duration, time_adjust, fmt="srt", language=language) -def parse_vtt(text: str, total_duration: Optional[float] = None, time_adjust: float = 0.0) -> Dict[str, Any]: +def parse_vtt( + text: str, + total_duration: Optional[float] = None, + time_adjust: float = 0.0, + language: Optional[str] = None, +) -> Dict[str, Any]: """ Parse a WebVTT subtitle file into structured data with word-level timestamps. @@ -185,10 +195,16 @@ def parse_vtt(text: str, total_duration: Optional[float] = None, time_adjust: fl if not blocks: return {"error": "No subtitle blocks found in VTT"} - return _blocks_to_result(blocks, total_duration, time_adjust, fmt="vtt") + return _blocks_to_result(blocks, total_duration, time_adjust, fmt="vtt", language=language) -def _blocks_to_result(blocks: List[Dict], total_duration: Optional[float], time_adjust: float, fmt: str) -> Dict[str, Any]: +def _blocks_to_result( + blocks: List[Dict], + total_duration: Optional[float], + time_adjust: float, + fmt: str, + language: Optional[str] = None, +) -> Dict[str, Any]: """Convert parsed subtitle blocks into the standard output format.""" all_words = [] segments = [] @@ -227,7 +243,10 @@ def _blocks_to_result(blocks: List[Dict], total_duration: Optional[float], time_ "words": all_words, "segments": segments, "duration": round(duration, 2), - "language": "en", + # These formats carry no language info of their own; caller-supplied + # language wins, "und" (undetermined) otherwise. Previously this + # always claimed "en", which is simply wrong for anything else. + "language": language or "und", "speakers": [], "speaker_segments": [], "imported": True, @@ -235,7 +254,12 @@ def _blocks_to_result(blocks: List[Dict], total_duration: Optional[float], time_ } -def detect_and_parse(text: str, total_duration: Optional[float] = None, time_adjust: float = 0.0) -> Dict[str, Any]: +def detect_and_parse( + text: str, + total_duration: Optional[float] = None, + time_adjust: float = 0.0, + language: Optional[str] = None, +) -> Dict[str, Any]: """ Auto-detect transcript format and parse accordingly. @@ -249,12 +273,12 @@ def detect_and_parse(text: str, total_duration: Optional[float] = None, time_adj # VTT detection if stripped.startswith("WEBVTT"): - return parse_vtt(text, total_duration=total_duration, time_adjust=time_adjust) + return parse_vtt(text, total_duration=total_duration, time_adjust=time_adjust, language=language) # SRT detection: first line is a digit, second line contains --> lines = stripped.split("\n", 3) if len(lines) >= 2 and re.match(r'^\d+$', lines[0].strip()) and '-->' in lines[1]: - return parse_srt(text, total_duration=total_duration, time_adjust=time_adjust) + return parse_srt(text, total_duration=total_duration, time_adjust=time_adjust, language=language) # JSON detection if stripped.startswith("{") or stripped.startswith("["): @@ -276,7 +300,9 @@ def detect_and_parse(text: str, total_duration: Optional[float] = None, time_adj "words": words, "segments": segments_list, "duration": round(duration, 2), - "language": "en", + "language": (data.get("language") if isinstance(data, dict) else None) + or language + or "und", "speakers": [], "speaker_segments": [], "imported": True, @@ -286,7 +312,9 @@ def detect_and_parse(text: str, total_duration: Optional[float] = None, time_adj pass # Fall through to speaker format # Default: speaker format - return parse_speaker_transcript(text, total_duration=total_duration, time_adjust=time_adjust) + return parse_speaker_transcript( + text, total_duration=total_duration, time_adjust=time_adjust, language=language + ) def parse_timestamp(ts: str) -> float: @@ -299,7 +327,12 @@ def parse_timestamp(ts: str) -> float: return 0.0 -def parse_speaker_transcript(raw_text: str, total_duration: Optional[float] = None, time_adjust: float = 0.0) -> Dict[str, Any]: +def parse_speaker_transcript( + raw_text: str, + total_duration: Optional[float] = None, + time_adjust: float = 0.0, + language: Optional[str] = None, +) -> Dict[str, Any]: """ Parse a speaker-labeled transcript into structured data with word-level timestamps. @@ -309,6 +342,8 @@ def parse_speaker_transcript(raw_text: str, total_duration: Optional[float] = No raw_text: The raw transcript text total_duration: Total duration of the podcast in seconds time_adjust: Seconds to add/subtract from all timestamps (e.g., -1.0 to shift 1s earlier) + language: Language of the transcript, if known. Defaults to "und" (undetermined) — + this format carries no language info of its own to detect it from. """ lines = raw_text.strip().split("\n") @@ -434,10 +469,17 @@ def parse_speaker_transcript(raw_text: str, total_duration: Optional[float] = No "words": all_words, "segments": segments, "duration": round(duration, 2), - "language": "en", + "language": language or "und", "speakers": speakers_list, + # words/segments above apply time_adjust and clamp to 0; this has to + # match or a speaker's segment boundaries drift out of sync with the + # words and segments supposedly inside them. "speaker_segments": [ - {"speaker": b["speaker"], "start": b["start"], "end": b["end"]} + { + "speaker": b["speaker"], + "start": round(max(0, b["start"] + time_adjust), 3), + "end": round(max(0, b["end"] + time_adjust), 3), + } for b in blocks ], "imported": True, diff --git a/src/server.ts b/src/server.ts index ccb72760..0e132b2f 100644 --- a/src/server.ts +++ b/src/server.ts @@ -2529,8 +2529,15 @@ export function createServer(): McpServer { .optional() .default(0) .describe("Offset in seconds to add to all timestamps"), + language: z + .string() + .optional() + .describe( + "ISO language code of the transcript (e.g. 'ka'). This format has no language " + + "info of its own, so omitting it labels the result 'und' rather than guessing.", + ), }, - async ({ file_path, raw_text, total_duration, time_adjust }) => { + async ({ file_path, raw_text, total_duration, time_adjust, language }) => { try { const res = await fetch(`${webServerUrl}/api/parse-transcript`, { method: "POST", @@ -2540,6 +2547,7 @@ export function createServer(): McpServer { raw_text, total_duration, time_adjust, + language, }), }); if (!res.ok) { diff --git a/src/ui/web-server.ts b/src/ui/web-server.ts index 87aa526f..4633e128 100644 --- a/src/ui/web-server.ts +++ b/src/ui/web-server.ts @@ -992,7 +992,7 @@ app.post("/api/import-transcript", (req, res) => { * Uses Python backend to generate word-level timestamps. */ app.post("/api/parse-transcript", async (req, res) => { - const { file_path, raw_text, total_duration, time_adjust = 0 } = req.body; + const { file_path, raw_text, total_duration, time_adjust = 0, language } = req.body; if (!file_path) { res.status(400).json({ error: "file_path is required" }); @@ -1008,6 +1008,7 @@ app.post("/api/parse-transcript", async (req, res) => { raw_text, total_duration: total_duration || null, time_adjust: time_adjust || 0, + language: language || null, }); if (result.data) { diff --git a/tests/test_transcript_parser.py b/tests/test_transcript_parser.py index 6c9f3878..9a78b49b 100644 --- a/tests/test_transcript_parser.py +++ b/tests/test_transcript_parser.py @@ -135,6 +135,36 @@ def test_hh_mm_ss_headers(self): # First word start near 3600s self.assertGreaterEqual(result["words"][0]["start"], 3600.0 - 1.0) + def test_language_defaults_to_undetermined_not_english(self): + raw = "Alice (00:00)\nHello\n" + result = tp.parse_speaker_transcript(raw, total_duration=5.0) + self.assertEqual(result["language"], "und") + + def test_language_passthrough(self): + raw = "Alice (00:00)\nHello\n" + result = tp.parse_speaker_transcript(raw, total_duration=5.0, language="ka") + self.assertEqual(result["language"], "ka") + + def test_detect_and_parse_threads_language_to_each_format(self): + speaker_raw = "Alice (00:00)\nHello\n" + srt_raw = "1\n00:00:00,000 --> 00:00:01,000\nHello\n" + vtt_raw = "WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nHello\n" + for raw in (speaker_raw, srt_raw, vtt_raw): + with self.subTest(raw=raw): + result = tp.detect_and_parse(raw, total_duration=5.0, language="ka") + self.assertEqual(result["language"], "ka") + + def test_speaker_segments_apply_time_adjust_like_words_and_segments(self): + # Regression: speaker_segments previously used the raw block + # start/end, ignoring time_adjust, while words and segments applied + # it — the three arrays drifted out of sync for any non-zero adjust. + raw = "Alice (00:10)\nHello there\n" + result = tp.parse_speaker_transcript(raw, total_duration=30.0, time_adjust=-2.0) + seg = result["speaker_segments"][0] + self.assertAlmostEqual(seg["start"], result["segments"][0]["start"], places=3) + self.assertAlmostEqual(seg["end"], result["segments"][0]["end"], places=3) + self.assertAlmostEqual(seg["start"], 8.0, places=3) + if __name__ == "__main__": unittest.main() From 6c76c7485120ef1563f1611215acd41b8035a5d1 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:36:31 +0400 Subject: [PATCH 022/110] Update podcli-installed PodStack commands on upgrade, keep user edits installCommands used to skip any command file that already existed, so a user never got command fixes/improvements after podcli update unless they deleted the file first. It now tracks a sha256 manifest (.claude/commands/.podcli-manifest.json) of what it last wrote per file: unmodified files refresh to the new embedded version, user-edited files are left alone and reported, and files that predate the manifest are adopted as a baseline rather than silently overwritten. Add the podstack package's first tests, covering first install, idempotent re-install, user-edit preservation, staleness updates, and pre-manifest adoption. --- cli/internal/podstack/podstack.go | 138 +++++++++++++++++-- cli/internal/podstack/podstack_test.go | 176 +++++++++++++++++++++++++ 2 files changed, 303 insertions(+), 11 deletions(-) create mode 100644 cli/internal/podstack/podstack_test.go diff --git a/cli/internal/podstack/podstack.go b/cli/internal/podstack/podstack.go index 6a05b373..5b07d6cf 100644 --- a/cli/internal/podstack/podstack.go +++ b/cli/internal/podstack/podstack.go @@ -6,7 +6,10 @@ package podstack import ( + "crypto/sha256" "embed" + "encoding/hex" + "encoding/json" "fmt" "io/fs" "os" @@ -75,28 +78,137 @@ func warnEmptyKnowledge() { fmt.Fprintf(os.Stderr, " %sKnowledge base is empty.%s Run %spodcli knowledge init%s first, or %s/bootstrap-knowledge%s in your agent, so the workflow knows your show.\n", colYellow, colReset, colAccent, colReset, colAccent, colReset) } +// manifestName is the install-tracking file, hidden among the slash commands +// it describes. It is not itself a command (no .md suffix), so Names() and +// IsCommand() never see it. +const manifestName = ".podcli-manifest.json" + +// installManifest records the sha256 of the content podcli last wrote for +// each command file, so a later install can tell "podcli wrote this and +// nothing has touched it since" (safe to overwrite with the new version) +// apart from "the user edited this" (leave it alone). +type installManifest struct { + Files map[string]string `json:"files"` +} + +func readManifest(dest string) installManifest { + m := installManifest{Files: map[string]string{}} + data, err := os.ReadFile(filepath.Join(dest, manifestName)) + if err != nil { + return m + } + _ = json.Unmarshal(data, &m) + if m.Files == nil { + m.Files = map[string]string{} + } + return m +} + +func writeManifest(dest string, m installManifest) error { + data, err := json.MarshalIndent(m, "", " ") + if err != nil { + return err + } + return os.WriteFile(filepath.Join(dest, manifestName), data, 0o644) +} + +func sha256Hex(data []byte) string { + sum := sha256.Sum256(data) + return hex.EncodeToString(sum[:]) +} + +// installReport summarizes what installCommands did, for callers that want +// to tell the user about files it left alone. +type installReport struct { + Installed []string // written for the first time + Updated []string // podcli-owned, unmodified by the user, refreshed to the new version + UserModified []string // changed since podcli last wrote them — left alone +} + +func (r installReport) hasUpdates() bool { + return len(r.Installed) > 0 || len(r.Updated) > 0 +} + // installCommands writes the embedded slash-command files into -// /.claude/commands so the agent can resolve /. Existing files are -// left untouched, so a user's local edits win. -func installCommands(project string) error { +// /.claude/commands so the agent can resolve /. A manifest +// tracks the hash of what podcli wrote for each file: on a later run (e.g. +// after `podcli update`), a file whose on-disk hash still matches that +// record gets refreshed to the new embedded version; a file the user edited +// is left untouched and reported instead of silently overwritten. A file +// that predates the manifest (no record at all) is treated the same way — +// left alone — and its current content becomes the new baseline, so podcli +// never clobbers an install from before this tracking existed. +func installCommands(project string) (installReport, error) { + var report installReport dest := filepath.Join(project, ".claude", "commands") if err := os.MkdirAll(dest, 0o755); err != nil { - return err + return report, err } - return fs.WalkDir(commands, "commands", func(p string, d fs.DirEntry, err error) error { + manifest := readManifest(dest) + changed := false + + err := fs.WalkDir(commands, "commands", func(p string, d fs.DirEntry, err error) error { if err != nil || d.IsDir() { return err } - target := filepath.Join(dest, filepath.Base(p)) - if _, err := os.Stat(target); err == nil { + name := filepath.Base(p) + target := filepath.Join(dest, name) + embedded, err := commands.ReadFile(p) + if err != nil { + return err + } + embeddedHash := sha256Hex(embedded) + + current, statErr := os.ReadFile(target) + if statErr != nil { + // Never installed here before. + if err := os.WriteFile(target, embedded, 0o644); err != nil { + return err + } + manifest.Files[name] = embeddedHash + changed = true + report.Installed = append(report.Installed, name) return nil } - data, err := commands.ReadFile(p) - if err != nil { + + lastInstalledHash, tracked := manifest.Files[name] + currentHash := sha256Hex(current) + + if !tracked { + // Predates the manifest: don't know if this is a stock file or a + // user edit, so leave it and adopt its current state as the baseline. + manifest.Files[name] = currentHash + changed = true + return nil + } + + if currentHash != lastInstalledHash { + // The user changed it since podcli last wrote it. + report.UserModified = append(report.UserModified, name) + return nil + } + + if currentHash == embeddedHash { + return nil // already current + } + + if err := os.WriteFile(target, embedded, 0o644); err != nil { return err } - return os.WriteFile(target, data, 0o644) + manifest.Files[name] = embeddedHash + changed = true + report.Updated = append(report.Updated, name) + return nil }) + if err != nil { + return report, err + } + if changed { + if err := writeManifest(dest, manifest); err != nil { + return report, err + } + } + return report, nil } // Run launches the agent for cmd with the remaining args as the slash-command @@ -137,9 +249,13 @@ func Run(cmd string, args []string) int { if err != nil { project = "." } - if err := installCommands(project); err != nil { + report, err := installCommands(project) + if err != nil { fmt.Fprintf(os.Stderr, " %swarning:%s could not install slash commands: %v\n", colYellow, colReset, err) } + if len(report.UserModified) > 0 { + fmt.Fprintf(os.Stderr, " %sKept your edits%s to: %s\n", colDim, colReset, strings.Join(report.UserModified, ", ")) + } warnEmptyKnowledge() prompt := "/" + cmd diff --git a/cli/internal/podstack/podstack_test.go b/cli/internal/podstack/podstack_test.go new file mode 100644 index 00000000..7ac5d372 --- /dev/null +++ b/cli/internal/podstack/podstack_test.go @@ -0,0 +1,176 @@ +package podstack + +import ( + "os" + "path/filepath" + "testing" +) + +func TestInstallCommandsWritesEveryFileOnFirstRun(t *testing.T) { + project := t.TempDir() + + report, err := installCommands(project) + if err != nil { + t.Fatalf("installCommands: %v", err) + } + if len(report.Installed) == 0 { + t.Fatal("expected files to be reported as installed") + } + if len(report.Updated) != 0 || len(report.UserModified) != 0 { + t.Fatalf("first run should only install, got updated=%v userModified=%v", report.Updated, report.UserModified) + } + + dest := filepath.Join(project, ".claude", "commands") + if _, err := os.Stat(filepath.Join(dest, "auto.md")); err != nil { + t.Fatalf("auto.md not written: %v", err) + } + if _, err := os.Stat(filepath.Join(dest, manifestName)); err != nil { + t.Fatalf("manifest not written: %v", err) + } +} + +func TestInstallCommandsIsIdempotentWhenNothingChanged(t *testing.T) { + project := t.TempDir() + if _, err := installCommands(project); err != nil { + t.Fatalf("first install: %v", err) + } + + report, err := installCommands(project) + if err != nil { + t.Fatalf("second install: %v", err) + } + if report.hasUpdates() || len(report.UserModified) != 0 { + t.Fatalf("expected no-op on second run, got %+v", report) + } +} + +func TestInstallCommandsLeavesUserEditsAloneAndReportsThem(t *testing.T) { + project := t.TempDir() + if _, err := installCommands(project); err != nil { + t.Fatalf("first install: %v", err) + } + dest := filepath.Join(project, ".claude", "commands") + target := filepath.Join(dest, "auto.md") + + installed, err := os.ReadFile(target) + if err != nil { + t.Fatalf("read installed file: %v", err) + } + edited := append(append([]byte{}, installed...), []byte("\n\n")...) + if err := os.WriteFile(target, edited, 0o644); err != nil { + t.Fatalf("simulate user edit: %v", err) + } + + report, err := installCommands(project) + if err != nil { + t.Fatalf("second install: %v", err) + } + found := false + for _, f := range report.UserModified { + if f == "auto.md" { + found = true + } + } + if !found { + t.Fatalf("expected auto.md to be reported as user-modified, got %+v", report) + } + + after, err := os.ReadFile(target) + if err != nil { + t.Fatalf("read after install: %v", err) + } + if string(after) != string(edited) { + t.Fatal("user-modified file was overwritten, but it should have been left alone") + } +} + +func TestInstallCommandsUpdatesFilesTheUserDidNotTouch(t *testing.T) { + project := t.TempDir() + if _, err := installCommands(project); err != nil { + t.Fatalf("first install: %v", err) + } + dest := filepath.Join(project, ".claude", "commands") + target := filepath.Join(dest, "auto.md") + + // Simulate an older podcli version: rewrite both the file and its + // manifest entry to some other content, consistently, as if that's what + // an earlier install wrote. The file is still untouched by the user + // relative to that record, so the next install should treat it as safe + // to refresh to the current embedded content. + stale := []byte("# an older auto.md\n") + if err := os.WriteFile(target, stale, 0o644); err != nil { + t.Fatalf("simulate stale install: %v", err) + } + manifest := readManifest(dest) + manifest.Files["auto.md"] = sha256Hex(stale) + if err := writeManifest(dest, manifest); err != nil { + t.Fatalf("rewrite manifest: %v", err) + } + + report, err := installCommands(project) + if err != nil { + t.Fatalf("second install: %v", err) + } + found := false + for _, f := range report.Updated { + if f == "auto.md" { + found = true + } + } + if !found { + t.Fatalf("expected auto.md to be reported as updated, got %+v", report) + } + if len(report.UserModified) != 0 { + t.Fatalf("unmodified file should not be reported as user-modified, got %v", report.UserModified) + } + + after, err := os.ReadFile(target) + if err != nil { + t.Fatalf("read after install: %v", err) + } + if string(after) == string(stale) { + t.Fatal("stale file was not refreshed to the current embedded content") + } +} + +func TestInstallCommandsAdoptsPreManifestFilesWithoutOverwriting(t *testing.T) { + project := t.TempDir() + dest := filepath.Join(project, ".claude", "commands") + if err := os.MkdirAll(dest, 0o755); err != nil { + t.Fatalf("mkdir: %v", err) + } + // A file that predates manifest tracking: some arbitrary content, no + // manifest entry for it at all. + legacy := filepath.Join(dest, "auto.md") + if err := os.WriteFile(legacy, []byte("# a pre-existing, possibly hand-edited auto.md\n"), 0o644); err != nil { + t.Fatalf("seed legacy file: %v", err) + } + + report, err := installCommands(project) + if err != nil { + t.Fatalf("installCommands: %v", err) + } + for _, f := range report.Installed { + if f == "auto.md" { + t.Fatal("pre-manifest file must not be reported as freshly installed") + } + } + for _, f := range report.Updated { + if f == "auto.md" { + t.Fatal("pre-manifest file must not be overwritten on first sight") + } + } + + content, err := os.ReadFile(legacy) + if err != nil { + t.Fatalf("read: %v", err) + } + if string(content) != "# a pre-existing, possibly hand-edited auto.md\n" { + t.Fatal("pre-manifest file content was changed, but it should have been adopted as-is") + } + + manifest := readManifest(dest) + if manifest.Files["auto.md"] == "" { + t.Fatal("expected a baseline hash to be recorded for the adopted file") + } +} From 4910272e0fe760d4134f5e7cb4b112b55f8d8e93 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:38:14 +0400 Subject: [PATCH 023/110] Replace whisper.cpp voice-snap's index matrix with an O(n) running sum _voiced_intervals built an (nf, 400) int64 index matrix to gather every 25ms window at once for its RMS calculation -- about 1GB of indices alone per hour of 16kHz audio, before even touching the gathered samples. A cumulative sum of squared samples gives the identical per-window RMS in O(n) memory. Verified against the old implementation on synthetic mixed voiced/silent audio across several seeds. --- backend/services/transcription_whispercpp.py | 11 +- tests/test_whispercpp_snap.py | 105 +++++++++++++++++++ 2 files changed, 114 insertions(+), 2 deletions(-) diff --git a/backend/services/transcription_whispercpp.py b/backend/services/transcription_whispercpp.py index 50ad1ffd..8a464994 100644 --- a/backend/services/transcription_whispercpp.py +++ b/backend/services/transcription_whispercpp.py @@ -106,8 +106,15 @@ def _voiced_intervals(wav_path: str, bridge: float = 0.3, thresh_ratio: float = if len(samples) < frame: return [] nf = 1 + (len(samples) - frame) // hop - idx = np.arange(nf)[:, None] * hop + np.arange(frame)[None, :] - rms = np.sqrt((samples[idx] ** 2).mean(axis=1)) + # The old (nf, frame) index matrix gathered every window at once — + # ~1GB of int64 indices alone for a 1-hour 16kHz file, before even + # touching the gathered samples. A running sum of squares gives the same + # per-frame RMS in O(n) memory instead of O(nf * frame). + sq = samples * samples + csum = np.concatenate(([0.0], np.cumsum(sq, dtype=np.float32))) + starts = np.arange(nf) * hop + window_sums = csum[starts + frame] - csum[starts] + rms = np.sqrt(np.maximum(window_sums, 0.0) / frame) peak = float(rms.max()) if peak <= 0: return [] diff --git a/tests/test_whispercpp_snap.py b/tests/test_whispercpp_snap.py index 9f09861c..2c254e93 100644 --- a/tests/test_whispercpp_snap.py +++ b/tests/test_whispercpp_snap.py @@ -5,6 +5,7 @@ import math import os +import random import struct import sys import tempfile @@ -16,9 +17,57 @@ if BACKEND_ROOT not in sys.path: sys.path.insert(0, BACKEND_ROOT) +import numpy as np + from services.transcription_whispercpp import _snap_words_to_voiced, _voiced_intervals +def _voiced_intervals_quadratic_reference(wav_path, bridge=0.3, thresh_ratio=0.07): + """The pre-fix implementation: gathers every window into an (nf, frame) + matrix before taking the RMS. Kept here only as a correctness oracle for + the O(n) rewrite in transcription_whispercpp._voiced_intervals — it must + never run on real audio, only the short synthetic clips in this test.""" + import wave as wave_mod + + try: + w = wave_mod.open(wav_path, "rb") + sr, width, n = w.getframerate(), w.getsampwidth(), w.getnframes() + raw = w.readframes(n) + w.close() + except Exception: + return [] + if width != 2 or sr <= 0 or not raw: + return [] + samples = np.frombuffer(raw, dtype=np.int16).astype(np.float32) + hop = max(1, int(sr * 0.010)) + frame = max(hop, int(sr * 0.025)) + if len(samples) < frame: + return [] + nf = 1 + (len(samples) - frame) // hop + idx = np.arange(nf)[:, None] * hop + np.arange(frame)[None, :] + rms = np.sqrt((samples[idx] ** 2).mean(axis=1)) + peak = float(rms.max()) + if peak <= 0: + return [] + voiced = rms > thresh_ratio * peak + intervals, start = [], None + for i, v in enumerate(voiced): + if v and start is None: + start = i * hop / sr + elif not v and start is not None: + intervals.append([start, i * hop / sr]) + start = None + if start is not None: + intervals.append([start, nf * hop / sr]) + merged = [] + for iv in intervals: + if merged and iv[0] - merged[-1][1] <= bridge: + merged[-1][1] = iv[1] + else: + merged.append(iv) + return merged + + def _make_wav(path, voiced_s=1.0, silence_s=1.0, sr=16000): w = wave.open(path, "wb") w.setnchannels(1) @@ -69,5 +118,61 @@ def test_all_silence_leaves_words_unchanged(self): os.remove(silent) +class VoicedIntervalsMatchesOldImplementationTests(unittest.TestCase): + """The O(n) running-sum rewrite has to produce the same voiced intervals + as the O(nf * frame) index-matrix version it replaces, on audio shaped + like what it actually has to handle: several voiced/silent transitions, + not just one clean on/off.""" + + def _make_mixed_wav(self, path, sr=16000, seed=0): + rng = random.Random(seed) + w = wave.open(path, "wb") + w.setnchannels(1) + w.setsampwidth(2) + w.setframerate(sr) + buf = bytearray() + # Alternating voiced tones and silence of varying length, plus a + # little noise floor during "silence" so it isn't a degenerate + # all-zero case. + for seg_i in range(6): + dur = 0.3 + 0.1 * seg_i + n = int(sr * dur) + if seg_i % 2 == 0: + for i in range(n): + val = int(7000 * math.sin(2 * math.pi * (180 + seg_i * 20) * i / sr)) + buf += struct.pack(" Date: Sat, 3 Oct 2026 16:39:07 +0400 Subject: [PATCH 024/110] Time multi-segment clip captions from each part's probed duration --- backend/services/clip_generator.py | 23 ++++++++--- backend/services/video_cut.py | 36 ++++++++++++++--- tests/test_video_cut.py | 63 ++++++++++++++++++++++++++++-- 3 files changed, 108 insertions(+), 14 deletions(-) diff --git a/backend/services/clip_generator.py b/backend/services/clip_generator.py index 34d8327e..84255ad5 100644 --- a/backend/services/clip_generator.py +++ b/backend/services/clip_generator.py @@ -1092,9 +1092,10 @@ def generate_clip( progress_callback(10, msg) segment_path = os.path.join(work_dir, "segment.mp4") + part_durations: Optional[list[float]] = None with timed("render", "cut", segments=len(keep_segments) if keep_segments else 1): if keep_segments and len(keep_segments) > 1: - cut_multi_segment(video_path, segment_path, keep_segments) + _, part_durations = cut_multi_segment(video_path, segment_path, keep_segments) else: cut_segment(video_path, segment_path, start_second, end_second) @@ -1103,24 +1104,36 @@ def generate_clip( if keep_segments and len(keep_segments) > 1 and transcript_words: remapped_words = [] cumulative_t = 0.0 - for seg in keep_segments: + for i, seg in enumerate(keep_segments): seg_words = [ w for w in transcript_words if w["end"] > seg["start"] and w["start"] < seg["end"] ] - seg_duration = seg["end"] - seg["start"] + # Each part is encoded separately before the concat, and an + # encoder snaps a cut to whole frames — its real duration is + # typically a few ms off the requested end - start. Advancing + # cumulative_t by the planned length instead of the probed + # one drifts captions further out of sync with every segment + # concatenated in. Fall back to the planned length only if + # the part couldn't be probed. + requested_duration = seg["end"] - seg["start"] + actual_duration = ( + part_durations[i] + if part_durations and part_durations[i] > 0 + else requested_duration + ) for w in seg_words: # Clamp to segment bounds to avoid negative/overflow timestamps # for words that straddle a segment boundary remapped_start = max(0, cumulative_t + (w["start"] - seg["start"])) - remapped_end = min(cumulative_t + seg_duration, cumulative_t + (w["end"] - seg["start"])) + remapped_end = min(cumulative_t + actual_duration, cumulative_t + (w["end"] - seg["start"])) if remapped_end > remapped_start: remapped_words.append({ **w, "start": round(remapped_start, 3), "end": round(remapped_end, 3), }) - cumulative_t += seg_duration + cumulative_t += actual_duration crop_words = remapped_words crop_clip_start = 0 caption_time_offset = 0 diff --git a/backend/services/video_cut.py b/backend/services/video_cut.py index ed08b1c0..7c2786b2 100644 --- a/backend/services/video_cut.py +++ b/backend/services/video_cut.py @@ -49,33 +49,57 @@ def cut_segment( return output_path +def probe_duration(path: str) -> float: + """Read media duration in seconds (best effort, 0.0 if unreadable).""" + try: + cmd = [ + "ffprobe", "-v", "error", + "-show_entries", "format=duration", + "-of", "default=nk=1:nw=1", + path, + ] + result = proc_run(cmd, timeout=10, check=False) + if result.returncode != 0: + return 0.0 + return float((result.stdout or "0").strip() or 0.0) + except Exception: + return 0.0 + + def cut_multi_segment( input_path: str, output_path: str, segments: list[dict], -) -> str: +) -> tuple[str, list[float]]: """Cut multiple time ranges and concatenate them seamlessly. segments: [{"start": 10.5, "end": 25.0}, {"start": 30.2, "end": 45.0}] - Each segment is cut individually with frame-accurate encoding, - then concatenated with stream copy (matching codecs means no - re-encode needed). + Each segment is cut individually with frame-accurate encoding, then + concatenated with stream copy (matching codecs means no re-encode + needed). Returns (output_path, part_durations): the probed duration of + each encoded part, not the requested one. An encoder snaps a cut to + whole frames, so the actual part is typically a few milliseconds off + the requested end - start; captions timed from the requested length + instead of the probed one drift further with every cut concatenated in. """ if len(segments) == 1: - return cut_segment( + out = cut_segment( input_path, output_path, segments[0]["start"], segments[0]["end"] ) + return out, [probe_duration(out)] work_dir = os.path.dirname(output_path) or "." part_paths: list[str] = [] concat_file = os.path.join(work_dir, "_concat_parts.txt") try: + part_durations: list[float] = [] for i, seg in enumerate(segments): part_path = os.path.join(work_dir, f"_part_{i}.mp4") cut_segment(input_path, part_path, seg["start"], seg["end"]) part_paths.append(part_path) + part_durations.append(probe_duration(part_path)) with open(concat_file, "w", encoding="utf-8") as f: for p in part_paths: @@ -93,7 +117,7 @@ def cut_multi_segment( if result.returncode != 0: raise RuntimeError(f"FFmpeg concat failed: {result.stderr[-500:]}") - return output_path + return output_path, part_durations finally: for p in part_paths: diff --git a/tests/test_video_cut.py b/tests/test_video_cut.py index 12d11e16..e8f16f3a 100644 --- a/tests/test_video_cut.py +++ b/tests/test_video_cut.py @@ -51,12 +51,15 @@ def test_raises_on_failure(self): class CutMultiSegmentTests(unittest.TestCase): def test_single_segment_delegates_to_cut_segment(self): - with mock.patch.object(video_cut, "cut_segment", return_value="/out.mp4") as cs: - out = video_cut.cut_multi_segment( + with mock.patch.object(video_cut, "cut_segment", return_value="/out.mp4") as cs, \ + mock.patch.object(video_cut, "probe_duration", return_value=5.0) as pd: + out, durations = video_cut.cut_multi_segment( "/in.mp4", "/out.mp4", [{"start": 0, "end": 5}] ) self.assertEqual(out, "/out.mp4") + self.assertEqual(durations, [5.0]) cs.assert_called_once_with("/in.mp4", "/out.mp4", 0, 5) + pd.assert_called_once_with("/out.mp4") def test_multi_segment_invokes_cut_then_concat(self): # Two-segment flow: two cut_segment calls + one proc_run for concat @@ -75,7 +78,7 @@ def fake_cut(input_path, out_path, start, end): ) as cs, mock.patch.object( video_cut, "proc_run", return_value=_ok() ) as mocked: - out = video_cut.cut_multi_segment( + out, durations = video_cut.cut_multi_segment( "/in.mp4", out_path, [ @@ -84,6 +87,7 @@ def fake_cut(input_path, out_path, start, end): ], ) self.assertEqual(out, out_path) + self.assertEqual(len(durations), 2) self.assertEqual(cs.call_count, 2) # concat call cmd = mocked.call_args.args[0] @@ -275,5 +279,58 @@ def test_verify_full_decode_flags_truncated_file(self): self.assertIsNotNone(video_cut.verify_full_decode(self.truncated)) +@unittest.skipUnless( + shutil.which("ffmpeg") and shutil.which("ffprobe"), "ffmpeg/ffprobe not installed" +) +class MultiSegmentPartDurationTests(unittest.TestCase): + """Captions for a multi-cut clip are timed by summing part durations + (clip_generator.py). Each part's actual encoded length can be a few ms + off the requested end - start, and that drift compounds across cuts — + so the probed durations cut_multi_segment now returns must track the + real concatenated output, not the planned request.""" + + @classmethod + def setUpClass(cls): + cls.tmpdir = tempfile.mkdtemp(prefix="podcli-multiseg-test-") + cls.src = os.path.join(cls.tmpdir, "src.mp4") + subprocess.run( + [ + "ffmpeg", "-y", "-loglevel", "error", + "-f", "lavfi", "-i", "testsrc=size=64x64:rate=30:duration=20", + "-f", "lavfi", "-i", "sine=frequency=440:duration=20", + "-c:v", "libx264", "-pix_fmt", "yuv420p", "-c:a", "aac", + "-g", "30", "-keyint_min", "30", "-sc_threshold", "0", + cls.src, + ], + check=True, capture_output=True, + ) + + @classmethod + def tearDownClass(cls): + shutil.rmtree(cls.tmpdir, ignore_errors=True) + + def test_sum_of_part_durations_matches_the_concatenated_output(self): + # Deliberately off-grid, irregular lengths so frame-snap rounding + # doesn't cancel out across segments. + segments = [ + {"start": 0.13, "end": 0.46}, + {"start": 1.02, "end": 1.58}, + {"start": 2.21, "end": 2.69}, + {"start": 3.07, "end": 3.91}, + {"start": 4.44, "end": 4.77}, + {"start": 5.10, "end": 5.88}, + ] + out_path = os.path.join(self.tmpdir, "multi.mp4") + _, part_durations = video_cut.cut_multi_segment(self.src, out_path, segments) + + self.assertEqual(len(part_durations), len(segments)) + actual_total = video_cut.probe_duration(out_path) + + # One frame at 30fps; the probed per-part durations summed should + # land within a frame of the real concatenated output. + frame = 1.0 / 30 + self.assertLessEqual(abs(sum(part_durations) - actual_total), frame) + + if __name__ == "__main__": unittest.main() From 9663ec09d12005e264029dbfedc859f8984d5676 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:39:09 +0400 Subject: [PATCH 025/110] Refuse a cloud pull against a moved timeline and keep removal reasons --- backend/services/multicam.py | 13 +++++++++++++ backend/services/multicam_cloud.py | 15 +++++++++++++-- tests/test_multicam_cloud.py | 26 ++++++++++++++++++++++++-- 3 files changed, 50 insertions(+), 4 deletions(-) diff --git a/backend/services/multicam.py b/backend/services/multicam.py index 0d50597c..029cf568 100644 --- a/backend/services/multicam.py +++ b/backend/services/multicam.py @@ -850,6 +850,19 @@ def needs_sync(session: MulticamSession) -> bool: return any(_in_use(session, s) and not s.virtual and not s.synced for s in session.sources) +def sync_basis_signature(session: MulticamSession) -> str: + """Fingerprint of where every in-use source currently sits on the timeline. + + A cut or removal made in an external editor (the podcli cloud editor) + references timeline seconds measured against this placement. If sync, + a nudge, or a re-map moves anything after the edit was sent out, applying + that cut back silently lands on the wrong footage: this lets callers + detect that before it happens. + """ + feeds = sorted((s.id, s.offset, s.speed) for s in session.sources if _in_use(session, s) and not s.virtual) + return hashlib.sha1(json.dumps(feeds, default=str).encode()).hexdigest()[:16] + + def _in_use(session: MulticamSession, s: Source) -> bool: """Mapped to something, or the recording a tile in use is cut from (even when the whole frame is ignored).""" return s.role != "ignore" or any(v.parent == s.id and v.role != "ignore" for v in session.sources) diff --git a/backend/services/multicam_cloud.py b/backend/services/multicam_cloud.py index 422ce202..df4eb45e 100644 --- a/backend/services/multicam_cloud.py +++ b/backend/services/multicam_cloud.py @@ -110,7 +110,7 @@ def push(session: mc.MulticamSession, *, model_size: str = "base", engine: Optio _request("POST", f"/v1/multicam/{opened['id']}/hybrid-complete", {"files": stored}) latest = mc.MulticamSession.load(session.session_id) - latest.cloud = {"id": opened["id"], "url": editor_url(opened["id"])} + latest.cloud = {"id": opened["id"], "url": editor_url(opened["id"]), "basis": mc.sync_basis_signature(latest)} latest.save() _emit(progress_callback, 100, "In the cloud editor") return latest @@ -134,6 +134,14 @@ def pull(session: mc.MulticamSession) -> mc.MulticamSession: edit_id = (session.cloud or {}).get("id") if not edit_id: raise MulticamCloudError("this edit was never sent to podcli cloud. Send it with --cloud first") + pushed_basis = (session.cloud or {}).get("basis") + current_basis = mc.sync_basis_signature(session) + if pushed_basis and pushed_basis != current_basis: + raise MulticamCloudError( + "sources moved on the timeline (re-synced, nudged, or re-mapped) since this edit was sent to the " + "podcli cloud editor. Its cuts and removals were made against the old positions and would land on " + "the wrong footage now. Send it again with --cloud before pulling." + ) edit = _request("GET", f"/v1/multicam/{edit_id}/edit") state = edit.get("state") or {} @@ -148,5 +156,8 @@ def pull(session: mc.MulticamSession) -> mc.MulticamSession: if cuts: session = mc.set_cuts(session, cuts) if isinstance(edit.get("removals"), list): - session = mc.set_removals(session, [{"start": r["start"], "end": r["end"]} for r in edit["removals"]]) + session = mc.set_removals(session, [ + {"start": r["start"], "end": r["end"], **({"reason": r["reason"]} if r.get("reason") else {})} + for r in edit["removals"] + ]) return session diff --git a/tests/test_multicam_cloud.py b/tests/test_multicam_cloud.py index aed15111..0a92149a 100644 --- a/tests/test_multicam_cloud.py +++ b/tests/test_multicam_cloud.py @@ -116,7 +116,7 @@ def test_sends_only_previews_and_renders_the_cloud_cut_here(episode, cloud, monk cloud["edit"] = { "state": {**sent["engine"], "people": [{"id": "nika", "name": "Nika"}, {"id": "ana", "name": "Ana Smith"}]}, "cuts": [{"start": start, "end": end, "source_id": one}], - "removals": [{"start": 10, "end": 12}], + "removals": [{"start": 10, "end": 12, "reason": "retake"}], "look": "warm", "revision": 4, } @@ -126,10 +126,32 @@ def test_sends_only_previews_and_renders_the_cloud_cut_here(episode, cloud, monk assert "Pulled the cloud edit: 1 shots" in out session = mc.MulticamSession.load(session.session_id) assert [p.name for p in session.people] == ["Nika", "Ana Smith"] - assert session.look == "warm" and session.removals == [{"start": 10.0, "end": 12.0}] + assert session.look == "warm" + assert session.removals == [{"start": 10.0, "end": 12.0, "reason": "retake"}] assert session.outputs["duration"] == pytest.approx(end - start - 2, abs=0.05) def test_pull_needs_an_edit_that_was_sent(sandbox): with pytest.raises(multicam_cloud.MulticamCloudError, match="No multicam edit on this computer"): multicam_cloud.resolve("11111111-2222-3333-4444-555555555555") + + +@pytest.mark.skipif(not shutil.which("ffmpeg"), reason="ffmpeg not installed") +def test_pull_refuses_when_sources_moved_since_the_edit_was_sent(episode, cloud, monkeypatch, capsys): + words = [{"start": 1.0, "end": 1.4, "text": "Hello", "person": "nika"}] + monkeypatch.setattr(mc, "transcript", lambda *a, **k: {"words": words}) + code = run_cli(monkeypatch, str(episode), "--people", "Nika, Ana", "--set", "cam_one=camera:nika", + "--set", "cam_two=camera:ana", "--cloud", "-y") + assert code == 0, capsys.readouterr().out + + session = mc.MulticamSession.load(mc.list_sessions()[0]["session_id"]) + one = next(s for s in session.sources if s.path.endswith("cam_one.mp4")) + cloud["edit"] = {"state": {}, "cuts": [], "removals": [], "revision": 1} + + # Nudging a source after the push moves it on the timeline, so the cuts + # the cloud editor made against the old position would land on the wrong + # footage if pulled. + mc.update_mapping(session, {"sources": [{"id": one.id, "nudge": 1.5}]}) + session = mc.MulticamSession.load(session.session_id) + with pytest.raises(multicam_cloud.MulticamCloudError, match="moved on the timeline"): + multicam_cloud.pull(session) From 7b38e6f68a7eae1178c468ee7b8a5e938085d6e5 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:42:12 +0400 Subject: [PATCH 026/110] Publish a render's video, stems and record atomically --- backend/services/multicam.py | 28 +++++++++++++++++++++++----- tests/test_multicam.py | 27 +++++++++++++++++++++++++++ 2 files changed, 50 insertions(+), 5 deletions(-) diff --git a/backend/services/multicam.py b/backend/services/multicam.py index 029cf568..77df8e6e 100644 --- a/backend/services/multicam.py +++ b/backend/services/multicam.py @@ -2064,19 +2064,37 @@ def run(item): "-map", "0:v:0", "-map", "1:a:0", "-c:v", "copy", "-c:a", "aac", "-b:a", "192k", "-shortest", "-movflags", "+faststart", str(partial), ], timeout=3600, check=True) - video = out_dir / "episode.mp4" - # shutil.move falls back to copy+delete when work and output sit on different volumes. - shutil.move(str(partial), str(video)) stem_paths = [] if stems: _emit(progress_callback, 96, "Writing separate mic tracks") - stem_paths = _render_stems(session, out_dir, start, end - start, splice) + # Built into `work`, not `out_dir`: if this fails, nothing at the + # canonical output path changes, and the video built above is + # discarded along with it instead of being published alone. + stem_paths = _render_stems(session, work, start, end - start, splice) + + # Every piece rendered; publish video, stems and the session record + # together. A crash between these renames can only ever leave either + # the previous complete render or this one in place, never a mix. + video = out_dir / "episode.mp4" + tmp_video = video.with_name(video.name + ".publishing") + # shutil.move falls back to copy+delete when work and output sit on different volumes. + shutil.move(str(partial), str(tmp_video)) + pending_stems = [] + for stem in stem_paths: + dest = out_dir / Path(stem).name + tmp_stem = dest.with_name(dest.name + ".publishing") + shutil.move(stem, str(tmp_stem)) + pending_stems.append((tmp_stem, dest)) + + os.replace(str(tmp_video), str(video)) + for tmp_stem, dest in pending_stems: + os.replace(str(tmp_stem), str(dest)) session.outputs = { **session.outputs, "video": str(video), - "stems": stem_paths, + "stems": [str(dest) for _, dest in pending_stems], "duration": round(duration, 3), "render_key": key, } diff --git a/tests/test_multicam.py b/tests/test_multicam.py index 99127f0f..91c06145 100644 --- a/tests/test_multicam.py +++ b/tests/test_multicam.py @@ -252,6 +252,33 @@ def length(path, stream): assert abs(mc.sync_session(session, force=True).source(wide.id).offset - 2.2) < 0.005 +@pytest.mark.skipif(not shutil.which("ffmpeg"), reason="ffmpeg not installed") +def test_a_stems_failure_does_not_strand_a_video_with_no_outputs_record(episode, monkeypatch): + session = mc.new_session(folder=str(episode), people=["Nika", "Ana"]) + mc.update_mapping(session, {"sources": [ + {"id": s.id, "role": "camera", "person": "nika" if "one" in os.path.basename(s.path) else "ana"} + for s in session.sources if s.kind == "video" + ]}) + session = mc.plan_session(mc.sync_session(session)) + out_dir = mc._output_dir(session) + + monkeypatch.setattr(mc, "_render_stems", lambda *a, **k: (_ for _ in ()).throw(RuntimeError("boom"))) + with pytest.raises(RuntimeError, match="boom"): + mc.render_session(session) + + # Nothing half-rendered should land at the canonical output path, and the + # session must not record a video it never finished publishing. + assert not (out_dir / "episode.mp4").exists() + assert not any(out_dir.glob("*.publishing")) + session = mc.MulticamSession.load(session.session_id) + assert not session.outputs.get("video") + + monkeypatch.undo() + outputs = mc.render_session(session) + assert os.path.exists(outputs["video"]) and len(outputs["stems"]) == 2 + assert not any(out_dir.glob("*.publishing")) + + @pytest.mark.skipif(not shutil.which("ffmpeg"), reason="ffmpeg not installed") def test_sync_measures_clock_drift(sandbox): folder = sandbox / "drift" From 580abb12aa2a5c763e3f2401eaadb17ab618f815 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:42:52 +0400 Subject: [PATCH 027/110] Give Codex the same MCP and PodStack install path as Claude MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit podcli setup and podcli mcp install now also register the MCP server with Codex (codex mcp add podcli -- mcp) when the codex CLI is on PATH, matching the existing Claude Code registration. installCodexSkills installs every PodStack command as a Codex skill under ~/.codex/skills//SKILL.md (confirmed against a local codex install — that's the real path it reads, not .codex/prompts), reusing the same install-manifest mechanism as .claude/commands so upgrades refresh unmodified skills and leave user edits alone. podstack.Run() no longer falls back from Claude to Codex on a nonzero exit code — only when Claude isn't on PATH at all. Running the same destructive workflow twice under two different agents because the first one's exit code was nonzero (e.g. the user cancelled) was never the intent. Add a root AGENTS.md pointing at CLAUDE.md and AGENTS.podstack.md, and correct AGENTS.podstack.md's host-compatibility table, which claimed a .codex/prompts install path nothing wrote. Add a Codex section to the Web UI's MCP setup page alongside Claude Desktop and Claude Code. --- AGENTS.md | 17 +++ AGENTS.podstack.md | 20 ++-- cli/internal/podstack/podstack.go | 137 ++++++++++++++++++++++++- cli/internal/podstack/podstack_test.go | 104 +++++++++++++++++++ cli/main.go | 82 +++++++++++++-- src/ui/client/McpSetupPage.tsx | 16 +++ 6 files changed, 355 insertions(+), 21 deletions(-) create mode 100644 AGENTS.md diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 00000000..c7b0d651 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,17 @@ +# AGENTS.md + +This file exists for coding agents (OpenAI Codex, opencode, Aider, Cursor Agent, +and others) that read `AGENTS.md` by convention. + +- `CLAUDE.md` is the primary instruction document: project layout, the MCP tool + table, the knowledge base, and the quality gate all live there and are not + repeated here. +- `AGENTS.podstack.md` covers cross-tool PodStack usage: how each host runs the + content-production commands (`/plan-episode`, `/process-transcript`, + `/generate-titles`, and the rest), and where each host installs them. +- `.claude/commands/*.md` are the PodStack command sources. Claude Code reads + them directly from that path; `podcli auto` (and the other PodStack + commands) also installs them as Codex skills under `~/.codex/skills/` when + the `codex` CLI is present. + +Start with `CLAUDE.md`, then `AGENTS.podstack.md` for command-by-command detail. diff --git a/AGENTS.podstack.md b/AGENTS.podstack.md index a1bae009..d74f6b7c 100644 --- a/AGENTS.podstack.md +++ b/AGENTS.podstack.md @@ -8,7 +8,7 @@ PodStack turns your AI tool into a podcast content team: Episode Architect, Cont ## How to use -Each skill below is a self-contained instruction file in `commands/` (or `.claude/commands/`, `.codex/prompts/`, `.cursor/rules/`, `.opencode/commands/`, depending on which host installed it). +Each skill below is a self-contained instruction file in `commands/` (or `.claude/commands/`, `~/.codex/skills//SKILL.md`, `.cursor/rules/`, `.opencode/commands/`, depending on which host installed it). **To run a skill:** ask your agent to "run the [skill-name] skill" or invoke its slash command (`/[skill-name]`) where supported. The agent opens the corresponding file and follows it step by step. @@ -114,16 +114,16 @@ Skill files read the 14 knowledge files at `.podcli/knowledge/`; the full file t PodStack ships one source-of-truth (`commands/`) and installs to the right location for each tool: -| Host | Install location | Primary doc | -|------|-----------------|-------------| -| Claude Code | `.claude/commands/*.md` | `CLAUDE.md` | -| OpenAI Codex | `.codex/prompts/*.md` | `AGENTS.podstack.md` (this file) | -| Cursor | `.cursor/rules/*.mdc` | `AGENTS.podstack.md` | -| opencode | `.opencode/commands/*.md` | `AGENTS.podstack.md` | -| Generic | `commands/*.md` | `AGENTS.podstack.md` | +| Host | Install location | Installed by | Primary doc | +|------|-----------------|--------------|-------------| +| Claude Code | `.claude/commands/*.md` (per project) | `podcli auto` / any PodStack command | `CLAUDE.md` | +| OpenAI Codex | `~/.codex/skills//SKILL.md` (global) | `podcli auto` / any PodStack command, when the `codex` CLI is on PATH | `AGENTS.podstack.md` (this file) | +| Cursor | `.cursor/rules/*.mdc` | not automated yet — copy by hand | `AGENTS.podstack.md` | +| opencode | `.opencode/commands/*.md` | not automated yet — copy by hand | `AGENTS.podstack.md` | +| Generic | `commands/*.md` | not automated yet — copy by hand | `AGENTS.podstack.md` | -These command files ship with podcli; place the set for your tool (left column) in -its command dir. See `README.md` for per-host usage examples. +Claude and Codex installs are automatic and kept in sync on upgrade; see `README.md` +for per-host usage examples and manual steps for the other hosts. --- diff --git a/cli/internal/podstack/podstack.go b/cli/internal/podstack/podstack.go index 5b07d6cf..7e4a0e05 100644 --- a/cli/internal/podstack/podstack.go +++ b/cli/internal/podstack/podstack.go @@ -211,6 +211,127 @@ func installCommands(project string) (installReport, error) { return report, nil } +// codexSkillsDir is where the installed Codex CLI reads skills from +// (confirmed against a local `codex` install: ~/.codex/skills//SKILL.md, +// one directory per skill). It is global, not per-project, unlike +// .claude/commands. +func codexSkillsDir() (string, error) { + home, err := os.UserHomeDir() + if err != nil { + return "", err + } + return filepath.Join(home, ".codex", "skills"), nil +} + +// frontmatterDescription pulls the `description:` field out of a command +// file's YAML frontmatter without a YAML dependency — the format here is a +// fixed, simple `key: value` list. +func frontmatterDescription(raw string) string { + lines := strings.Split(raw, "\n") + inFrontmatter := false + for i, line := range lines { + if i == 0 && strings.TrimSpace(line) == "---" { + inFrontmatter = true + continue + } + if !inFrontmatter { + return "" + } + if strings.TrimSpace(line) == "---" { + return "" + } + if rest, ok := strings.CutPrefix(line, "description:"); ok { + return strings.TrimSpace(rest) + } + } + return "" +} + +// codexSkillContent rewraps a PodStack command file as a Codex SKILL.md: +// same body, frontmatter translated from this project's `description:` / +// `argument-hint:` shape to the `name:` / `description:` shape codex reads. +func codexSkillContent(name, raw string) string { + desc := frontmatterDescription(raw) + if desc == "" { + desc = "PodStack command: " + name + } + body := raw + if end := strings.Index(raw, "\n---\n"); strings.HasPrefix(raw, "---\n") && end != -1 { + body = strings.TrimPrefix(raw[end+len("\n---\n"):], "\n") + } + return fmt.Sprintf("---\nname: %s\ndescription: %s\n---\n\n%s", name, desc, body) +} + +// installCodexSkills writes every PodStack command as a Codex skill under +// ~/.codex/skills//SKILL.md. Skills are global (not per-project like +// .claude/commands), so this always targets the user's home directory +// regardless of which project podcli was run from. Like installCommands, a +// manifest tracks what podcli last wrote so an upgrade can refresh +// unmodified skills and leave user edits alone. +func installCodexSkills() error { + skillsDir, err := codexSkillsDir() + if err != nil { + return err + } + if err := os.MkdirAll(skillsDir, 0o755); err != nil { + return err + } + manifest := readManifest(skillsDir) + changed := false + + err = fs.WalkDir(commands, "commands", func(p string, d fs.DirEntry, err error) error { + if err != nil || d.IsDir() { + return err + } + name := strings.TrimSuffix(filepath.Base(p), ".md") + raw, err := commands.ReadFile(p) + if err != nil { + return err + } + content := []byte(codexSkillContent(name, string(raw))) + contentHash := sha256Hex(content) + + skillDir := filepath.Join(skillsDir, name) + target := filepath.Join(skillDir, "SKILL.md") + current, statErr := os.ReadFile(target) + if statErr != nil { + if err := os.MkdirAll(skillDir, 0o755); err != nil { + return err + } + if err := os.WriteFile(target, content, 0o644); err != nil { + return err + } + manifest.Files[name] = contentHash + changed = true + return nil + } + + lastInstalledHash, tracked := manifest.Files[name] + currentHash := sha256Hex(current) + if !tracked { + manifest.Files[name] = currentHash + changed = true + return nil + } + if currentHash != lastInstalledHash || currentHash == contentHash { + return nil // user-modified, or already current + } + if err := os.WriteFile(target, content, 0o644); err != nil { + return err + } + manifest.Files[name] = contentHash + changed = true + return nil + }) + if err != nil { + return err + } + if changed { + return writeManifest(skillsDir, manifest) + } + return nil +} + // Run launches the agent for cmd with the remaining args as the slash-command // arguments. Engine selection: --claude / --codex / --ai , else // PODCLI_AI, else auto (Claude preferred, Codex fallback). @@ -267,6 +388,12 @@ func Run(cmd string, args []string) int { claudeBin, _ := exec.LookPath("claude") codexBin, _ := exec.LookPath("codex") + if codexBin != "" { + if err := installCodexSkills(); err != nil { + fmt.Fprintf(os.Stderr, " %swarning:%s could not install Codex skills: %v\n", colYellow, colReset, err) + } + } + if engine == "codex" && codexBin == "" { fmt.Fprintf(os.Stderr, "\n %sCodex not found in PATH.%s\n Install it, then run:\n %scodex --cd %q %q%s\n\n", colBold, colReset, colAccent, project, codexPrompt, colReset) return 1 @@ -276,12 +403,14 @@ func Run(cmd string, args []string) int { return 1 } + // Fall back to Codex only when Claude isn't installed at all. Claude + // exiting nonzero (the user cancelled, a tool failed mid-run, etc.) is + // not a reason to silently relaunch the whole workflow under a + // different agent — it previously was, which could run the same + // destructive command twice under two different engines. if engine != "codex" && claudeBin != "" { fmt.Fprintf(os.Stderr, "\n %s▶%s Launching Claude Code with: %s%s%s\n %scwd: %s%s\n\n", colGreen, colReset, colAccent, prompt, colReset, colDim, project, colReset) - if code := runIn(project, claudeBin, prompt); code == 0 || engine == "claude" || codexBin == "" { - return code - } - fmt.Fprintf(os.Stderr, "\n %s⚠%s Claude exited nonzero; trying Codex...\n", colYellow, colReset) + return runIn(project, claudeBin, prompt) } if codexBin == "" { diff --git a/cli/internal/podstack/podstack_test.go b/cli/internal/podstack/podstack_test.go index 7ac5d372..c4e99bf3 100644 --- a/cli/internal/podstack/podstack_test.go +++ b/cli/internal/podstack/podstack_test.go @@ -3,9 +3,35 @@ package podstack import ( "os" "path/filepath" + "runtime" + "strings" "testing" ) +// withHome points os.UserHomeDir() (and HOME/USERPROFILE) at a temp dir for +// the duration of the test, so installCodexSkills never touches the real +// ~/.codex/skills. +func withHome(t *testing.T) string { + t.Helper() + home := t.TempDir() + envVar := "HOME" + if runtime.GOOS == "windows" { + envVar = "USERPROFILE" + } + old, hadOld := os.LookupEnv(envVar) + if err := os.Setenv(envVar, home); err != nil { + t.Fatalf("setenv: %v", err) + } + t.Cleanup(func() { + if hadOld { + os.Setenv(envVar, old) + } else { + os.Unsetenv(envVar) + } + }) + return home +} + func TestInstallCommandsWritesEveryFileOnFirstRun(t *testing.T) { project := t.TempDir() @@ -174,3 +200,81 @@ func TestInstallCommandsAdoptsPreManifestFilesWithoutOverwriting(t *testing.T) { t.Fatal("expected a baseline hash to be recorded for the adopted file") } } + +func TestFrontmatterDescriptionExtractsTheDescriptionField(t *testing.T) { + raw := "---\ndescription: Full pipeline from transcript to publish-ready package\nallowed-tools: Read\n---\n\n# body\n" + if got := frontmatterDescription(raw); got != "Full pipeline from transcript to publish-ready package" { + t.Fatalf("got %q", got) + } +} + +func TestFrontmatterDescriptionIsEmptyWithoutFrontmatter(t *testing.T) { + if got := frontmatterDescription("# just a body\n"); got != "" { + t.Fatalf("got %q, want empty", got) + } +} + +func TestCodexSkillContentTranslatesFrontmatterAndKeepsBody(t *testing.T) { + raw := "---\ndescription: does a thing\nallowed-tools: Read\n---\n\n# /auto\n\nbody text\n" + got := codexSkillContent("auto", raw) + if !strings.HasPrefix(got, "---\nname: auto\ndescription: does a thing\n---\n\n") { + t.Fatalf("unexpected frontmatter translation: %q", got) + } + if !strings.Contains(got, "# /auto\n\nbody text\n") { + t.Fatal("body was dropped or altered") + } +} + +func TestInstallCodexSkillsWritesOneSkillDirPerCommand(t *testing.T) { + withHome(t) + + if err := installCodexSkills(); err != nil { + t.Fatalf("installCodexSkills: %v", err) + } + + skillsDir, err := codexSkillsDir() + if err != nil { + t.Fatalf("codexSkillsDir: %v", err) + } + for _, name := range Names() { + skillFile := filepath.Join(skillsDir, name, "SKILL.md") + data, err := os.ReadFile(skillFile) + if err != nil { + t.Fatalf("%s: %v", skillFile, err) + } + if !strings.HasPrefix(string(data), "---\nname: "+name+"\n") { + t.Fatalf("%s: missing expected frontmatter, got: %q", skillFile, string(data)[:min(60, len(data))]) + } + } +} + +func TestInstallCodexSkillsLeavesUserEditsAlone(t *testing.T) { + withHome(t) + if err := installCodexSkills(); err != nil { + t.Fatalf("first install: %v", err) + } + skillsDir, err := codexSkillsDir() + if err != nil { + t.Fatalf("codexSkillsDir: %v", err) + } + target := filepath.Join(skillsDir, "auto", "SKILL.md") + installed, err := os.ReadFile(target) + if err != nil { + t.Fatalf("read installed skill: %v", err) + } + edited := append(append([]byte{}, installed...), []byte("\n\n")...) + if err := os.WriteFile(target, edited, 0o644); err != nil { + t.Fatalf("simulate user edit: %v", err) + } + + if err := installCodexSkills(); err != nil { + t.Fatalf("second install: %v", err) + } + after, err := os.ReadFile(target) + if err != nil { + t.Fatalf("read after reinstall: %v", err) + } + if string(after) != string(edited) { + t.Fatal("user-edited skill file was overwritten") + } +} diff --git a/cli/main.go b/cli/main.go index b615dfb4..24d54ce0 100644 --- a/cli/main.go +++ b/cli/main.go @@ -375,14 +375,24 @@ func setup(args []string) int { } if engine.MCPServer() != "" { if mcpRegisteredToSelf() { - fmt.Printf(" mcp: already registered\n") + fmt.Printf(" mcp: already registered with Claude Code\n") } else if _, err := exec.LookPath("claude"); err != nil { // Claude MCP registration is optional; Codex users do not need this. } else if err := registerMCPServer(); err != nil { - fmt.Fprintf(os.Stderr, " mcp: not registered (%v) - run `podcli mcp install`\n", err) + fmt.Fprintf(os.Stderr, " mcp: not registered with Claude Code (%v) - run `podcli mcp install`\n", err) } else { fmt.Printf(" mcp: registered with Claude Code\n") } + + if codexMCPRegisteredToSelf() { + fmt.Printf(" mcp: already registered with Codex\n") + } else if _, err := exec.LookPath("codex"); err != nil { + // Codex MCP registration is optional; Claude users do not need this. + } else if err := registerCodexMCPServer(); err != nil { + fmt.Fprintf(os.Stderr, " mcp: not registered with Codex (%v)\n", err) + } else { + fmt.Printf(" mcp: registered with Codex\n") + } } fmt.Println("Done.") return 0 @@ -419,14 +429,72 @@ func mcpRegisteredToSelf() bool { return err == nil && strings.Contains(string(out), self) } +// registerCodexMCPServer points Codex at this binary's `mcp` command. Unlike +// Claude's `mcp add`, `codex mcp add` overwrites an existing entry by name, +// so no remove-first step is needed to stay idempotent. +func registerCodexMCPServer() error { + codex, err := exec.LookPath("codex") + if err != nil { + return fmt.Errorf("Codex CLI not found on PATH") + } + self, err := os.Executable() + if err != nil { + return err + } + if out, err := exec.Command(codex, "mcp", "add", "podcli", "--", self, "mcp").CombinedOutput(); err != nil { + return fmt.Errorf("%v: %s", err, strings.TrimSpace(string(out))) + } + return nil +} + +func codexMCPRegisteredToSelf() bool { + codex, err := exec.LookPath("codex") + if err != nil { + return false + } + self, err := os.Executable() + if err != nil { + return false + } + out, err := exec.Command(codex, "mcp", "get", "podcli").CombinedOutput() + return err == nil && strings.Contains(string(out), self) +} + func mcpInstall() int { - if err := registerMCPServer(); err != nil { - self, _ := os.Executable() - fmt.Fprintf(os.Stderr, "podcli: %v\n", err) - fmt.Fprintf(os.Stderr, "Register manually: claude mcp add podcli -- %s mcp\n", self) + self, _ := os.Executable() + _, claudeErr := exec.LookPath("claude") + _, codexErr := exec.LookPath("codex") + registeredAny := false + failedAny := false + + if claudeErr == nil { + if err := registerMCPServer(); err != nil { + fmt.Fprintf(os.Stderr, "podcli: %v\n", err) + fmt.Fprintf(os.Stderr, "Register manually: claude mcp add podcli -- %s mcp\n", self) + failedAny = true + } else { + fmt.Println("Registered podcli MCP server with Claude Code.") + registeredAny = true + } + } + if codexErr == nil { + if err := registerCodexMCPServer(); err != nil { + fmt.Fprintf(os.Stderr, "podcli: %v\n", err) + fmt.Fprintf(os.Stderr, "Register manually: codex mcp add podcli -- %s mcp\n", self) + failedAny = true + } else { + fmt.Println("Registered podcli MCP server with Codex.") + registeredAny = true + } + } + + if !registeredAny && !failedAny { + fmt.Fprintln(os.Stderr, "podcli: neither Claude Code nor Codex CLI found on PATH") + return 1 + } + if failedAny { return 1 } - fmt.Println("Registered podcli MCP server with Claude Code.") return 0 } diff --git a/src/ui/client/McpSetupPage.tsx b/src/ui/client/McpSetupPage.tsx index b96999e9..b16f46cb 100644 --- a/src/ui/client/McpSetupPage.tsx +++ b/src/ui/client/McpSetupPage.tsx @@ -17,6 +17,7 @@ export default function McpSetupPage() { const [statusText, setStatusText] = useState("Checking…"); const desktopRef = useRef(null); const codeRef = useRef(null); + const codexRef = useRef(null); useEffect(() => { api("/integration-info") @@ -78,6 +79,21 @@ export default function McpSetupPage() {
podcli mcp install
+ +
+
Codex
+

+ podcli mcp install registers with Codex too when the codex CLI is on PATH. To + register by hand instead: +

+
+
+ terminal + codexRef.current?.innerText ?? ""} /> +
+
{`codex mcp add podcli -- node ${serverPath}`}
+
+
); } From de0920a5f2365d5e67215b603bb213c07003e2ef Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:44:11 +0400 Subject: [PATCH 028/110] Warn when a source's frame rate looks variable or had to be guessed --- backend/services/multicam.py | 25 ++++++++++++++++++++----- tests/test_multicam.py | 27 +++++++++++++++++++++++++++ 2 files changed, 47 insertions(+), 5 deletions(-) diff --git a/backend/services/multicam.py b/backend/services/multicam.py index 77df8e6e..5414412e 100644 --- a/backend/services/multicam.py +++ b/backend/services/multicam.py @@ -114,6 +114,7 @@ class Source: timecode: float = 0.0 # embedded start timecode in seconds; editors address media from here file_size: int = 0 # size at probe time; a mismatch on reopen means the file changed underneath us file_mtime_ns: int = 0 # mtime at probe time, nanosecond resolution + fps_warning: str = "" # set when the container's frame rate looks variable or had to be guessed role: str = "ignore" # "camera" | "mic" | "ignore" # camera: a person id or "wide"; mic: a person id, or "" for a shared room mic person: str = "" @@ -343,13 +344,26 @@ def probe_source(path: str) -> Source: if duration <= 0: raise RuntimeError(f"Could not read the duration of {os.path.basename(path)}") fps = 0.0 + fps_warning = "" if video: - num, _, den = str(video.get("avg_frame_rate") or video.get("r_frame_rate") or "0/1").partition("/") - try: - fps = float(num) / float(den or 1) if float(den or 1) else 0.0 - except ValueError: - fps = 0.0 + def rate(field: str) -> float: + num, _, den = str(video.get(field) or "0/1").partition("/") + try: + return float(num) / float(den or 1) if float(den or 1) else 0.0 + except ValueError: + return 0.0 + + avg_fps, r_fps = rate("avg_frame_rate"), rate("r_frame_rate") + fps = avg_fps or r_fps + # r_frame_rate is the stream's time base, a ceiling on how fast frames + # could appear; avg_frame_rate is what actually played out. A material + # gap between them means some frames held longer than others, so a shot + # cut can't assume a fixed grid: the frame at a given timestamp drifts. + if avg_fps and r_fps and abs(avg_fps - r_fps) / r_fps > 0.01: + fps_warning = (f"Variable frame rate detected ({avg_fps:.3g} fps average, " + f"{r_fps:.3g} fps max): cuts may drift by a frame or more.") if not 1 <= fps <= 240: + fps_warning = f"Could not read a usable frame rate ({fps or 'none'}); assuming 30 fps." fps = 30.0 sample_rate = int(audio.get("sample_rate") or 0) if audio else 0 stat = os.stat(path) @@ -368,6 +382,7 @@ def probe_source(path: str) -> Source: timecode=round(_timecode_seconds(info, fps, sample_rate), 6), file_size=stat.st_size, file_mtime_ns=stat.st_mtime_ns, + fps_warning=fps_warning, ) diff --git a/tests/test_multicam.py b/tests/test_multicam.py index 91c06145..9b35a36a 100644 --- a/tests/test_multicam.py +++ b/tests/test_multicam.py @@ -1117,3 +1117,30 @@ def test_timecode_seconds_handles_ntsc_pulldown_and_drop_frame(fps, tc, expected def test_timecode_seconds_falls_back_to_time_reference_without_embedded_timecode(): info = {"format": {"tags": {"time_reference": "48000"}}} assert mc._timecode_seconds(info, 29.97, 48000) == pytest.approx(1.0, abs=1e-6) + + +def _fake_probe(tmp_path, monkeypatch, video_stream): + path = tmp_path / "cam.mov" + path.write_bytes(b"0") + info = {"format": {"duration": "10.0", "tags": {}}, "streams": [{"codec_type": "video", **video_stream}]} + monkeypatch.setattr(mc, "get_video_info", lambda p: info) + return mc.probe_source(str(path)) + + +def test_probe_source_warns_on_variable_frame_rate(tmp_path, monkeypatch): + # avg_frame_rate (what actually played) is far below r_frame_rate (the + # stream's time base): frames held variable lengths. + src = _fake_probe(tmp_path, monkeypatch, {"avg_frame_rate": "24/1", "r_frame_rate": "60/1"}) + assert src.fps == pytest.approx(24.0) + assert "variable frame rate" in src.fps_warning.lower() + + +def test_probe_source_does_not_warn_on_a_steady_frame_rate(tmp_path, monkeypatch): + src = _fake_probe(tmp_path, monkeypatch, {"avg_frame_rate": "30000/1001", "r_frame_rate": "30000/1001"}) + assert src.fps_warning == "" + + +def test_probe_source_warns_when_falling_back_to_the_default_frame_rate(tmp_path, monkeypatch): + src = _fake_probe(tmp_path, monkeypatch, {"avg_frame_rate": "0/0", "r_frame_rate": "0/0"}) + assert src.fps == 30.0 + assert "assuming 30 fps" in src.fps_warning.lower() From 46adf133f1a76dd724e4be91588ebd93ac1a490f Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:45:30 +0400 Subject: [PATCH 029/110] Add sample-mode transcription: test a language on a short window first MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit transcribe_podcast/transcribe_start take optional start_seconds and duration_seconds; transcribe_file extracts just that window (via a trimmed 16kHz mono wav) instead of decoding the whole file, skips diarization and face analysis, and marks the result complete: false with sample_offset_seconds carrying where in the source the window started. A sample never reads or writes the main transcript cache, the packed markdown view, the energy/event signal cache, or the UI/session transcript state — all of those are keyed by, and assumed to describe, the whole file. No CLI wiring: there is no standalone 'transcribe' subcommand in cli.py to attach --start-seconds/--duration-seconds to; 'process' runs the full clip pipeline, which a 40s sample isn't meant to feed. Left out rather than bolting the flags onto a command they don't fit. --- backend/main.py | 89 ++++++++++++-------- backend/services/audio_extract.py | 17 +++- backend/services/transcription.py | 63 ++++++++++++++ src/handlers/transcribe.handler.test.ts | 107 ++++++++++++++++++++++++ src/handlers/transcribe.handler.ts | 47 ++++++++++- src/models/index.ts | 5 ++ src/server.ts | 31 +++++-- src/ui/web-server.ts | 28 ++++++- tests/test_transcription_engine.py | 82 ++++++++++++++++++ 9 files changed, 417 insertions(+), 52 deletions(-) create mode 100644 src/handlers/transcribe.handler.test.ts diff --git a/backend/main.py b/backend/main.py index 567a83c2..26c9abe6 100644 --- a/backend/main.py +++ b/backend/main.py @@ -95,14 +95,21 @@ def handle_transcribe(task_id: str, params: dict): if params.get("assemblyai_api_key"): os.environ["ASSEMBLYAI_API_KEY"] = params["assemblyai_api_key"] + start_seconds = params.get("start_seconds") + duration_seconds = params.get("duration_seconds") + is_sample = start_seconds is not None or duration_seconds is not None + # One shared 16 kHz mono wav feeds transcription, energy and reactions - # instead of decoding the source three times. + # instead of decoding the source three times. Skipped for a sample run — + # transcribe_file extracts its own trimmed window, and decoding the full + # source here would undo the whole point of a quick sample. shared_wav = None - try: - from services.audio_extract import extract_wav_16k_mono - shared_wav = extract_wav_16k_mono(file_path) - except Exception: - shared_wav = None + if not is_sample: + try: + from services.audio_extract import extract_wav_16k_mono + shared_wav = extract_wav_16k_mono(file_path) + except Exception: + shared_wav = None try: result = transcribe_file( @@ -112,46 +119,54 @@ def handle_transcribe(task_id: str, params: dict): language=params.get("language"), enable_diarization=params.get("enable_diarization", True), num_speakers=params.get("num_speakers"), + start_seconds=start_seconds, + duration_seconds=duration_seconds, progress_callback=lambda pct, msg: emit_progress(task_id, "transcribing", pct, msg), wav_path=shared_wav, ) # Apply word corrections (Whisper misheard proper nouns) apply_corrections(result.get("words", []), result.get("segments", [])) - energy_data = None - try: - from services.audio_analyzer import extract_audio_energy - energy_data = extract_audio_energy(file_path, wav_path=shared_wav) - except Exception: - pass # energy is a nice-to-have + # A sample is a throwaway language/quality check on a slice of the + # source — energy/event signals and the packed view are keyed by the + # full file and meant to describe the whole episode, so skip them + # rather than caching partial (or source-wide-but-wrongly-expensive) + # data under those keys. + if not is_sample: + energy_data = None + try: + from services.audio_analyzer import extract_audio_energy + energy_data = extract_audio_energy(file_path, wav_path=shared_wav) + except Exception: + pass # energy is a nice-to-have - events_data = None - try: - from services.audio_events import extract_audio_events - events_data = extract_audio_events(file_path, wav_path=shared_wav) - except Exception: - pass # reactions are a nice-to-have + events_data = None + try: + from services.audio_events import extract_audio_events + events_data = extract_audio_events(file_path, wav_path=shared_wav) + except Exception: + pass # reactions are a nice-to-have - # Cached so clip suggestion reuses these instead of decoding the source again. - from services.signal_cache import save_signals - save_signals(file_path, energy_data=energy_data, events_data=events_data) + # Cached so clip suggestion reuses these instead of decoding the source again. + from services.signal_cache import save_signals + save_signals(file_path, energy_data=energy_data, events_data=events_data) - # Auto-pack: emit compact LLM-readable markdown alongside raw JSON. - # Pulls energy data so the packed view includes peak moments for clip reasoning. - try: - cache_hash = compute_cache_hash(file_path) + engine_cache_suffix(result.get("engine") or engine) - packed_path, packed_md = write_packed( - result, - cache_hash, - source_label=os.path.basename(file_path), - energy_data=energy_data, - events_data=events_data, - ) - result["packed_path"] = packed_path - result["packed_size_bytes"] = len(packed_md.encode("utf-8")) - except Exception as e: - # Non-fatal — transcription result is still useful without the packed view - emit_progress(task_id, "packing", 99, f"Packer skipped: {e}") + # Auto-pack: emit compact LLM-readable markdown alongside raw JSON. + # Pulls energy data so the packed view includes peak moments for clip reasoning. + try: + cache_hash = compute_cache_hash(file_path) + engine_cache_suffix(result.get("engine") or engine) + packed_path, packed_md = write_packed( + result, + cache_hash, + source_label=os.path.basename(file_path), + energy_data=energy_data, + events_data=events_data, + ) + result["packed_path"] = packed_path + result["packed_size_bytes"] = len(packed_md.encode("utf-8")) + except Exception as e: + # Non-fatal — transcription result is still useful without the packed view + emit_progress(task_id, "packing", 99, f"Packer skipped: {e}") finally: if previous_engine is None: os.environ.pop("PODCLI_ENGINE", None) diff --git a/backend/services/audio_extract.py b/backend/services/audio_extract.py index f9a6883a..31a23175 100644 --- a/backend/services/audio_extract.py +++ b/backend/services/audio_extract.py @@ -17,18 +17,29 @@ def extract_wav_16k_mono( media_path: str, wav_path: Optional[str] = None, timeout: int = 1800, + start_seconds: Optional[float] = None, + duration_seconds: Optional[float] = None, ) -> str: """Extract audio as 16 kHz mono 16-bit PCM WAV. Returns the wav path. When wav_path is None a temp file is created; the caller owns cleanup. + start_seconds/duration_seconds trim the output to a window of the source + — used for sample-mode transcription (test a language on a short clip + instead of the full episode). """ owns_wav = wav_path is None if owns_wav: fd, wav_path = tempfile.mkstemp(prefix="podcli_audio_", suffix=".wav") os.close(fd) - cmd = [ - "ffmpeg", "-y", "-loglevel", "error", - "-i", media_path, + cmd = ["ffmpeg", "-y", "-loglevel", "error"] + if start_seconds: + # Before -i: fast (keyframe-seek) trim. Precision to the frame + # doesn't matter for a language-check sample. + cmd += ["-ss", str(start_seconds)] + cmd += ["-i", media_path] + if duration_seconds: + cmd += ["-t", str(duration_seconds)] + cmd += [ "-vn", "-acodec", "pcm_s16le", "-ar", "16000", diff --git a/backend/services/transcription.py b/backend/services/transcription.py index 3bc48d70..a451e257 100644 --- a/backend/services/transcription.py +++ b/backend/services/transcription.py @@ -497,6 +497,8 @@ def transcribe_file( num_speakers: Optional[int] = None, progress_callback: Optional[Callable[[int, str], None]] = None, wav_path: Optional[str] = None, + start_seconds: Optional[float] = None, + duration_seconds: Optional[float] = None, ) -> dict: """ Transcribe a video/audio file with word-level timestamps and speaker detection. @@ -504,6 +506,15 @@ def transcribe_file( wav_path: optional pre-extracted 16 kHz mono WAV shared across analysis stages — used by whisper.cpp and diarization instead of re-decoding. + start_seconds/duration_seconds: sample mode — transcribe only a window of + the source (e.g. to test a language on 40s before committing to a full + run) instead of the whole file. The result is marked complete: False and + its timestamps are relative to the sample window, not the source; + sample_offset_seconds carries where in the source the window started. + Diarization and face analysis are skipped — a throwaway sample isn't + worth the extra passes, and both would need frame/audio access to the + original file that the trimmed clip doesn't carry. + Returns: { "transcript": str, @@ -518,6 +529,58 @@ def transcribe_file( if not os.path.exists(file_path): raise FileNotFoundError(f"File not found: {file_path}") + is_sample = start_seconds is not None or duration_seconds is not None + if is_sample: + from services.audio_extract import extract_wav_16k_mono + + sample_wav = extract_wav_16k_mono( + file_path, + start_seconds=start_seconds or 0.0, + duration_seconds=duration_seconds, + ) + try: + result = _transcribe_file_inner( + sample_wav, + model_size=model_size, + engine=engine, + language=language, + enable_diarization=False, + num_speakers=num_speakers, + progress_callback=progress_callback, + wav_path=sample_wav, + ) + finally: + try: + os.unlink(sample_wav) + except OSError: + pass + result["complete"] = False + result["sample_offset_seconds"] = start_seconds or 0.0 + return result + + return _transcribe_file_inner( + file_path, + model_size=model_size, + engine=engine, + language=language, + enable_diarization=enable_diarization, + num_speakers=num_speakers, + progress_callback=progress_callback, + wav_path=wav_path, + ) + + +def _transcribe_file_inner( + file_path: str, + model_size: str = "base", + engine: Optional[str] = None, + language: Optional[str] = None, + enable_diarization: bool = True, + num_speakers: Optional[int] = None, + progress_callback: Optional[Callable[[int, str], None]] = None, + wav_path: Optional[str] = None, +) -> dict: + requested = engine if engine is not None else os.environ.get("PODCLI_ENGINE", "") engine = normalize_engine(requested) use_cpp = engine == "whispercpp" diff --git a/src/handlers/transcribe.handler.test.ts b/src/handlers/transcribe.handler.test.ts new file mode 100644 index 00000000..afa74d64 --- /dev/null +++ b/src/handlers/transcribe.handler.test.ts @@ -0,0 +1,107 @@ +import { describe, it, expect, vi, beforeEach } from "vitest"; + +const executeMock = vi.fn(); +const cacheGetMock = vi.fn(); +const cacheSetMock = vi.fn(); +const cacheGetPackedMock = vi.fn(); +const cacheGetHashForEngineMock = vi.fn(); + +vi.mock("../services/python-executor.js", () => ({ + PythonExecutor: class { + execute = executeMock; + }, +})); + +vi.mock("../services/transcript-cache.js", () => ({ + TranscriptCache: class { + get = cacheGetMock; + set = cacheSetMock; + getPackedMarkdown = cacheGetPackedMock; + getFileHashForEngine = cacheGetHashForEngineMock; + }, + hasSpeakerLabels: () => true, +})); + +vi.mock("../services/engine-resolve.js", () => ({ + resolveTranscribeEngine: vi.fn().mockResolvedValue("whispercpp"), +})); + +const { handleTranscribe } = await import("./transcribe.handler.js"); + +describe("handleTranscribe — sample mode", () => { + beforeEach(() => { + executeMock.mockReset(); + cacheGetMock.mockReset(); + cacheSetMock.mockReset(); + cacheGetPackedMock.mockReset(); + }); + + it("never reads the main cache when start_seconds/duration_seconds are set", async () => { + executeMock.mockResolvedValue({ + data: { transcript: "hi", segments: [], words: [], duration: 40, language: "ka", engine: "whispercpp" }, + }); + + const result = await handleTranscribe({ + file_path: "/video.mp4", + start_seconds: 120, + duration_seconds: 40, + }); + + expect(cacheGetMock).not.toHaveBeenCalled(); + const parsed = JSON.parse(result); + expect(parsed.cached).toBe(false); + expect(parsed.packed_ready).toBe(false); + }); + + it("never writes the main cache for a sample result", async () => { + executeMock.mockResolvedValue({ + data: { + transcript: "hi", + segments: [], + words: [], + duration: 40, + language: "ka", + engine: "whispercpp", + complete: false, + sample_offset_seconds: 120, + }, + }); + + const result = await handleTranscribe({ + file_path: "/video.mp4", + start_seconds: 120, + duration_seconds: 40, + }); + + expect(cacheSetMock).not.toHaveBeenCalled(); + const parsed = JSON.parse(result); + expect(parsed.complete).toBe(false); + expect(parsed.sample_offset_seconds).toBe(120); + }); + + it("passes start_seconds/duration_seconds through to the transcribe task", async () => { + executeMock.mockResolvedValue({ + data: { transcript: "hi", segments: [], words: [], duration: 40, language: "ka", engine: "whispercpp" }, + }); + + await handleTranscribe({ file_path: "/video.mp4", start_seconds: 10, duration_seconds: 30 }); + + expect(executeMock).toHaveBeenCalledWith( + "transcribe", + expect.objectContaining({ start_seconds: 10, duration_seconds: 30 }), + ); + }); + + it("still uses the main cache for a normal (non-sample) request", async () => { + cacheGetMock.mockResolvedValue(null); + executeMock.mockResolvedValue({ + data: { transcript: "hi", segments: [], words: [], duration: 400, language: "en", engine: "whispercpp" }, + }); + cacheGetPackedMock.mockResolvedValue("# packed"); + + await handleTranscribe({ file_path: "/video.mp4" }); + + expect(cacheGetMock).toHaveBeenCalled(); + expect(cacheSetMock).toHaveBeenCalled(); + }); +}); diff --git a/src/handlers/transcribe.handler.ts b/src/handlers/transcribe.handler.ts index e0eb3529..de8c1ccd 100644 --- a/src/handlers/transcribe.handler.ts +++ b/src/handlers/transcribe.handler.ts @@ -15,6 +15,8 @@ export interface TranscribeInput { language?: string; enable_diarization?: boolean; num_speakers?: number; + start_seconds?: number; + duration_seconds?: number; } export const transcribeToolDef = { @@ -69,6 +71,18 @@ export const transcribeToolDef = { "Exact number of speakers if known (e.g. 2 for a two-person podcast). " + "Leave empty to auto-detect (2-5 speakers).", }, + start_seconds: { + type: "number", + description: + "Sample mode: only transcribe a window starting here (seconds into the source), " + + "instead of the whole file — e.g. to test a language on 40s before committing to " + + "a full run. Pair with duration_seconds. The result is marked complete: false and " + + "is not written to the main transcript cache.", + }, + duration_seconds: { + type: "number", + description: "Sample mode window length in seconds. Defaults start_seconds to 0 if omitted.", + }, }, required: ["file_path"], }, @@ -81,6 +95,9 @@ export async function handleTranscribe(input: TranscribeInput): Promise const language = input.language; const enableDiarization = input.enable_diarization !== false; const numSpeakers = input.num_speakers; + const startSeconds = input.start_seconds; + const durationSeconds = input.duration_seconds; + const isSample = startSeconds !== undefined || durationSeconds !== undefined; // Resolve before reading the cache: an unset engine is written under // whatever transcribe_file actually ran (e.g. "whispercpp" on a native @@ -88,9 +105,10 @@ export async function handleTranscribe(input: TranscribeInput): Promise const resolvedEngine = await resolveTranscribeEngine(executor, engine, modelSize); const cacheKey = { engine: resolvedEngine, model: modelSize, language }; - // Check cache first. A cached transcript without speakers cannot answer a - // request for them, so serving it makes re-transcribing look like a no-op. - const cachedRaw = await cache.get(filePath, cacheKey); + // A sample is a throwaway check on a slice of the file — it must never + // serve (or pollute) the main transcript cache, which is keyed by the + // whole file and assumed complete. + const cachedRaw = isSample ? null : await cache.get(filePath, cacheKey); const cached = cachedRaw && enableDiarization && !hasSpeakerLabels(cachedRaw) ? null : cachedRaw; if (cached) { @@ -122,6 +140,8 @@ export async function handleTranscribe(input: TranscribeInput): Promise language, enable_diarization: enableDiarization, num_speakers: numSpeakers, + start_seconds: startSeconds, + duration_seconds: durationSeconds, }); if (!result.data) { @@ -130,6 +150,12 @@ export async function handleTranscribe(input: TranscribeInput): Promise const data = result.data; const actualEngine = data.engine ?? resolvedEngine; + if (isSample) { + // Not cached and not packed — it's a slice of the file, not the whole + // transcript the cache/packed-view keys assume. + return JSON.stringify({ cached: false, packed_ready: false, ...formatResult(data) }); + } + // Cache the raw result under what it actually ran with, not the prediction // above — resolveTranscribeEngine can't see a model-load failure that only // shows up once transcribe_file tries it for real. @@ -155,6 +181,9 @@ function formatResult(data: TranscriptResult) { word_count: (data.words ?? []).length, segment_count: (data.segments ?? []).length, speakers: data.speakers ?? { num_speakers: 0, speakers: {} }, + ...(data.complete === false + ? { complete: false, sample_offset_seconds: data.sample_offset_seconds ?? 0 } + : {}), next_step: "Read the transcript via get_ui_state(include_transcript: true), then suggest_clips.", }; } @@ -189,6 +218,16 @@ export const transcribeStartToolDef = { }, enable_diarization: { type: "boolean", default: true }, num_speakers: { type: "number" }, + start_seconds: { + type: "number", + description: + "Sample mode: only transcribe a window starting here (seconds into the source). " + + "Pair with duration_seconds. Not written to the main transcript cache.", + }, + duration_seconds: { + type: "number", + description: "Sample mode window length in seconds. Defaults start_seconds to 0 if omitted.", + }, }, required: ["file_path"], }, @@ -206,6 +245,8 @@ export async function handleTranscribeStart(input: TranscribeInput): Promise { try { const result = await handleTranscribe({ @@ -309,22 +323,29 @@ export function createServer(): McpServer { language, enable_diarization, num_speakers, + start_seconds, + duration_seconds, }); // Push FULL transcript (words[] + segments[]) to Web UI state from the // on-disk cache — NOT the trimmed MCP response. Without words in UI // state, downstream batch_create_clips can't burn captions. + // Skipped for a sample: it's never written to this cache, and it's a + // slice of the file, not something that belongs in the UI session. + const isSample = start_seconds !== undefined || duration_seconds !== undefined; try { // handleTranscribe's result carries the engine it actually resolved // to and cached under; the request-time `engine` can be unset while // the write landed under "whispercpp", so reading with it misses. const resolvedEngine = (JSON.parse(result) as { engine?: string }).engine ?? engine; - const cached = await transcriptCache.get(file_path, { - engine: resolvedEngine, - model: model_size, - language, - }); + const cached = isSample + ? null + : await transcriptCache.get(file_path, { + engine: resolvedEngine, + model: model_size, + language, + }); if (cached) { await uiPing({ videoPath: file_path, diff --git a/src/ui/web-server.ts b/src/ui/web-server.ts index 4633e128..6f52ac18 100644 --- a/src/ui/web-server.ts +++ b/src/ui/web-server.ts @@ -1042,21 +1042,26 @@ app.post("/api/transcribe", async (req, res) => { language, enable_diarization = true, num_speakers, + start_seconds, + duration_seconds, } = req.body; if (!file_path || !existsSync(file_path)) { res.status(400).json({ error: "File not found" }); return; } + const isSample = start_seconds !== undefined || duration_seconds !== undefined; + // Resolve before reading the cache: an unset engine is written under // whatever transcribe_file actually ran (e.g. "whispercpp" on a native // install), so reading with the raw unset request always misses. const resolvedEngine = await resolveTranscribeEngine(executor, engine, model_size); const cacheKey = { engine: resolvedEngine, model: model_size, language }; - // Check cache first. A cached transcript without speakers cannot answer a - // request for them, so serving it makes re-transcribing look like a no-op. - const cachedRaw = await cache.get(file_path, cacheKey); + // A sample is a throwaway check on a slice of the file — never serve or + // populate the session/UI state from the main cache (keyed by, and + // assumed to describe, the whole file). + const cachedRaw = isSample ? null : await cache.get(file_path, cacheKey); const cached = cachedRaw && enable_diarization && !hasSpeakerLabels(cachedRaw) ? null : cachedRaw; if (cached) { @@ -1094,7 +1099,17 @@ app.post("/api/transcribe", async (req, res) => { executor .execute( "transcribe", - { file_path, model_size, engine, assemblyai_api_key, language, enable_diarization, num_speakers }, + { + file_path, + model_size, + engine, + assemblyai_api_key, + language, + enable_diarization, + num_speakers, + start_seconds, + duration_seconds, + }, (event) => { job.progress = event.percent; job.message = event.message; @@ -1105,6 +1120,11 @@ app.post("/api/transcribe", async (req, res) => { job.progress = 100; job.message = "Transcription complete"; job.result = result.data; + + // A sample result is a slice, not the episode — leave the session/UI + // state and the main cache alone. The caller reads it via job_status. + if (isSample) return; + sessionTranscripts.set( file_path, result.data as unknown as ServerTranscript, diff --git a/tests/test_transcription_engine.py b/tests/test_transcription_engine.py index d882e5bc..d92bfab7 100644 --- a/tests/test_transcription_engine.py +++ b/tests/test_transcription_engine.py @@ -80,6 +80,88 @@ def test_no_fallback_when_whispercpp_unavailable(self): tr.transcribe_file(self._tmp.name, model_size="base", enable_diarization=False) +class SampleModeTests(unittest.TestCase): + """start_seconds/duration_seconds should transcribe a trimmed window + instead of the full file, mark the result incomplete, and skip + diarization/face analysis (the trimmed clip has no video track for face + analysis to read, and a throwaway sample isn't worth either pass).""" + + def setUp(self): + self._orig_wcpp = tr._transcribe_with_whispercpp + self._orig_ready = tr._whispercpp_ready + self._saved_engine = os.environ.pop("PODCLI_ENGINE", None) + os.environ["PODCLI_ENGINE"] = "whispercpp" + self._tmp = tempfile.NamedTemporaryFile(suffix=".mp4", delete=False) + self._tmp.write(b"not a real video") + self._tmp.close() + self.extract_calls = [] + + def fake_extract(file_path, wav_path=None, timeout=1800, start_seconds=None, duration_seconds=None): + self.extract_calls.append( + {"file_path": file_path, "start_seconds": start_seconds, "duration_seconds": duration_seconds} + ) + fd, path = tempfile.mkstemp(suffix=".wav") + os.close(fd) + return path + + self._fake_extract = fake_extract + import services.audio_extract as audio_extract + + self._orig_extract = audio_extract.extract_wav_16k_mono + audio_extract.extract_wav_16k_mono = fake_extract + + def fake_wcpp(*args, **kwargs): + # The inner call must receive the sample wav as both file_path + # and wav_path, and diarization must be forced off. + self.inner_file_path = args[0] if args else kwargs.get("file_path") + self.inner_wav_path = kwargs.get("wav_path") + return {"engine": "whispercpp", "words": [], "segments": [], "duration": 40.0} + + tr._transcribe_with_whispercpp = fake_wcpp + tr._whispercpp_ready = lambda size: True + + def tearDown(self): + tr._transcribe_with_whispercpp = self._orig_wcpp + tr._whispercpp_ready = self._orig_ready + import services.audio_extract as audio_extract + + audio_extract.extract_wav_16k_mono = self._orig_extract + os.unlink(self._tmp.name) + if self._saved_engine is None: + os.environ.pop("PODCLI_ENGINE", None) + else: + os.environ["PODCLI_ENGINE"] = self._saved_engine + + def test_sample_window_is_passed_to_extraction(self): + tr.transcribe_file( + self._tmp.name, model_size="base", start_seconds=120.0, duration_seconds=40.0, + enable_diarization=False, + ) + self.assertEqual(len(self.extract_calls), 1) + self.assertEqual(self.extract_calls[0]["start_seconds"], 120.0) + self.assertEqual(self.extract_calls[0]["duration_seconds"], 40.0) + + def test_result_marked_incomplete_with_offset(self): + result = tr.transcribe_file( + self._tmp.name, model_size="base", start_seconds=120.0, duration_seconds=40.0, + enable_diarization=False, + ) + self.assertFalse(result["complete"]) + self.assertEqual(result["sample_offset_seconds"], 120.0) + + def test_full_run_is_marked_neither_incomplete_nor_offset(self): + result = tr.transcribe_file(self._tmp.name, model_size="base", enable_diarization=False) + self.assertNotIn("complete", result) + self.assertNotIn("sample_offset_seconds", result) + + def test_default_start_is_zero_when_only_duration_given(self): + result = tr.transcribe_file( + self._tmp.name, model_size="base", duration_seconds=40.0, enable_diarization=False, + ) + self.assertEqual(self.extract_calls[0]["start_seconds"], 0.0) + self.assertEqual(result["sample_offset_seconds"], 0.0) + + class ResolveEngineInfoTests(unittest.TestCase): """resolve_engine_info predicts transcribe_file's engine choice so a caller can build a matching cache key before deciding to transcribe.""" From 47670c4f6a25baea5eeaeb7fe174a2667ef51281 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:46:04 +0400 Subject: [PATCH 030/110] Skip case transforms for caseless scripts like Georgian in captions and thumbnails --- backend/services/caption_renderer.py | 5 +-- backend/services/thumbnail_html.py | 13 +++++--- backend/utils/text.py | 25 +++++++++++++++ remotion/src/Audiogram.tsx | 7 +++-- remotion/src/components/HormoziCaptions.tsx | 3 +- remotion/src/components/Scene.tsx | 9 ++++-- remotion/src/components/TopicChip.tsx | 7 +++-- remotion/src/text.test.ts | 24 +++++++++++++++ remotion/src/text.ts | 34 +++++++++++++++++++++ src/ui/client/EpisodeWorkspace.jsx | 4 +-- src/ui/client/lib.test.ts | 20 ++++++++++++ src/ui/client/lib.ts | 31 +++++++++++++++++++ tests/test_text_utils.py | 22 ++++++++++++- tests/test_thumbnail_html.py | 18 +++++++++++ 14 files changed, 206 insertions(+), 16 deletions(-) create mode 100644 remotion/src/text.test.ts create mode 100644 remotion/src/text.ts diff --git a/backend/services/caption_renderer.py b/backend/services/caption_renderer.py index 60141c54..f8bb5d44 100644 --- a/backend/services/caption_renderer.py +++ b/backend/services/caption_renderer.py @@ -12,6 +12,7 @@ from config.caption_styles import get_style from utils.timing_utils import seconds_to_ass +from utils.text import safe_upper def _sanitize_ass_text(text: str) -> str: @@ -227,7 +228,7 @@ def _render_hormozi(words: list[dict], style: dict, offset: float) -> str: parts = [] for w in chunk: duration_cs = int((w["end"] - w["start"]) * 100) - text = _sanitize_ass_text(w["word"].upper() if uppercase else w["word"]) + text = _sanitize_ass_text(safe_upper(w["word"]) if uppercase else w["word"]) parts.append(f"{{\\kf{duration_cs}}}{text}") # \c = active (filled) color, \2c = inactive (unfilled) color @@ -566,7 +567,7 @@ def _render_branded(words: list[dict], style: dict, offset: float) -> str: for j, w in enumerate(chunk): text = _sanitize_ass_text(_normalize_case(w["word"])) if j == 0: - text = text[0].upper() + text[1:] if len(text) > 1 else text.upper() + text = safe_upper(text[0]) + text[1:] if len(text) > 1 else safe_upper(text) normalized.append(text) # Measure word widths for pill positioning. diff --git a/backend/services/thumbnail_html.py b/backend/services/thumbnail_html.py index 81f6f119..7f96f8ee 100644 --- a/backend/services/thumbnail_html.py +++ b/backend/services/thumbnail_html.py @@ -16,6 +16,7 @@ import subprocess from config.paths import paths from utils.proc import run as proc_run, ProcError +from utils.text import safe_upper import sys import tempfile from typing import Optional @@ -429,8 +430,8 @@ def _build_html( # Clean any stray slashes from title split line1 = line1.strip().strip("/").strip() line2 = line2.strip().strip("/").strip() - l1 = line1.upper() if cfg.get("line1_uppercase", True) else line1 - l2 = line2.upper() if cfg.get("line2_uppercase", True) else line2 + l1 = safe_upper(line1) if cfg.get("line1_uppercase", True) else line1 + l2 = safe_upper(line2) if cfg.get("line2_uppercase", True) else line2 has_photo = photo_path and os.path.exists(str(photo_path)) @@ -643,7 +644,10 @@ def _fit_size(text, base_size, base_spacing): font-weight: {l1_weight}; letter-spacing: {l1_spacing}; color: {l1_color}; - text-transform: uppercase; + /* Casing is already resolved in Python via safe_upper(), which skips + caseless scripts (e.g. Georgian) that CSS text-transform would + otherwise still remap to a different alphabet (Mkhedruli -> Mtavruli). */ + text-transform: none; text-align: center; line-height: {l1_lh}; margin-bottom: {l1_mb}; @@ -663,7 +667,8 @@ def _fit_size(text, base_size, base_spacing): font-weight: {l2_weight}; font-style: {l2_style}; letter-spacing: {l2_spacing}; - text-transform: uppercase; + /* See .line1 above — casing is resolved in Python, not here. */ + text-transform: none; line-height: {l2_lh}; background: {hl_color}; color: {l2_text_color}; diff --git a/backend/utils/text.py b/backend/utils/text.py index 1f734ba8..9d481c02 100644 --- a/backend/utils/text.py +++ b/backend/utils/text.py @@ -3,6 +3,31 @@ TITLE_MAX = 55 FILENAME_MAX = 50 +# Scripts with no case distinction whose letters Unicode nonetheless assigns +# an uppercase mapping for display styling (Georgian Mkhedruli -> Mtavruli). +# str.upper() applies that mapping uncritically, so a caption or thumbnail +# style that uppercases for emphasis silently switches alphabets for these +# scripts instead of just emphasizing them. +_CASELESS_SCRIPT_RANGES = ( + (0x10A0, 0x10FF), # Georgian (Mkhedruli, Asomtavruli) + (0x1C90, 0x1CBF), # Georgian Extended (Mtavruli) + (0x2D00, 0x2D2F), # Georgian Supplement +) + + +def _is_caseless_script_char(ch: str) -> bool: + cp = ord(ch) + return any(lo <= cp <= hi for lo, hi in _CASELESS_SCRIPT_RANGES) + + +def safe_upper(text: str) -> str: + """Uppercase text, except for scripts where Unicode's uppercase mapping + would change the alphabet rather than just the case (see above). + """ + if not text or not any(_is_caseless_script_char(c) for c in text): + return text.upper() if text else text + return "".join(c if _is_caseless_script_char(c) else c.upper() for c in text) + _TRAILING_PUNCT = " ,;:-–—" # Windows refuses these as a filename stem whatever the extension, so "CON.fcpxml" diff --git a/remotion/src/Audiogram.tsx b/remotion/src/Audiogram.tsx index 1749a826..6089a2fd 100644 --- a/remotion/src/Audiogram.tsx +++ b/remotion/src/Audiogram.tsx @@ -5,6 +5,7 @@ import { KaraokeCaptions } from "./components/KaraokeCaptions"; import { SubtleCaptions } from "./components/SubtleCaptions"; import { BrandedCaptions } from "./components/BrandedCaptions"; import type { Word, CaptionStyle } from "./types"; +import { safeUpper } from "./text"; /** * What an episode that was never filmed looks like. @@ -94,10 +95,12 @@ export const Audiogram: React.FC = ({ fontWeight: 700, fontSize: Math.round(height * 0.022), letterSpacing: "0.08em", - textTransform: "uppercase", + // Resolved below via safeUpper — CSS text-transform would still + // remap caseless scripts like Georgian to a different alphabet. + textTransform: "none", }} > - {title} + {safeUpper(title)} )} diff --git a/remotion/src/components/HormoziCaptions.tsx b/remotion/src/components/HormoziCaptions.tsx index b52df0cd..b36206f9 100644 --- a/remotion/src/components/HormoziCaptions.tsx +++ b/remotion/src/components/HormoziCaptions.tsx @@ -8,6 +8,7 @@ import { captionScale } from "../types"; import { buildChunks, activeChunkAt } from "../chunks"; import { MOTION, motionAt } from "../motion"; import type { Motion } from "../motion"; +import { safeUpper } from "../text"; interface Props { words: Word[]; @@ -73,7 +74,7 @@ export const HormoziCaptions: React.FC = ({ {activeChunk.words.map((word, i) => { const isActive = currentTime >= word.start && currentTime < word.end; const text = style.uppercase - ? word.word.toUpperCase() + ? safeUpper(word.word) : word.word; // The word being spoken still wins: the sweep is what this style is. // Emphasis colours it for the rest of the chunk, which is the part diff --git a/remotion/src/components/Scene.tsx b/remotion/src/components/Scene.tsx index a5e93404..f4dca379 100644 --- a/remotion/src/components/Scene.tsx +++ b/remotion/src/components/Scene.tsx @@ -10,6 +10,7 @@ import { sizeOf, } from "../scene"; import type { Align, Block, Gap, Layout, Size, Tone } from "../scene"; +import { safeUpper } from "../text"; /** * Blocks, drawn. @@ -92,13 +93,17 @@ const Piece: React.FC<{ block: Block; paint: Paint }> = ({ block, paint }) => { fontWeight: WEIGHT[size], lineHeight: LINE_HEIGHT[size], letterSpacing: TRACKING[size] * unit, - textTransform: block.caps ? "uppercase" : undefined, + // Resolved per-run below via safeUpper — CSS text-transform would + // still remap caseless scripts like Georgian to a different alphabet. + textTransform: "none", color: toneOf(brand, accent, block.tone, size === "xs" ? "muted" : "ink"), textAlign: block.align ?? "start", }} > {emphasisRuns(block.text, block.emphasis).map((run, i) => ( - {run.text} + + {block.caps ? safeUpper(run.text) : run.text} + ))} ); diff --git a/remotion/src/components/TopicChip.tsx b/remotion/src/components/TopicChip.tsx index 2606a000..550a65f7 100644 --- a/remotion/src/components/TopicChip.tsx +++ b/remotion/src/components/TopicChip.tsx @@ -2,6 +2,7 @@ import React from "react"; import { useVideoConfig } from "remotion"; import { captionScale, FONT, LOGO_EDGE, LOGO_INSET } from "../types"; import type { LogoPosition } from "../types"; +import { safeUpper } from "../text"; export interface TopicChipProps { /** What the clip is about, in two or three words. */ @@ -62,7 +63,9 @@ export const TopicChip: React.FC = ({ fontSize: 30 * s, fontWeight: 700, letterSpacing: 3 * s, - textTransform: "uppercase", + // Resolved below via safeUpper, not here — CSS text-transform would + // still remap caseless scripts like Georgian to a different alphabet. + textTransform: "none", color, background, ...(padded @@ -70,7 +73,7 @@ export const TopicChip: React.FC = ({ : { textShadow: "0 2px 12px rgba(0,0,0,0.8)" }), }} > - {label} + {safeUpper(label)} ); }; diff --git a/remotion/src/text.test.ts b/remotion/src/text.test.ts new file mode 100644 index 00000000..e4951e51 --- /dev/null +++ b/remotion/src/text.test.ts @@ -0,0 +1,24 @@ +import { describe, expect, it } from "vitest"; +import { safeUpper } from "./text"; + +describe("safeUpper", () => { + it("uppercases plain ASCII text", () => { + expect(safeUpper("hello world")).toBe("HELLO WORLD"); + }); + + it("leaves Georgian Mkhedruli text unchanged instead of switching to Mtavruli", () => { + const georgian = "მიშა"; + expect(safeUpper(georgian)).toBe(georgian); + // Sanity check the bug this guards against: toUpperCase() alone does + // remap Mkhedruli to Mtavruli. + expect(georgian.toUpperCase()).not.toBe(georgian); + }); + + it("uppercases the Latin parts of a mixed-script string and leaves Georgian alone", () => { + expect(safeUpper("hello მიშა")).toBe("HELLO მიშა"); + }); + + it("passes through empty strings", () => { + expect(safeUpper("")).toBe(""); + }); +}); diff --git a/remotion/src/text.ts b/remotion/src/text.ts new file mode 100644 index 00000000..8dae0a82 --- /dev/null +++ b/remotion/src/text.ts @@ -0,0 +1,34 @@ +// Scripts with no case distinction whose letters Unicode nonetheless assigns +// an uppercase mapping for display styling (Georgian Mkhedruli -> Mtavruli). +// String.prototype.toUpperCase() applies that mapping uncritically, so a +// caption style that uppercases for emphasis silently switches alphabets for +// these scripts instead of just emphasizing them. +const CASELESS_SCRIPT_RANGES: Array<[number, number]> = [ + [0x10a0, 0x10ff], // Georgian (Mkhedruli, Asomtavruli) + [0x1c90, 0x1cbf], // Georgian Extended (Mtavruli) + [0x2d00, 0x2d2f], // Georgian Supplement +]; + +function isCaselessScriptChar(ch: string): boolean { + const cp = ch.codePointAt(0) ?? 0; + return CASELESS_SCRIPT_RANGES.some(([lo, hi]) => cp >= lo && cp <= hi); +} + +/** + * Uppercase text, except for scripts where Unicode's uppercase mapping + * would change the alphabet rather than just the case (see above). + */ +export function safeUpper(text: string): string { + if (!text) return text; + let hasCaseless = false; + for (const ch of text) { + if (isCaselessScriptChar(ch)) { + hasCaseless = true; + break; + } + } + if (!hasCaseless) return text.toUpperCase(); + return Array.from(text) + .map((ch) => (isCaselessScriptChar(ch) ? ch : ch.toUpperCase())) + .join(""); +} diff --git a/src/ui/client/EpisodeWorkspace.jsx b/src/ui/client/EpisodeWorkspace.jsx index 474f9b8f..da432cf5 100644 --- a/src/ui/client/EpisodeWorkspace.jsx +++ b/src/ui/client/EpisodeWorkspace.jsx @@ -40,7 +40,7 @@ import MomentTrim from './MomentTrim'; import { useDialog } from './useDialog'; import { PageHeader } from './Page'; import { buildPreviewChunks, activePreviewChunk, selectPreviewWords } from './captionChunks'; -import { findClipResult, resultBoundsKey, clipKey, buildEnergyMap, dropEnergy, clampClipIndex, resolveAssetName, formatTranscriptText } from './lib'; +import { findClipResult, resultBoundsKey, clipKey, buildEnergyMap, dropEnergy, clampClipIndex, resolveAssetName, formatTranscriptText, safeUpper } from './lib'; // Mirrors backend/services/formats.py. A horizontal cutdown is minutes long, // so a slider capped at 60s could not express one. @@ -294,7 +294,7 @@ const onKeyActivate = (fn) => (e) => { function PhoneCaptionBody({ chunk, activeWordInChunk, cfg, singleLine = false }) { if (!chunk || !chunk.length) return null; - const fmt = (w) => (cfg.uppercase ? w.toUpperCase() : w); + const fmt = (w) => (cfg.uppercase ? safeUpper(w) : w); const renderWord = (w, i, isActive) => { if (cfg.activePill && isActive) { diff --git a/src/ui/client/lib.test.ts b/src/ui/client/lib.test.ts index cf3615d8..dc877ffd 100644 --- a/src/ui/client/lib.test.ts +++ b/src/ui/client/lib.test.ts @@ -9,6 +9,7 @@ import { clampClipIndex, resolveAssetName, formatTranscriptText, + safeUpper, } from "./lib"; describe("fmt", () => { @@ -212,3 +213,22 @@ describe("formatTranscriptText", () => { ); }); }); + +describe("safeUpper", () => { + it("uppercases plain ASCII text", () => { + expect(safeUpper("hello world")).toBe("HELLO WORLD"); + }); + + it("leaves Georgian Mkhedruli text unchanged instead of switching to Mtavruli", () => { + const georgian = "მიშა"; + expect(safeUpper(georgian)).toBe(georgian); + }); + + it("uppercases the Latin parts of a mixed-script string and leaves Georgian alone", () => { + expect(safeUpper("hello მიშა")).toBe("HELLO მიშა"); + }); + + it("passes through empty strings", () => { + expect(safeUpper("")).toBe(""); + }); +}); diff --git a/src/ui/client/lib.ts b/src/ui/client/lib.ts index eca88ee7..1acb4043 100644 --- a/src/ui/client/lib.ts +++ b/src/ui/client/lib.ts @@ -26,6 +26,37 @@ export const labelStyle: CSSProperties = { display: "block", }; +// Scripts with no case distinction whose letters Unicode nonetheless assigns +// an uppercase mapping for display styling (Georgian Mkhedruli -> Mtavruli). +// Kept in sync with remotion/src/text.ts and backend/utils/text.py's +// safe_upper — this is the same fix for the studio's live caption/thumbnail +// preview, so it doesn't show a different alphabet than the final render. +const CASELESS_SCRIPT_RANGES: Array<[number, number]> = [ + [0x10a0, 0x10ff], // Georgian (Mkhedruli, Asomtavruli) + [0x1c90, 0x1cbf], // Georgian Extended (Mtavruli) + [0x2d00, 0x2d2f], // Georgian Supplement +]; + +function isCaselessScriptChar(ch: string): boolean { + const cp = ch.codePointAt(0) ?? 0; + return CASELESS_SCRIPT_RANGES.some(([lo, hi]) => cp >= lo && cp <= hi); +} + +export function safeUpper(text: string): string { + if (!text) return text; + let hasCaseless = false; + for (const ch of text) { + if (isCaselessScriptChar(ch)) { + hasCaseless = true; + break; + } + } + if (!hasCaseless) return text.toUpperCase(); + return Array.from(text) + .map((ch) => (isCaselessScriptChar(ch) ? ch : ch.toUpperCase())) + .join(""); +} + export class ApiError extends Error { constructor( message: string, diff --git a/tests/test_text_utils.py b/tests/test_text_utils.py index 64a2187b..066771be 100644 --- a/tests/test_text_utils.py +++ b/tests/test_text_utils.py @@ -3,7 +3,7 @@ sys.path.insert(0, os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "backend")) -from utils.text import clean_title, safe_filename, truncate_title # noqa: E402 +from utils.text import clean_title, safe_filename, safe_upper, truncate_title # noqa: E402 def test_clean_title_keeps_full_text(): @@ -87,3 +87,23 @@ def test_safe_filename_escapes_windows_reserved_device_names(): def test_safe_filename_leaves_names_merely_containing_reserved_words(): assert safe_filename("Conference") == "Conference" assert safe_filename("CON artists") == "CON_artists" + + +def test_safe_upper_uppercases_plain_ascii(): + assert safe_upper("hello world") == "HELLO WORLD" + + +def test_safe_upper_leaves_georgian_mkhedruli_unchanged(): + georgian = "მიშა" + # Sanity check the bug this guards against: str.upper() alone does remap + # Mkhedruli to Mtavruli, a different alphabet, not just a different case. + assert georgian.upper() != georgian + assert safe_upper(georgian) == georgian + + +def test_safe_upper_uppercases_latin_in_mixed_script_text(): + assert safe_upper("hello მიშა") == "HELLO მიშა" + + +def test_safe_upper_handles_empty_string(): + assert safe_upper("") == "" diff --git a/tests/test_thumbnail_html.py b/tests/test_thumbnail_html.py index 0ae5751b..899fcac7 100644 --- a/tests/test_thumbnail_html.py +++ b/tests/test_thumbnail_html.py @@ -177,5 +177,23 @@ def test_nothing_at_all_when_no_layers_were_set(self): self.assertEqual(th._layer_html([]), "") +class GeorgianCasingTests(unittest.TestCase): + """Georgian is caseless; uppercasing for emphasis must not swap it to a + different alphabet (Mkhedruli -> Mtavruli), whether that uppercasing + happens in Python or in the CSS the HTML carries.""" + + def test_build_html_uppercases_latin_but_not_georgian(self): + html = th._build_html("hello მიშა", "second line", config={}) + self.assertIn("HELLO", html) + self.assertIn("მიშა", html) + self.assertNotIn("Მიშა".upper(), html) + + def test_build_html_css_does_not_redundantly_uppercase(self): + # text-transform: uppercase in the CSS would re-break Georgian in the + # headless browser even after the Python-side fix above. + html = th._build_html("hello world", "second line", config={}) + self.assertNotIn("text-transform: uppercase", html) + + if __name__ == "__main__": unittest.main() From 0639bfda21d80bd84eaf308d5e867eea282ec33f Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:46:53 +0400 Subject: [PATCH 031/110] Make podcli doctor actually run checks instead of only printing paths Previously doctor() only printed where things were supposed to be and always exited 0. It now runs each provisioned binary (ffmpeg -version, ffprobe -version, whisper-cli --help, node --version), imports a sample of the Python backend's own modules under whichever interpreter podcli would actually use, hashes every provisioned whisper.cpp/VAD model against its pinned sha256, and checks MCP registration with both Claude and Codex when their CLIs are present. Each becomes a pass/fail doctorCheck; doctor exits nonzero if any failed, and --json emits the full report as JSON for scripting. provision.go exports KnownModelSizes, ModelSHA256, and VADModelSHA256 so doctor can validate hashes without duplicating the pinned values. Removed presence() and humanBytes(), both now dead with the old print-only models section gone. --- cli/doctor_test.go | 107 +++++++++++++ cli/internal/provision/provision.go | 25 +++ cli/main.go | 237 ++++++++++++++++++++++------ 3 files changed, 323 insertions(+), 46 deletions(-) create mode 100644 cli/doctor_test.go diff --git a/cli/doctor_test.go b/cli/doctor_test.go new file mode 100644 index 00000000..5fa83481 --- /dev/null +++ b/cli/doctor_test.go @@ -0,0 +1,107 @@ +package main + +import ( + "os" + "path/filepath" + "testing" + + "podcli/internal/provision" +) + +func TestFirstLine(t *testing.T) { + if got := firstLine("a\nb\nc"); got != "a" { + t.Fatalf("got %q", got) + } + if got := firstLine(" padded \n"); got != "padded" { + t.Fatalf("got %q", got) + } + if got := firstLine(""); got != "" { + t.Fatalf("got %q", got) + } +} + +// go is guaranteed present wherever these tests run (it built the test +// binary), so it stands in for "a real hermetic/PATH tool that works". +func TestRunCheckSucceedsForARealBinary(t *testing.T) { + c := runCheck("go", "", "go", "version") + if !c.OK { + t.Fatalf("expected ok, got %+v", c) + } + if c.Detail == "" { + t.Fatal("expected a detail message") + } +} + +func TestRunCheckFailsWhenBinaryIsMissing(t *testing.T) { + c := runCheck("nope", "", "podcli-definitely-not-a-real-binary-xyz") + if c.OK { + t.Fatalf("expected failure, got %+v", c) + } +} + +func TestRunCheckFailsWhenHermeticPathDoesNotRun(t *testing.T) { + // A hermetic path that resolves but can't actually run (e.g. corrupted + // install) must fail the check, not just report "found". + dir := t.TempDir() + bogus := filepath.Join(dir, "not-executable") + if err := os.WriteFile(bogus, []byte("not a real binary"), 0o644); err != nil { + t.Fatalf("seed bogus binary: %v", err) + } + c := runCheck("bogus", bogus, "bogus") + if c.OK { + t.Fatalf("expected failure, got %+v", c) + } +} + +func TestModelChecksSkipsUnprovisionedModels(t *testing.T) { + home := t.TempDir() + old, hadOld := os.LookupEnv("PODCLI_HOME") + os.Setenv("PODCLI_HOME", home) + t.Cleanup(func() { + if hadOld { + os.Setenv("PODCLI_HOME", old) + } else { + os.Unsetenv("PODCLI_HOME") + } + }) + + checks := modelChecks() + if len(checks) != 0 { + t.Fatalf("expected no checks for an install with no models downloaded, got %+v", checks) + } +} + +func TestModelChecksFlagsAHashMismatch(t *testing.T) { + home := t.TempDir() + old, hadOld := os.LookupEnv("PODCLI_HOME") + os.Setenv("PODCLI_HOME", home) + t.Cleanup(func() { + if hadOld { + os.Setenv("PODCLI_HOME", old) + } else { + os.Unsetenv("PODCLI_HOME") + } + }) + + modelPath := provision.ModelPath("base") + if err := os.MkdirAll(filepath.Dir(modelPath), 0o755); err != nil { + t.Fatalf("mkdir models dir: %v", err) + } + if err := os.WriteFile(modelPath, []byte("not the real model bytes"), 0o644); err != nil { + t.Fatalf("seed fake model: %v", err) + } + + checks := modelChecks() + found := false + for _, c := range checks { + if c.Name == "model base" { + found = true + if c.OK { + t.Fatalf("expected a hash mismatch to fail, got %+v", c) + } + } + } + if !found { + t.Fatalf("expected a check for the present-but-wrong base model, got %+v", checks) + } +} diff --git a/cli/internal/provision/provision.go b/cli/internal/provision/provision.go index 2af7c718..5d73ac77 100644 --- a/cli/internal/provision/provision.go +++ b/cli/internal/provision/provision.go @@ -61,6 +61,31 @@ func VADModelPath() string { return filepath.Join(paths.ModelsDir(), "ggml-silero-v5.1.2.bin") } +// KnownModelSizes lists the whisper.cpp model sizes podcli knows how to +// provision, for callers (doctor) that need to check every one that's +// actually present rather than assuming a single size. +func KnownModelSizes() []string { + sizes := make([]string, 0, len(models)) + for size := range models { + sizes = append(sizes, size) + } + return sizes +} + +// ModelSHA256 returns the pinned hash for a whisper.cpp model size. +func ModelSHA256(size string) (string, bool) { + m, ok := models[size] + if !ok { + return "", false + } + return m.SHA256, true +} + +// VADModelSHA256 returns the pinned hash for the VAD model. +func VADModelSHA256() string { + return vadSHA +} + func have(p string) bool { if fi, err := os.Stat(p); err == nil && fi.Size() > 0 { return true diff --git a/cli/main.go b/cli/main.go index 24d54ce0..84ad89d1 100644 --- a/cli/main.go +++ b/cli/main.go @@ -4,11 +4,13 @@ package main import ( "bytes" + "encoding/json" "fmt" "os" "os/exec" "path/filepath" "runtime" + "sort" "strings" "podcli/internal/backend" @@ -33,7 +35,7 @@ func main() { case "version", "--version", "-v": fmt.Printf("podcli %s\n", Version) case "doctor": - doctor() + os.Exit(doctor(args[1:])) case "update": os.Exit(update.Run(Version)) case "uninstall": @@ -745,17 +747,183 @@ func backendStamp(root string) string { } } -func doctor() { +// doctorCheck is one pass/fail probe doctor actually runs, as opposed to the +// path/engine-resolution report above it, which only states what was found. +type doctorCheck struct { + Name string `json:"name"` + OK bool `json:"ok"` + Detail string `json:"detail"` +} + +type doctorReport struct { + Version string `json:"version"` + Paths map[string]string `json:"paths"` + Checks []doctorCheck `json:"checks"` + OK bool `json:"ok"` +} + +func firstLine(s string) string { + s = strings.TrimSpace(s) + if i := strings.IndexByte(s, '\n'); i != -1 { + s = s[:i] + } + return s +} + +// runCheck resolves a hermetic binary (falling back to PATH), then actually +// runs it. A binary that resolves but fails to run (missing shared lib, bad +// install) is exactly the failure mode `podcli doctor` otherwise can't see. +func runCheck(name, hermetic, pathFallback string, args ...string) doctorCheck { + bin := hermetic + source := "hermetic" + if bin == "" { + if p, err := exec.LookPath(pathFallback); err == nil { + bin = p + source = "PATH" + } + } + if bin == "" { + return doctorCheck{Name: name, OK: false, Detail: "not found (hermetic or PATH)"} + } + out, err := exec.Command(bin, args...).CombinedOutput() + if err != nil { + return doctorCheck{Name: name, OK: false, Detail: fmt.Sprintf("%s (%s): %v: %s", bin, source, err, firstLine(string(out)))} + } + return doctorCheck{Name: name, OK: true, Detail: fmt.Sprintf("%s (%s): %s", bin, source, firstLine(string(out)))} +} + +func pythonBackendCheck() doctorCheck { + root, ok := engine.BackendRoot() + if !ok { + return doctorCheck{Name: "python backend", OK: false, Detail: "backend not found (set PODCLI_BACKEND or run inside the repo)"} + } + // A representative sample, not every module: enough to catch "the + // interpreter can't even import the backend's own services" without + // reimplementing the whole import graph here. + modules := []string{"services.ai_cli", "services.multicam", "services.caption_renderer", "services.transcription"} + script := fmt.Sprintf("import sys; sys.path.insert(0, %q); import %s", root, strings.Join(modules, ", ")) + out, err := exec.Command(engine.Python(), "-c", script).CombinedOutput() + if err != nil { + return doctorCheck{Name: "python backend", OK: false, Detail: fmt.Sprintf("%s: %v: %s", engine.Python(), err, firstLine(string(out)))} + } + return doctorCheck{Name: "python backend", OK: true, Detail: fmt.Sprintf("%s imports %s", engine.Python(), strings.Join(modules, ", "))} +} + +func modelChecks() []doctorCheck { + var checks []doctorCheck + sizes := provision.KnownModelSizes() + sort.Strings(sizes) + for _, size := range sizes { + p := provision.ModelPath(size) + if !fileExists(p) { + continue // not provisioned; that's a valid state, not a failure + } + want, _ := provision.ModelSHA256(size) + got, err := provision.Sha256File(p) + name := "model " + size + if err != nil { + checks = append(checks, doctorCheck{Name: name, OK: false, Detail: fmt.Sprintf("could not hash %s: %v", p, err)}) + continue + } + if got != want { + checks = append(checks, doctorCheck{Name: name, OK: false, Detail: fmt.Sprintf("%s hash mismatch: got %s, want %s", p, got, want)}) + continue + } + checks = append(checks, doctorCheck{Name: name, OK: true, Detail: p}) + } + if vp := provision.VADModelPath(); fileExists(vp) { + got, err := provision.Sha256File(vp) + if err != nil { + checks = append(checks, doctorCheck{Name: "model vad", OK: false, Detail: fmt.Sprintf("could not hash %s: %v", vp, err)}) + } else if want := provision.VADModelSHA256(); got != want { + checks = append(checks, doctorCheck{Name: "model vad", OK: false, Detail: fmt.Sprintf("%s hash mismatch: got %s, want %s", vp, got, want)}) + } else { + checks = append(checks, doctorCheck{Name: "model vad", OK: true, Detail: vp}) + } + } + return checks +} + +func mcpRegistrationChecks() []doctorCheck { + var checks []doctorCheck + if _, err := exec.LookPath("claude"); err == nil { + if mcpRegisteredToSelf() { + checks = append(checks, doctorCheck{Name: "mcp registration (Claude)", OK: true, Detail: "registered"}) + } else { + checks = append(checks, doctorCheck{Name: "mcp registration (Claude)", OK: false, Detail: "not registered - run `podcli mcp install`"}) + } + } + if _, err := exec.LookPath("codex"); err == nil { + if codexMCPRegisteredToSelf() { + checks = append(checks, doctorCheck{Name: "mcp registration (Codex)", OK: true, Detail: "registered"}) + } else { + checks = append(checks, doctorCheck{Name: "mcp registration (Codex)", OK: false, Detail: "not registered - run `podcli mcp install`"}) + } + } + return checks +} + +func runDoctorChecks() []doctorCheck { + var checks []doctorCheck + checks = append(checks, runCheck("ffmpeg", engine.FFmpeg(), "ffmpeg", "-version")) + checks = append(checks, runCheck("ffprobe", engine.FFprobe(), "ffprobe", "-version")) + checks = append(checks, runCheck("whisper-cli", engine.WhisperCLI(), "whisper-cli", "--help")) + checks = append(checks, runCheck("node", engine.Node(), "node", "--version")) + checks = append(checks, pythonBackendCheck()) + checks = append(checks, modelChecks()...) + checks = append(checks, mcpRegistrationChecks()...) + return checks +} + +func doctor(args []string) int { + asJSON := false + for _, a := range args { + if a == "--json" { + asJSON = true + } + } + + pathInfo := map[string]string{ + "home": paths.Home(), + "runtime": paths.RuntimeDir(), + "models": paths.ModelsDir(), + } + if out := os.Getenv("PODCLI_OUTPUT"); out != "" { + pathInfo["clips"] = out + } else if cwd, err := os.Getwd(); err == nil { + pathInfo["clips"] = filepath.Join(cwd, "podcli-clips") + } + + checks := runDoctorChecks() + allOK := true + for _, c := range checks { + if !c.OK { + allOK = false + } + } + + if asJSON { + report := doctorReport{Version: Version, Paths: pathInfo, Checks: checks, OK: allOK} + data, err := json.MarshalIndent(report, "", " ") + if err != nil { + fmt.Fprintln(os.Stderr, "podcli: doctor:", err) + return 1 + } + fmt.Println(string(data)) + if !allOK { + return 1 + } + return 0 + } + fmt.Printf("podcli %s\n\n", Version) fmt.Println("Paths") - fmt.Printf(" home: %s\n", paths.Home()) - fmt.Printf(" runtime: %s\n", paths.RuntimeDir()) - fmt.Printf(" models: %s\n", paths.ModelsDir()) + fmt.Printf(" home: %s\n", pathInfo["home"]) + fmt.Printf(" runtime: %s\n", pathInfo["runtime"]) + fmt.Printf(" models: %s\n", pathInfo["models"]) fmt.Printf(" presets/knowledge/assets/history/cache: %s (global - follow you everywhere)\n", paths.Home()) - if out := os.Getenv("PODCLI_OUTPUT"); out != "" { - fmt.Printf(" clips: %s (PODCLI_OUTPUT)\n", out) - } else if cwd, err := os.Getwd(); err == nil { - fmt.Printf(" clips: %s (rendered into your working directory)\n", filepath.Join(cwd, "podcli-clips")) + if clips, ok := pathInfo["clips"]; ok { + fmt.Printf(" clips: %s\n", clips) } fmt.Println("\nEngine resolution") if root, ok := engine.BackendRoot(); ok { @@ -771,24 +939,6 @@ func doctor() { fmt.Printf(" backend: NOT FOUND (set PODCLI_BACKEND or run inside the repo)\n") } fmt.Printf(" python: %s\n", engine.Python()) - if ff := engine.FFmpeg(); ff != "" { - fmt.Printf(" ffmpeg: %s (hermetic)\n", ff) - } else { - fmt.Printf(" ffmpeg: PATH fallback (not yet hermetic)\n") - } - if fp := engine.FFprobe(); fp != "" { - fmt.Printf(" ffprobe: %s (hermetic)\n", fp) - } - if wc := engine.WhisperCLI(); wc != "" { - fmt.Printf(" whisper: %s (hermetic)\n", wc) - } else { - fmt.Printf(" whisper: PATH fallback (install whisper-cli, or provisioned once hosted)\n") - } - if nd := engine.Node(); nd != "" { - fmt.Printf(" node: %s (hermetic)\n", nd) - } else { - fmt.Printf(" node: PATH fallback (Web UI uses system Node, or run `podcli setup`)\n") - } if ss := engine.StudioServer(); ss != "" { fmt.Printf(" studio: %s\n", ss) } else { @@ -804,16 +954,22 @@ func doctor() { } else { fmt.Printf(" remotion: not provisioned (captions/thumbnails need a published release)\n") } - fmt.Println("\nModels") - fmt.Printf(" base: %s\n", presence(provision.ModelPath("base"))) - fmt.Printf(" vad: %s\n", presence(provision.VADModelPath())) -} -func presence(p string) string { - if fi, err := os.Stat(p); err == nil && fi.Size() > 0 { - return fmt.Sprintf("%s (%s)", p, humanBytes(fi.Size())) + fmt.Println("\nChecks") + for _, c := range checks { + mark := "OK " + if !c.OK { + mark = "FAIL" + } + fmt.Printf(" [%s] %-28s %s\n", mark, c.Name, c.Detail) } - return "not provisioned - run `podcli setup`" + + if !allOK { + fmt.Println("\npodcli doctor found problems above.") + return 1 + } + fmt.Println("\nAll checks passed.") + return 0 } func fileExists(p string) bool { @@ -821,17 +977,6 @@ func fileExists(p string) bool { return err == nil } -func humanBytes(n int64) string { - switch { - case n >= 1<<20: - return fmt.Sprintf("%d MB", n>>20) - case n >= 1<<10: - return fmt.Sprintf("%d KB", n>>10) - default: - return fmt.Sprintf("%d B", n) - } -} - func printHelp() { fmt.Printf(`podcli %s - AI podcast clip generator From c51f14aabe7134c3361f816c46767bf8a8bf88e3 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:48:16 +0400 Subject: [PATCH 032/110] Support selecting which audio stream a source uses --- backend/services/multicam.py | 29 +++++++++++++++----- src/server.ts | 1 + src/ui/client/multicam-types.ts | 3 +++ tests/test_multicam.py | 48 +++++++++++++++++++++++++++++++++ 4 files changed, 75 insertions(+), 6 deletions(-) diff --git a/backend/services/multicam.py b/backend/services/multicam.py index 5414412e..08b8f8ba 100644 --- a/backend/services/multicam.py +++ b/backend/services/multicam.py @@ -115,6 +115,9 @@ class Source: file_size: int = 0 # size at probe time; a mismatch on reopen means the file changed underneath us file_mtime_ns: int = 0 # mtime at probe time, nanosecond resolution fps_warning: str = "" # set when the container's frame rate looks variable or had to be guessed + audio_stream_count: int = 1 # separate audio streams in the container (not channels within one stream) + audio_stream_index: int = 0 # which audio stream to use, for MXF-style cameras with one mono stream per mic + audio_stream_channels: list[int] = field(default_factory=list) # channel count per audio stream, by index role: str = "ignore" # "camera" | "mic" | "ignore" # camera: a person id or "wide"; mic: a person id, or "" for a shared room mic person: str = "" @@ -337,7 +340,8 @@ def probe_source(path: str) -> Source: and not (s.get("disposition") or {}).get("attached_pic")), None, ) - audio = next((s for s in streams if s.get("codec_type") == "audio"), None) + audio_streams = [s for s in streams if s.get("codec_type") == "audio"] + audio = audio_streams[0] if audio_streams else None durations = [float(info.get("format", {}).get("duration") or 0)] durations += [float(s.get("duration") or 0) for s in streams] duration = max(durations) @@ -383,6 +387,8 @@ def rate(field: str) -> float: file_size=stat.st_size, file_mtime_ns=stat.st_mtime_ns, fps_warning=fps_warning, + audio_stream_count=len(audio_streams), + audio_stream_channels=[int(a.get("channels") or 0) for a in audio_streams], ) @@ -952,6 +958,15 @@ def update_mapping(session: MulticamSession, params: dict) -> MulticamSession: if "channel_people" in edit: chans = [c if c in valid else "" for c in (edit["channel_people"] or [])] s.channel_people = chans[: max(0, s.audio_channels)] if any(chans) else [] + if "audio_stream_index" in edit: + idx = edit["audio_stream_index"] + if not isinstance(idx, int) or not 0 <= idx < max(1, s.audio_stream_count): + raise ValueError(f"{os.path.basename(s.path)} has {s.audio_stream_count} audio " + f"stream(s); audio_stream_index must be between 0 and {s.audio_stream_count - 1}") + s.audio_stream_index = idx + if idx < len(s.audio_stream_channels): + s.audio_channels = s.audio_stream_channels[idx] + s.channel_people = [] if (("offset" in edit and edit["offset"] is not None) or "nudge" in edit) and s.virtual: raise ValueError("A tile or split screen moves with the files it comes from. Move those instead.") if "offset" in edit and edit["offset"] is not None: @@ -1008,7 +1023,7 @@ def update_mapping(session: MulticamSession, params: dict) -> MulticamSession: def _cut_inputs(session: MulticamSession) -> str: return json.dumps([ - [(s.id, s.role, s.person, s.channel_people, s.offset) for s in session.sources], + [(s.id, s.role, s.person, s.channel_people, s.audio_stream_index, s.offset) for s in session.sources], [(p.id, p.role) for p in session.people], session.range_start, session.range_end, session.cut_settings, session.speaker_map, ], default=str) @@ -1033,7 +1048,7 @@ def _extract(source: Source, out: Path, channel: int = -1, rate: int = SYNC_RATE # clock the render and editors use; audio often starts 20-100 ms after video. proc_run([ "ffmpeg", "-y", "-hide_banner", "-loglevel", "error", - "-i", source.path, "-vn", "-map", "0:a:0", + "-i", source.path, "-vn", "-map", f"0:a:{source.audio_stream_index}", "-af", f"aresample=async=1:first_pts=0,{pick}", "-ar", str(rate), "-acodec", "pcm_s16le", str(tmp), ], timeout=3600, check=True) @@ -1259,7 +1274,8 @@ def wav(s: Source) -> np.ndarray: # --------------------------------------------------------------------------- def _activity_key(session: MulticamSession) -> str: - feeds = [(s.id, _file_identity(s), ch, pid, s.offset, s.speed) for s, ch, pid in session.person_mics()] + feeds = [(s.id, _file_identity(s), s.audio_stream_index, ch, pid, s.offset, s.speed) + for s, ch, pid in session.person_mics()] blob = json.dumps([feeds, session.speaker_map, session.person_ids()], sort_keys=True, default=str) return hashlib.sha1(blob.encode()).hexdigest()[:16] @@ -1831,7 +1847,8 @@ def _write_mix(session: MulticamSession, out: Path, start: float, duration: floa def _mix_key(session: MulticamSession) -> str: """Changes whenever which mics are mixed, or where they sit, changes.""" - feeds = [(s.id, ch, s.offset, s.speed, os.path.getmtime(s.path)) for s, ch in _audio_inputs(session)] + feeds = [(s.id, s.audio_stream_index, ch, s.offset, s.speed, os.path.getmtime(s.path)) + for s, ch in _audio_inputs(session)] return hashlib.sha1(json.dumps(feeds, default=str).encode()).hexdigest()[:12] @@ -2014,7 +2031,7 @@ def set_removals(session: MulticamSession, removals: list) -> MulticamSession: def _render_key(session: MulticamSession, stems: bool) -> str: """Everything the MP4 depends on, so an unchanged edit isn't rendered twice.""" used = {c["source_id"] for c in session.cuts} | {s.id for s, _ in _audio_inputs(session)} - files = [(s.id, s.offset, s.speed, s.channel_people, s.person, os.path.getmtime(s.path)) + files = [(s.id, s.offset, s.speed, s.channel_people, s.audio_stream_index, s.person, os.path.getmtime(s.path)) for s in session.sources if s.id in used and os.path.exists(s.path)] blob = json.dumps([session.cuts, session.removals, session.look, stems, files], sort_keys=True, default=str) return hashlib.sha1(blob.encode()).hexdigest()[:16] diff --git a/src/server.ts b/src/server.ts index 2419bd83..fdd2b6a5 100644 --- a/src/server.ts +++ b/src/server.ts @@ -2260,6 +2260,7 @@ export function createServer(): McpServer { role: z.enum(["camera", "mic", "ignore"]).optional(), person: z.string().optional().describe("Camera: a person id or 'wide'. Mic: a person id, or '' for a shared room mic"), channel_people: z.array(z.string()).optional().describe("Mic: one person id per channel when a recorder puts two people on L/R"), + audio_stream_index: z.number().int().min(0).optional().describe("Which audio stream in the container to use, for cameras (often MXF) that carry one mono stream per mic instead of packing channels into a single stream"), offset: z.number().optional().describe("Timeline seconds where this file starts, to override sync"), nudge: z.number().optional().describe("Seconds to shift the synced offset by"), }), diff --git a/src/ui/client/multicam-types.ts b/src/ui/client/multicam-types.ts index 442d5722..7c52f551 100644 --- a/src/ui/client/multicam-types.ts +++ b/src/ui/client/multicam-types.ts @@ -30,9 +30,12 @@ export interface McSource { duration: number; has_audio: boolean; audio_channels: number; + audio_stream_count?: number; + audio_stream_index?: number; width?: number; height?: number; fps?: number; + fps_warning?: string; role: McRole; person: string; channel_people: string[]; diff --git a/tests/test_multicam.py b/tests/test_multicam.py index 9b35a36a..abb45b47 100644 --- a/tests/test_multicam.py +++ b/tests/test_multicam.py @@ -1127,6 +1127,54 @@ def _fake_probe(tmp_path, monkeypatch, video_stream): return mc.probe_source(str(path)) +def test_probe_source_counts_multiple_audio_streams(tmp_path, monkeypatch): + path = tmp_path / "cam.mxf" + path.write_bytes(b"0") + info = {"format": {"duration": "10.0", "tags": {}}, "streams": [ + {"codec_type": "video", "avg_frame_rate": "30/1", "r_frame_rate": "30/1"}, + {"codec_type": "audio", "channels": 1}, + {"codec_type": "audio", "channels": 1}, + ]} + monkeypatch.setattr(mc, "get_video_info", lambda p: info) + src = mc.probe_source(str(path)) + assert src.audio_stream_count == 2 + assert src.audio_stream_channels == [1, 1] + assert src.audio_stream_index == 0 + + +def test_update_mapping_validates_audio_stream_index(sandbox): + session = mc.MulticamSession( + session_id="abc123abc999", name="ep", + people=[mc.Person("nika", "Nika")], + sources=[mc.Source(id="a", path="/x/cam_a.mov", kind="video", duration=9, has_audio=True, + audio_channels=1, audio_stream_count=2, audio_stream_channels=[1, 2])], + ) + with pytest.raises(ValueError, match="audio_stream_index"): + mc.update_mapping(session, {"sources": [{"id": "a", "audio_stream_index": 5}]}) + session = mc.update_mapping(session, {"sources": [{"id": "a", "audio_stream_index": 1}]}) + src = session.source("a") + assert src.audio_stream_index == 1 + assert src.audio_channels == 2 # recomputed from the selected stream, not the first one + + +@pytest.mark.skipif(not shutil.which("ffmpeg"), reason="ffmpeg not installed") +def test_extract_reads_the_selected_audio_stream_not_always_the_first(sandbox): + path = sandbox / "two_streams.mp4" + _ffmpeg("-f", "lavfi", "-i", "color=s=160x90:d=2", "-f", "lavfi", "-i", "anullsrc=r=48000:cl=mono", + "-f", "lavfi", "-i", "sine=f=440:r=48000:d=2", + "-map", "0:v", "-map", "1:a", "-map", "2:a", "-shortest", + "-c:v", "libx264", "-pix_fmt", "yuv420p", "-c:a", "aac", str(path)) + src = mc.probe_source(str(path)) + assert src.audio_stream_count == 2 + + silent = mc._read_wav(mc._extract(src, sandbox / "stream0.wav")) + assert np.abs(silent).mean() < 50 # stream 0 is anullsrc: near silence + + src.audio_stream_index = 1 + tone = mc._read_wav(mc._extract(src, sandbox / "stream1.wav")) + assert np.abs(tone).mean() > 1000 # stream 1 is a 440 Hz tone + + def test_probe_source_warns_on_variable_frame_rate(tmp_path, monkeypatch): # avg_frame_rate (what actually played) is far below r_frame_rate (the # stream's time base): frames held variable lengths. From f594cd66def42ec0a8fa85646a3d37425b7a5259 Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:48:18 +0400 Subject: [PATCH 033/110] Flag a selected clip as changed when it is edited after approval --- src/models/index.ts | 4 ++ src/server.ts | 6 ++- src/ui/web-server.ts | 25 +++++++-- src/utils/selection-hash.test.ts | 88 ++++++++++++++++++++++++++++++++ src/utils/selection-hash.ts | 34 ++++++++++++ 5 files changed, 153 insertions(+), 4 deletions(-) create mode 100644 src/utils/selection-hash.test.ts create mode 100644 src/utils/selection-hash.ts diff --git a/src/models/index.ts b/src/models/index.ts index f52ffad9..5933c9d6 100644 --- a/src/models/index.ts +++ b/src/models/index.ts @@ -110,6 +110,10 @@ export interface SuggestedClip { content_type?: string; score?: number; rank?: number; + /** Fingerprint of render-relevant fields, stamped when the clip is selected for export. */ + selectionHash?: string; + /** True when the clip was edited after selectionHash was stamped — the selection may be stale. */ + changedSinceSelection?: boolean; } export interface UIState { diff --git a/src/server.ts b/src/server.ts index 2419bd83..c100d5cb 100644 --- a/src/server.ts +++ b/src/server.ts @@ -1393,7 +1393,11 @@ export function createServer(): McpServer { ? `${Math.round(end - start)}s` : "?"; const style = clip.suggested_caption_style || "hormozi"; - const tag = deselected.includes(i) ? " [DESELECTED]" : ""; + const tag = deselected.includes(i) + ? " [DESELECTED]" + : clip.changedSinceSelection + ? " [CHANGED SINCE SELECTION — re-confirm before exporting]" + : ""; lines.push( ` #${num}: "${title}" (${start}s–${end}s, ${duration}) [${style}]${tag}`, ); diff --git a/src/ui/web-server.ts b/src/ui/web-server.ts index 78cdaf9b..72f11520 100644 --- a/src/ui/web-server.ts +++ b/src/ui/web-server.ts @@ -53,6 +53,7 @@ import { } from "../utils/transcript.js"; import { formatSrtTime, formatVttTime } from "../utils/srt-time.js"; import { computeVideoIdentity, type VideoIdentity } from "../utils/video-identity.js"; +import { computeSelectionHash } from "../utils/selection-hash.js"; import { errMsg } from "../utils/errors.js"; import { resolveByteRange } from "../utils/http-range.js"; import { @@ -4329,12 +4330,20 @@ app.post("/api/suggestions/modify", (req, res) => { res.status(400).json({ error: "selected must be a boolean for action 'toggle'" }); return; } + clip = uiState.suggestions[index]; if (selected) { uiState.deselectedIndices = uiState.deselectedIndices.filter((i) => i !== index); - } else if (!uiState.deselectedIndices.includes(index)) { - uiState.deselectedIndices = [...uiState.deselectedIndices, index]; + // Stamp what's being approved so a later edit to the same clip can be + // caught instead of silently rendering something else. + clip.selectionHash = computeSelectionHash(clip, uiState.transcript?.words); + clip.changedSinceSelection = false; + } else { + if (!uiState.deselectedIndices.includes(index)) { + uiState.deselectedIndices = [...uiState.deselectedIndices, index]; + } + delete clip.selectionHash; + delete clip.changedSinceSelection; } - clip = uiState.suggestions[index]; } else if (action === "update") { const upd = updates || {}; clip = uiState.suggestions[index]; @@ -4368,6 +4377,16 @@ app.post("/api/suggestions/modify", (req, res) => { const fmtTime = (s: number) => `${Math.floor(s / 60)}:${Math.floor(s % 60).toString().padStart(2, "0")}`; clip.timestamp_display = `${fmtTime(clip.start_second)} → ${fmtTime(clip.end_second)}`; + + // Approval was for a specific render; if this edit changed any + // render-relevant field on an already-selected clip, flag it instead of + // silently exporting something other than what got approved. + if (clip.selectionHash && !uiState.deselectedIndices.includes(index)) { + const currentHash = computeSelectionHash(clip, uiState.transcript?.words); + if (currentHash !== clip.selectionHash) { + clip.changedSinceSelection = true; + } + } } else { res.status(400).json({ error: `Unknown action: ${action}` }); return; diff --git a/src/utils/selection-hash.test.ts b/src/utils/selection-hash.test.ts new file mode 100644 index 00000000..2486dbb3 --- /dev/null +++ b/src/utils/selection-hash.test.ts @@ -0,0 +1,88 @@ +import { describe, it, expect } from "vitest"; +import { computeSelectionHash } from "./selection-hash.js"; + +function clip(overrides: Partial[0]> = {}) { + return { + start_second: 10, + end_second: 20, + segments: undefined, + suggested_caption_style: "hormozi", + title: "A title", + ...overrides, + }; +} + +const words = [ + { word: "one", start: 9, end: 9.5, confidence: 1 }, + { word: "two", start: 10, end: 10.5, confidence: 1 }, + { word: "three", start: 15, end: 15.5, confidence: 1 }, + { word: "four", start: 20, end: 20.5, confidence: 1 }, +]; + +describe("computeSelectionHash", () => { + it("is stable for the same clip and words", () => { + const a = computeSelectionHash(clip(), words); + const b = computeSelectionHash(clip(), words); + expect(a).toBe(b); + }); + + it("changes when the range changes", () => { + const a = computeSelectionHash(clip(), words); + const b = computeSelectionHash(clip({ end_second: 25 }), words); + expect(a).not.toBe(b); + }); + + it("changes when the caption style changes", () => { + const a = computeSelectionHash(clip(), words); + const b = computeSelectionHash(clip({ suggested_caption_style: "karaoke" }), words); + expect(a).not.toBe(b); + }); + + it("changes when the title changes", () => { + const a = computeSelectionHash(clip(), words); + const b = computeSelectionHash(clip({ title: "A different title" }), words); + expect(a).not.toBe(b); + }); + + it("changes when segments change", () => { + const a = computeSelectionHash(clip(), words); + const b = computeSelectionHash( + clip({ segments: [{ start: 10, end: 15 }] }), + words, + ); + expect(a).not.toBe(b); + }); + + it("changes when the transcript words inside the range change", () => { + const a = computeSelectionHash(clip(), words); + const editedWords = words.map((w) => + w.word === "three" ? { ...w, word: "THREE-EDITED" } : w, + ); + const b = computeSelectionHash(clip(), editedWords); + expect(a).not.toBe(b); + }); + + it("ignores words outside the clip range", () => { + const a = computeSelectionHash(clip(), words); + const extraOutside = [...words, { word: "far-away", start: 500, end: 501, confidence: 1 }]; + // "far-away" is outside [10, 20) only because of its start time — adding + // a word inside the range should matter, outside should not. + const b = computeSelectionHash(clip(), extraOutside); + expect(a).toBe(b); + }); + + it("is unaffected by a hook field that isn't set", () => { + const a = computeSelectionHash(clip(), words); + const b = computeSelectionHash({ ...clip(), hook: undefined }, words); + expect(a).toBe(b); + }); + + it("changes when a hook is added", () => { + const a = computeSelectionHash(clip(), words); + const b = computeSelectionHash( + { ...clip(), hook: { start: 10, end: 12, mode: "repeat" } }, + words, + ); + expect(a).not.toBe(b); + }); +}); diff --git a/src/utils/selection-hash.ts b/src/utils/selection-hash.ts new file mode 100644 index 00000000..2baffb00 --- /dev/null +++ b/src/utils/selection-hash.ts @@ -0,0 +1,34 @@ +import { createHash } from "crypto"; +import type { SuggestedClip, WordTimestamp } from "../models/index.js"; + +/** + * Fingerprints everything about a clip that changes what actually renders: + * its range, its editorial segments and hook (if any), caption style, title + * text, and the transcript words that fall inside its range. Selecting a + * clip for export stores this; if the clip is edited afterward the stored + * hash no longer matches, which is the signal that the export no longer + * describes what was approved. + */ +export function computeSelectionHash( + clip: Pick< + SuggestedClip, + "start_second" | "end_second" | "segments" | "suggested_caption_style" | "title" + > & { hook?: unknown }, + transcriptWords: WordTimestamp[] | undefined | null, +): string { + const wordsInRange = (transcriptWords || []) + .filter((w) => w.start >= clip.start_second && w.start < clip.end_second) + .map((w) => `${w.word}:${w.start}:${w.end}`); + + const payload = JSON.stringify({ + start: clip.start_second, + end: clip.end_second, + segments: clip.segments ?? null, + hook: clip.hook ?? null, + captionStyle: clip.suggested_caption_style ?? null, + title: clip.title ?? null, + words: wordsInRange, + }); + + return createHash("sha1").update(payload).digest("hex"); +} From 41a5a54d2b90ca8e6f152271d72aa561be548dac Mon Sep 17 00:00:00 2001 From: Nika Siradze Date: Sat, 3 Oct 2026 16:50:49 +0400 Subject: [PATCH 034/110] Add engine comparison report: transcribe a sample with two engines, diff the output New services/engine_comparison.py transcribes the same sample window with two engines and reports per-20s-window word-level disagreement: Levenshtein over NFKC-normalized, casefolded words (letters/marks/numbers only), words assigned to windows by midpoint so a straddling word isn't double-counted. Windows where both engines produced nothing are flagged rather than scored as agreement. Writes comparison.json plus a self-contained comparison.html with a sample audio player and per-window seek buttons; all transcript text renders through textContent (not innerHTML), and the embedded JSON escapes ' str: + """NFKC-normalize, casefold, and strip everything but letters/marks/numbers.""" + normalized = unicodedata.normalize("NFKC", word or "").casefold() + return "".join(ch for ch in normalized if unicodedata.category(ch)[0] in _KEEP_CATEGORY_PREFIXES) + + +def normalize_words(words: list[str]) -> list[str]: + return [w for w in (normalize_word(w) for w in words) if w] + + +def word_levenshtein(a: list[str], b: list[str]) -> int: + """Levenshtein distance over word sequences (insert/delete/substitute, cost 1 each).""" + if a == b: + return 0 + if not a: + return len(b) + if not b: + return len(a) + prev = list(range(len(b) + 1)) + for i, wa in enumerate(a, 1): + curr = [i] + [0] * len(b) + for j, wb in enumerate(b, 1): + cost = 0 if wa == wb else 1 + curr[j] = min( + prev[j] + 1, # delete from a + curr[j - 1] + 1, # insert into a + prev[j - 1] + cost, # substitute + ) + prev = curr + return prev[-1] + + +def disagreement_ratio(text_a: str, text_b: str) -> float: + """Levenshtein over normalized words, divided by the longer word count. + 0.0 = identical (after normalization), 1.0 = completely different. + Two empty texts disagree 0.0 (nothing to disagree about) — callers should + flag that case separately rather than reading it as agreement.""" + words_a = normalize_words(text_a.split()) + words_b = normalize_words(text_b.split()) + denom = max(len(words_a), len(words_b)) + if denom == 0: + return 0.0 + return word_levenshtein(words_a, words_b) / denom + + +def _words_in_window(words: list[dict], window_start: float, window_end: float) -> str: + """Join words whose midpoint falls in [window_start, window_end) — matches + the midpoint-ownership rule used elsewhere so a word isn't double-counted + in two adjacent windows.""" + picked = [] + for w in words: + try: + start, end = float(w.get("start", 0.0)), float(w.get("end", 0.0)) + except (TypeError, ValueError): + continue + midpoint = (start + end) / 2.0 + if window_start <= midpoint < window_end: + text = str(w.get("word", "")).strip() + if text: + picked.append(text) + return " ".join(picked) + + +def build_windows( + words_a: list[dict], + words_b: list[dict], + total_duration: float, + window_seconds: float = 20.0, +) -> list[dict]: + windows = [] + n_windows = max(1, int(total_duration // window_seconds) + (1 if total_duration % window_seconds else 0)) + for i in range(n_windows): + start = i * window_seconds + end = min(start + window_seconds, total_duration) + text_a = _words_in_window(words_a, start, end) + text_b = _words_in_window(words_b, start, end) + both_empty = not text_a and not text_b + windows.append({ + "start": round(start, 3), + "end": round(end, 3), + "text_a": text_a, + "text_b": text_b, + "both_empty": both_empty, + "disagreement": 0.0 if both_empty else round(disagreement_ratio(text_a, text_b), 4), + }) + return windows + + +def compare_engines( + file_path: str, + engine_a: str, + engine_b: str, + *, + start_seconds: float = 0.0, + duration_seconds: Optional[float] = None, + window_seconds: float = 20.0, + model_size: str = "base", + language: Optional[str] = None, + output_dir: Optional[str] = None, + transcribe_fn=None, +) -> dict: + """Transcribe the same [start_seconds, start_seconds + duration_seconds) + window with engine_a and engine_b, and report per-window disagreement. + + transcribe_fn defaults to services.transcription.transcribe_file; tests + inject a fake to avoid running real engines. + """ + if transcribe_fn is None: + from services.transcription import transcribe_file as transcribe_fn + + result_a = transcribe_fn( + file_path, model_size=model_size, engine=engine_a, language=language, + enable_diarization=False, start_seconds=start_seconds, duration_seconds=duration_seconds, + ) + result_b = transcribe_fn( + file_path, model_size=model_size, engine=engine_b, language=language, + enable_diarization=False, start_seconds=start_seconds, duration_seconds=duration_seconds, + ) + + duration = max( + float(result_a.get("duration", 0.0) or 0.0), + float(result_b.get("duration", 0.0) or 0.0), + duration_seconds or 0.0, + ) + windows = build_windows( + result_a.get("words") or [], result_b.get("words") or [], duration, window_seconds + ) + scored = [w["disagreement"] for w in windows if not w["both_empty"]] + overall_disagreement = round(sum(scored) / len(scored), 4) if scored else 0.0 + flagged_empty = sum(1 for w in windows if w["both_empty"]) + + report = { + "file_path": file_path, + "engine_a": engine_a, + "engine_b": engine_b, + "start_seconds": start_seconds, + "duration_seconds": duration, + "window_seconds": window_seconds, + "windows": windows, + "overall_disagreement": overall_disagreement, + "both_empty_window_count": flagged_empty, + "note": "disagreement between the two engines' output, not accuracy against a transcript", + } + + if output_dir: + os.makedirs(output_dir, exist_ok=True) + json_path = os.path.join(output_dir, "comparison.json") + with open(json_path, "w", encoding="utf-8") as f: + json.dump(report, f, indent=2, ensure_ascii=False) + report["json_path"] = json_path + + # A fresh extraction of the same window, purely for the report's + #