From 304305c8b78edd1781b160e7c0e01a39bd2c27aa Mon Sep 17 00:00:00 2001 From: Joseph Leon <887316+JosephLeon@users.noreply.github.com> Date: Fri, 19 Jun 2026 08:39:41 -0500 Subject: [PATCH] Avoid re-running the AI pipeline on derived artifacts Five complementary changes so users stop accidentally re-paying for transcription + classification on content the pipeline already saw. Pattern 1+6 (render-lineage detection on add): - New renderLineage.ts walks the active project's render_history when a user drags in a file. If the path matches an entry's output, the AddPanel routes through a new "That's one of your renders" modal instead of the normal copy/reference flow. - "Continue editing intro.mov" jumps the AI tab to the original source via setActive(), so the next render reuses cached artifacts. - "Add anyway" stays as an escape hatch. Pattern 2 (UI copy): - Pacing tab + Audio tab render buttons now spell out "No AI cost on re-render": cuts and overrides reuse the cached classification; audio enhancement / denoise / ducking never touch the LLM. Pattern 4 (content-hash cache for /analyze): - New _analysis_cache.py keys cached AnalysisBundle JSONs by SHA256 of the extracted mic WAV. /analyze hashes after ingest (~1s), looks up, and skips the Groq call on hit. Cache miss falls through to the normal Whisper pass and stores the result. - Catches re-imports, copies across projects, symlinks of the same audio. Doesn't catch trims / different mic tracks (correctly). - Saves ~$0.05 per duplicate analyze. Pattern 5 (no code changes): - Verified runAllStages is idempotent on completed stages and that override / custom-cut mutations route only through the store + /render's re-plan path, never re-triggering /classify. Pattern 3 (deferred): - Splice-aware Cadence (talking to the model about a spliced output without re-running transcription on each clip) is real architecture work: artifact builder with offset math, dispatcher routing, custom cuts at splice-time. Full design captured in docs/design-splice-classification-reuse.md for a future PR. Drive-bys: re-applied two tsc -b strict-mode fixes (SplicingView discriminated-union narrowing + projectDigest export-type) that lived on the deleted public-release branch and didn't make it to master. --- app/src/components/MediaAddPanel.tsx | 159 ++++++++++++++++++++- app/src/components/RightPanel.tsx | 10 +- app/src/components/SplicingView.tsx | 7 +- app/src/lib/projectDigest.ts | 5 +- app/src/lib/renderLineage.ts | 116 +++++++++++++++ docs/design-splice-classification-reuse.md | 104 ++++++++++++++ src/cadence_lab/_analysis_cache.py | 91 ++++++++++++ src/cadence_lab/server.py | 58 +++++++- 8 files changed, 535 insertions(+), 15 deletions(-) create mode 100644 app/src/lib/renderLineage.ts create mode 100644 docs/design-splice-classification-reuse.md create mode 100644 src/cadence_lab/_analysis_cache.py diff --git a/app/src/components/MediaAddPanel.tsx b/app/src/components/MediaAddPanel.tsx index 6f17f2a..044843d 100644 --- a/app/src/components/MediaAddPanel.tsx +++ b/app/src/components/MediaAddPanel.tsx @@ -1,6 +1,12 @@ import { useRef, useState } from "react"; import { api } from "../api/client"; import { useActiveProject } from "../stores/activeProject"; +import { useProject } from "../stores/project"; +import { + describeLineage, + detectRenderLineage, + type RenderLineageMatch, +} from "../lib/renderLineage"; const VIDEO_EXTS = ["mov", "mp4", "mkv", "m4v", "avi", "webm"]; const isVideoFile = (name: string) => @@ -26,17 +32,38 @@ export function MediaAddPanel() { | { paths: string[]; suggestedMode: "copy" | "reference" } | null >(null); + // Lineage prompt: shown when the user tries to add a file that's + // actually a render in this project's history. Distinct from the add + // modal so the user can pick "continue editing source" without going + // through the copy-vs-reference dance. + const [lineagePrompt, setLineagePrompt] = useState< + | { addedPath: string; match: RenderLineageMatch; suggestedMode: "copy" | "reference" } + | null + >(null); const [uploads, setUploads] = useState>({}); const [error, setError] = useState(null); const fileInputRef = useRef(null); const projectAvailable = !!project; - /** Append a path to the pending-add queue, opening the modal if needed. */ + /** Append a path to the pending-add queue, opening the modal if needed. + * + * Special case: if the path is one of this project's previous renders, + * we surface a "continue editing the source instead?" prompt rather + * than letting the user accidentally add a derived MP4 as a fresh + * source (which would force a full pipeline re-run and burn tokens). + */ const queueAdd = (path: string, suggestedMode: "copy" | "reference") => { const trimmed = path.trim(); if (!trimmed) return; setError(null); + if (project) { + const match = detectRenderLineage(project, trimmed); + if (match) { + setLineagePrompt({ addedPath: trimmed, match, suggestedMode }); + return; + } + } setPendingAdd((prev) => { if (prev) { if (prev.paths.includes(trimmed)) return prev; @@ -46,6 +73,31 @@ export function MediaAddPanel() { }); }; + /** "Continue editing source" button on the lineage prompt: jump the + * AI tab to the source video the render was derived from. Source must + * already be in this project's sources (which it almost always is — + * a render can't exist without its source having been added). */ + const continueEditingSource = () => { + if (!lineagePrompt?.match.sourceAbsPath) return; + useProject.getState().setActive(lineagePrompt.match.sourceAbsPath); + setLineagePrompt(null); + }; + + /** "Add anyway" escape hatch: drop the lineage match and treat the file + * as a brand-new source. The user is taking the token hit knowingly. */ + const addAnywayDespiteLineage = () => { + if (!lineagePrompt) return; + const { addedPath, suggestedMode } = lineagePrompt; + setLineagePrompt(null); + setPendingAdd((prev) => { + if (prev) { + if (prev.paths.includes(addedPath)) return prev; + return { ...prev, paths: [...prev.paths, addedPath] }; + } + return { paths: [addedPath], suggestedMode }; + }); + }; + const performAdd = async (mode: "copy" | "reference") => { if (!project || !pendingAdd) return; try { @@ -150,6 +202,111 @@ export function MediaAddPanel() { onConfirm={performAdd} /> )} + + {lineagePrompt && ( + setLineagePrompt(null)} + /> + )} + + ); +} + +/** + * "Hey, this is one of your renders" prompt. + * + * Surfaced whenever a user tries to add a file that matches an entry in + * the project's render_history. The default path here is to send them + * back to the original source so their next render reuses the cached + * analysis and classification, costing zero new tokens. The "Add anyway" + * button is the escape hatch for users who genuinely want to treat the + * render as a fresh source (e.g. to A/B different classifier settings + * against the rendered output, or because they edited it externally). + */ +function RenderLineageModal({ + addedPath, + match, + onContinueEditingSource, + onAddAnyway, + onCancel, +}: { + addedPath: string; + match: RenderLineageMatch; + onContinueEditingSource: () => void; + onAddAnyway: () => void; + onCancel: () => void; +}) { + const fileName = addedPath.split("/").pop() ?? addedPath; + const sourceName = match.sourceAbsPath?.split("/").pop() ?? null; + const canContinue = Boolean(match.sourceAbsPath); + + return ( +
+
e.stopPropagation()} + > +

+ That's one of your renders +

+

+ {fileName} was produced by this + project: {describeLineage(match)}. +

+

+ {canContinue ? ( + <> + Adding it as a new source would re-run the full AI pipeline + (transcription + classification) on the rendered audio, + charging Anthropic and Groq tokens again. If you just want to + iterate on this render, continue editing{" "} + {sourceName}: tweak cuts, + overrides, or audio settings and re-render. No new AI cost. + + ) : ( + <> + This is a splice render, so there isn't a single source to + jump back to. Adding it as a new source will re-run the full + pipeline and charge new tokens. + + )} +

+ +
+ + + {canContinue && ( + + )} +
+
); } diff --git a/app/src/components/RightPanel.tsx b/app/src/components/RightPanel.tsx index d25983e..b6aef88 100644 --- a/app/src/components/RightPanel.tsx +++ b/app/src/components/RightPanel.tsx @@ -553,7 +553,8 @@ function PacingTab({ item, onOpenReview, onRender }: PacingTabProps) { {audioOn ? " and bakes in the audio settings from the Audio tab." : "."}{" "} - Produces a new MP4 in the project's renders folder. + Produces a new MP4 in the project's renders folder. No AI cost on + re-render: cuts and overrides reuse the cached classification.

{isRendering ? (
@@ -688,9 +689,10 @@ function AudioTab({ item, onChange, onRender }: AudioTabProps) {

- Applies just the audio settings — no AI cuts. Produces a new MP4 - in the project's renders folder. Doesn't require running the - pipeline first. + Audio-only render. No AI cost: tweak enhancement, denoise engine, + or ducking and re-render as often as you like without burning any + Anthropic or Groq tokens. Produces a new MP4 in the project's + renders folder.

{isRendering ? (
diff --git a/app/src/components/SplicingView.tsx b/app/src/components/SplicingView.tsx index 08d416c..6eea689 100644 --- a/app/src/components/SplicingView.tsx +++ b/app/src/components/SplicingView.tsx @@ -198,8 +198,11 @@ function Preview() { const at = clipAtPlayhead(timeline, playhead); const total = totalDuration(timeline); - const onVideoClip = at && at.clip.kind === "video"; - const currentSrc = onVideoClip ? api.sourceUrl(at!.clip.sourcePath) : null; + // Narrow `at.clip` inline so TypeScript sees the "video" variant + // (discriminated union narrowing doesn't follow through an extracted + // boolean). + const currentSrc = + at && at.clip.kind === "video" ? api.sourceUrl(at.clip.sourcePath) : null; const lastSrcRef = useRef(null); // Seek + swap source whenever the underlying clip changes. Only meaningful diff --git a/app/src/lib/projectDigest.ts b/app/src/lib/projectDigest.ts index dbdb39c..ce3646d 100644 --- a/app/src/lib/projectDigest.ts +++ b/app/src/lib/projectDigest.ts @@ -257,5 +257,6 @@ function relTime(iso: string): string { return `${Math.floor(diff / 86400)}d ago`; } -// Re-export so callers don't reach across modules for these. -export { ProjectSpliceClip }; +// Re-export so callers don't reach across modules for these. `export type` +// (not `export {}`) is required by tsconfig's `verbatimModuleSyntax: true`. +export type { ProjectSpliceClip }; diff --git a/app/src/lib/renderLineage.ts b/app/src/lib/renderLineage.ts new file mode 100644 index 0000000..ecf1b69 --- /dev/null +++ b/app/src/lib/renderLineage.ts @@ -0,0 +1,116 @@ +import type { Project, ProjectRenderHistoryEntry } from "../api/types"; + +/** + * Render lineage detection. + * + * Goal: stop users from accidentally treating a previously-rendered MP4 + * as a fresh source. Doing that would force the full pipeline to run + * again (analyze + classify, ~$0.55 in tokens) when in reality the user + * is just trying to iterate on a render they already produced. The + * correct workflow is to go back to the original source, tweak the + * project state (custom cuts, overrides, audio settings), and re-render. + * + * The project manifest's `render_history` is the source of truth: every + * render writes an entry with `{output, source, settings, label, …}`, + * including the project-relative output path. When a user drops a file + * onto the AddPanel, we check whether that path matches any entry's + * output. If so, we know: + * - what original source it was derived from + * - what settings produced it (cuts, audio, etc.) + * - how to label the offer in the UI + * + * This module is the lookup-side; the UI uses the result to surface a + * "this is a render of X — continue editing the source instead?" modal. + */ + +export interface RenderLineageMatch { + /** The render_history entry that matched. */ + entry: ProjectRenderHistoryEntry; + /** Absolute filesystem path of the source video this render was made + * from. Null when the entry didn't record one (older splice renders). */ + sourceAbsPath: string | null; +} + +/** + * Test whether ``addedPath`` is a previously-recorded render of the + * given project. Returns the matching history entry (with resolved + * source path) or null. + * + * Matching is by absolute path: the render_history stores + * project-relative paths (e.g. `renders/r001.intro.paced.mp4`), so we + * combine with `project.path` to get the absolute form and compare. + */ +export function detectRenderLineage( + project: Project, + addedPath: string, +): RenderLineageMatch | null { + const normalized = stripFileScheme(addedPath); + const projectRoot = project.path; + if (!projectRoot) return null; + + // Walk newest-first so if a user has somehow re-rendered to the same + // path (rare; we use rNNN prefixes that should be unique) the most + // recent metadata wins. + for (const entry of [...project.render_history].reverse()) { + const absOutput = joinPath(projectRoot, entry.output); + if (samePath(absOutput, normalized)) { + return { + entry, + sourceAbsPath: entry.source + ? joinPath(projectRoot, entry.source) + : null, + }; + } + } + return null; +} + +/** + * Human-readable one-liner for the lineage modal: + * "AI render of intro.mov (15 paced cuts + medium neural denoise)" + * + * Falls back to the stored `label` when settings parsing isn't useful. + */ +export function describeLineage(match: RenderLineageMatch): string { + const e = match.entry; + const sourceName = match.sourceAbsPath + ? baseName(match.sourceAbsPath) + : null; + const kind = e.type === "splice_render" ? "Splice render" : "AI render"; + if (sourceName) { + return `${kind} of ${sourceName} (${e.label})`; + } + return `${kind} (${e.label})`; +} + +// ─── helpers ─────────────────────────────────────────────────────────────── + +function stripFileScheme(p: string): string { + return p.startsWith("file://") ? p.slice("file://".length) : p; +} + +function joinPath(root: string, rel: string): string { + // Strip leading "./", "/" so we always join exactly one slash. + const cleanRel = rel.replace(/^\.?\/+/, ""); + return root.endsWith("/") + ? root + cleanRel + : `${root}/${cleanRel}`; +} + +/** + * Compare two absolute paths for equality. Tolerates a single trailing + * slash difference and a double-slash collapse, both of which crop up + * with concatenated path components. Case-sensitive (we only run on + * macOS / Linux where APFS / ext4 default to case-sensitive matches + * for tools like ffprobe — being stricter here than the filesystem is + * the safer error direction). + */ +function samePath(a: string, b: string): boolean { + const norm = (s: string) => s.replace(/\/+/g, "/").replace(/\/$/, ""); + return norm(a) === norm(b); +} + +function baseName(p: string): string { + const segs = p.split("/"); + return segs[segs.length - 1] || p; +} diff --git a/docs/design-splice-classification-reuse.md b/docs/design-splice-classification-reuse.md new file mode 100644 index 0000000..295dd02 --- /dev/null +++ b/docs/design-splice-classification-reuse.md @@ -0,0 +1,104 @@ +# Design: splice classification reuse + +**Status:** deferred (not implemented). Captured here so the next contributor +isn't designing from scratch. + +## Problem + +When a user assembles a splice timeline from sub-ranges of multiple source +videos, renders it to a single MP4, and then says *"Cadence, remove every +sniffle from this"*, the obvious-but-wrong move is to treat the spliced +MP4 as a new source. That triggers the full pipeline (analyze + classify ++ events scan) on the rendered audio, charging fresh Anthropic and Groq +tokens for transcription/classification work we already did on each +source clip independently. + +The right move is to recognize that we have: +- Original analyses (`/artifacts/.analysis.json`) for + every source clip in the timeline +- Optional audio-event scans (`.events.json`) for any source the + user already scanned +- Optional visual indexes (`.frames.npz`) likewise + +So for splice-aware Cadence queries, the answer is to walk each splice +clip, look up the corresponding artifacts for its source, *translate +timestamps from source-time to splice-output-time*, and produce a +synthetic merged view that Cadence's tools can read from. Zero new +tokens; everything is already on disk. + +## What needs to change + +### Backend + +1. **A "splice digest" artifact builder.** Given a `SpliceState.timeline`, + walk each clip and emit a merged view that mirrors the shape of the + per-source artifacts (`pauses`, `fillers`, `audio_events`) but with + timestamps offset by `cumulative_output_time - source_clip_start`. + Probably lives in a new `src/cadence_lab/splice_artifacts.py`. + +2. **`/cadence/query` accepts splice context.** When the active view is + the splice tab, the digest builder above runs, the merged artifacts + replace per-source `list_pauses` / `list_fillers` / `list_audio_events` + tool results, and any propose actions (`add_custom_cut`, + `create_highlight_clip`) are recorded against the splice timeline, + not against a source. + +3. **Custom cuts in the splice render.** Today + [`splice_render`](../src/cadence_lab/renderer.py) does pure ffmpeg + concat with no cut layer. Need to add a per-clip "cuts" pass that + trims sub-ranges out of each splice clip before concatenation, so + "remove the sniffle at 0:34 of splice output" actually modifies the + final MP4. + +### Frontend + +4. **Splice-time custom cuts in state.** The `useSplicing` store needs + a `customCuts: CustomCut[]` field per timeline (or per-clip) so the + user / Cadence can stage cuts before render. + +5. **Cadence dispatcher routes to splice ops in splice view.** The + existing `applyCadenceAction` is source-centric (mutates + `useProject.activeMedia`). When the active view is splice, the same + action types need to mutate `useSplicing` instead. + +6. **UI for the splice cuts.** Same audition / preview / remove + affordances as the AI tab's custom-cut list, but bound to the splice + timeline. + +## Hard parts + +- **Audio event timestamps span clip boundaries**: a sniffle whose end + falls in the next splice clip needs careful range math. Easiest fix: + drop any event that straddles a boundary; document it. +- **The visual index is per-source.** Visual search across a splice + ("find the part of the splice where the dog appears") requires + searching each source's index separately and aggregating with + timestamp translation. Plausible, just more code. +- **Re-rendering the splice with cuts changes the timeline shape**, so + output-time-to-source-time math has to re-run on every render. State + needs to be reactive. + +## Why deferred + +The pattern below it (the lineage prompt) covers the common case: most +users producing a splice want to iterate by tweaking the splice itself, +not by piping the rendered MP4 back through the AI pipeline. The lineage +prompt prevents the most expensive footgun (~$0.55 in tokens per +accidental round-trip). Splice-aware Cadence is the next level of +polish and worth a few days of focused work, not a sprinkle into an +already-large PR. + +## Quick recap of what *does* work today + +- Single-source AI tab: full pipeline runs once, then custom cuts + + overrides + audio settings re-render for free. +- Splice render: assembles clips, no AI calls on render. +- Audio-only render: applies enhancement, denoise, ducking with zero + AI calls. +- Lineage prompt: blocks the "treat a render as a source" footgun. +- Content-hash cache: re-importing the same audio skips the Whisper + call. + +The gap this design closes is specifically *"talk to Cadence about a +spliced output"*. Until built, the UX answer is: ask Cadence about the +individual sources before splicing, not after. diff --git a/src/cadence_lab/_analysis_cache.py b/src/cadence_lab/_analysis_cache.py new file mode 100644 index 0000000..89ff9cb --- /dev/null +++ b/src/cadence_lab/_analysis_cache.py @@ -0,0 +1,91 @@ +"""Content-hash cache for analysis artifacts. + +Reanalyzing a source video is expensive: Groq Whisper transcription is the +biggest line item per video at default settings. But the inputs to that pass +are deterministic functions of the *audio stream* (we extract a 16 kHz mono +mic-only WAV, then run Whisper on it). If the bytes of the extracted mic +WAV are the same, the resulting transcript will be too. + +So we keep a content-addressed cache keyed by the SHA256 of the extracted +mic WAV. ``ingest`` runs first and produces that WAV anyway; we hash its +bytes (cheap, ~1s for a 30-min recording) and look up a cached analysis +before paying for transcription. + +Cache shape:: + + /analysis_by_hash/ + .json # the speech-side of AnalysisBundle, with audio_path + # left as a placeholder that gets rewritten on restore + +When ``/analyze`` runs, we hash the new source's extracted mic WAV. If a +cache file exists, we restore the cached speech object, rewrite paths to +point at the current source, and skip Whisper. Cache miss falls through +to the normal pipeline and writes a new cache entry on success. + +This catches: +- Re-importing the same file (different filename, same bytes) +- Copying a source between projects +- Re-running analyze on the same file after a project move +- Symlinked / hard-linked duplicates + +It does NOT catch trims, edits, or different mic-track selections, since +those produce different audio bytes and will hash differently. That's the +correct behavior. +""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +from .paths import cache_dir + + +def _cache_dir() -> Path: + d = cache_dir() / "analysis_by_hash" + d.mkdir(parents=True, exist_ok=True) + return d + + +def hash_audio_file( + audio_path: Path, + chunk_bytes: int = 1 << 20, # 1 MB chunks +) -> str: + """SHA256 of a file's bytes. Streamed in chunks so memory stays flat + regardless of file size. Suitable for the extracted mic WAVs since + they're already canonicalized (16 kHz mono PCM).""" + hasher = hashlib.sha256() + with audio_path.open("rb") as f: + while True: + chunk = f.read(chunk_bytes) + if not chunk: + break + hasher.update(chunk) + return hasher.hexdigest() + + +def lookup(audio_hash: str) -> dict | None: + """Return the cached AnalysisBundle JSON for this hash, or None.""" + path = _cache_dir() / f"{audio_hash}.json" + if not path.exists(): + return None + try: + return json.loads(path.read_text(encoding="utf-8")) + except json.JSONDecodeError: + # Corrupt cache entry — drop it so the next run can rewrite. + try: + path.unlink() + except OSError: + pass + return None + + +def store(audio_hash: str, bundle_dict: dict) -> None: + """Persist an AnalysisBundle dict under its audio-hash key.""" + path = _cache_dir() / f"{audio_hash}.json" + # Atomic-ish: write to a sibling temp + rename, so a crash mid-write + # doesn't leave a half-written JSON that lookup() then can't parse. + tmp = path.with_suffix(".json.tmp") + tmp.write_text(json.dumps(bundle_dict, indent=2), encoding="utf-8") + tmp.replace(path) diff --git a/src/cadence_lab/server.py b/src/cadence_lab/server.py index a6596e7..ef9c44e 100644 --- a/src/cadence_lab/server.py +++ b/src/cadence_lab/server.py @@ -682,16 +682,62 @@ def run() -> None: cb = _progress_for(job) cb(0.0, "Extracting mic-only audio...") ing = ingest(source=src, mic_track_index=req.mic_track) - speech = run_analyze( - audio_path=ing.normalized_audio_path, - backend=req.backend, - language=req.language, - progress=cb, + + # Content-hash cache: hash the extracted mic WAV and skip the + # Whisper call if we've transcribed identical audio before + # (re-imports, copies across projects, etc.). The hash takes + # ~1s for a 30-min recording; transcription takes 30-60s on + # Groq and costs ~$0.05. + from . import _analysis_cache + + audio_hash = _analysis_cache.hash_audio_file( + ing.normalized_audio_path ) + cached = _analysis_cache.lookup(audio_hash) + if cached is not None: + cb( + 0.95, + "Reusing cached transcription (matching audio found, no AI cost)…", + ) + # Rewrite the audio_path so downstream consumers reading + # the bundle find the WAV at *this* source's canonical + # location, not the original. + speech_dict = cached["speech"] + speech_dict["audio_path"] = str(ing.normalized_audio_path) + from .models import SpeechAnalysis as _SpeechAnalysis + + speech = _SpeechAnalysis.model_validate(speech_dict) + else: + speech = run_analyze( + audio_path=ing.normalized_audio_path, + backend=req.backend, + language=req.language, + progress=cb, + ) + bundle = AnalysisBundle(ingest=ing, speech=speech) out = analysis_path(src) out.write_text(json.dumps(bundle.model_dump(mode="json"), indent=2)) - job.finish("done", result={"analysis_path": str(out)}) + + # On cache miss, persist the bundle so the next analyze of the + # same audio (same hash) skips Whisper entirely. + if cached is None: + try: + _analysis_cache.store( + audio_hash, bundle.model_dump(mode="json") + ) + except Exception: + # Cache write failures are non-fatal — the analysis + # still succeeded; we just don't get the speedup next time. + pass + + job.finish( + "done", + result={ + "analysis_path": str(out), + "cache_hit": cached is not None, + }, + ) except Exception as e: job.finish("error", error=str(e))