diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 8a2b1fe..eaa67c7 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -19,14 +19,13 @@ "name": "bmad-manticore", "source": "./", "description": "AI video production pipeline: brain dump to a rough cut sitting in your editor, in your own words, with approval gates at every taste decision.", - "version": "2.0.0", + "version": "3.0.0", "author": { "name": "Brian (BMad) Madison" }, "skills": [ "./skills/mc-agent", "./skills/mc-setup", - "./skills/mc-ograf", "./skills/mc-pipeline", "./skills/mc-new", "./skills/mc-braindump", diff --git a/.github/workflows/quality.yaml b/.github/workflows/quality.yaml index 3b7f6c8..fe2ac65 100644 --- a/.github/workflows/quality.yaml +++ b/.github/workflows/quality.yaml @@ -3,7 +3,7 @@ name: Quality & Validation # Runs the module's own gates on every PR: # - every skill test suite (uv run, PEP 723; ffmpeg and chromium installed # so the render and HTML-graphics suites exercise instead of skip) -# - the genericity release gate (nothing user-, brand-, or show-specific ships) +# Ship-hygiene rules are review rules, not lint: docs/review-rules.md. "on": pull_request: @@ -42,5 +42,3 @@ jobs: done exit $fail - - name: Genericity release gate - run: uv run skills/mc-setup/scripts/lint_genericity.py skills/ docs/ README.md CHANGELOG.md diff --git a/.gitignore b/.gitignore index 9719e87..90944b9 100644 --- a/.gitignore +++ b/.gitignore @@ -4,3 +4,7 @@ _inbox/ __pycache__/ *.pyc node_modules/ + +# Quality-scan reports (generated by bmad-workflow-builder analyze) +.analysis/ +.pytest_cache/ diff --git a/AGENTS.md b/AGENTS.md index 6f30b6e..764aaee 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -12,33 +12,41 @@ Taste lives in files. Mechanics live in scripts. Skills are thin routers between |---|---| | `.claude-plugin/marketplace.json` | Module manifest (install via `npx bmad-method install --custom-source `) | | `skills/module.yaml` + `skills/module-help.csv` | Module identity and the help-catalog rows (canonical BMad schema). The installer merges every module's help csv into `{project-root}/_bmad/_config/bmad-help.csv`; mc-agent and the bmad-help core skill read that merged catalog. A new or changed skill must update its module-help.csv row | -| `skills/mc-agent/` | Manny the Manticore, the persona agent and studio front door: a skill whose `[agent]` block in `customize.toml` carries the persona and capabilities menu (the BMad agent pattern); routes to the other skills, never does stage mechanics itself | +| `skills/mc-agent/` | Manny the Manticore, the persona agent and studio front door: its SKILL.md carries the persona and the capabilities menu; routes to the other skills, never does stage mechanics itself | | `skills/mc-pipeline/` | The router; owns `PIPELINE.md`, the master stage/gate/project.json contract | -| `skills/mc-setup/` | Configuration skill: writes the studio config (`[modules.manticore]` in `{project-root}/_bmad/custom/config.toml`); its `customize.toml` carries the full `[defaults]`; `assets/` holds the templates it copies into the studio (tokens, blacklist starter, voice-bible spec, format profiles) | -| `skills/mc-ograf/` | OGraf graphics authoring (scaffold, verify, spec references); gated on `[editor] ograf-editable` for the editor lane, always available for the OBS/SPX-GC live lane | +| `skills/mc-setup/` | Configuration skill: writes the studio config (`[modules.manticore]` in `{project-root}/_bmad/custom/config.toml`); `assets/studio-defaults.toml` carries the full `[defaults]` seed, and the rest of `assets/` holds the templates it copies into the studio (tokens, blacklist starter, craft checklist, the production-bible and voice-bible specs, format profiles) | | `skills/mc-audio/` | Audio service skill, not a stage: farms sound for other skills (Kokoro TTS/dialogue, MusicGen beds, AudioLDM2 SFX) from the `[audio]` lanes, local-first with paid rungs opt-in; heavy venv and model caches live in the creator's `{engines-path}/audio-lab` workspace | -| `skills/mc-*/` | The 12 stage skills; each resolves the studio config + its own `customize.toml` on activation and stops at gates | +| `skills/mc-*/` | The 11 stage skills; each resolves the studio config on activation and stops at gates | | `docs/user-guide.md` | "Configure your own Manticore studio", the end-user walkthrough | ## Conventions (binding when editing this module) +Reviewers: [docs/review-rules.md](docs/review-rules.md) is the ship-hygiene checklist, checked by code review agents rather than any lint. + - Nothing user-specific ships in the module. The creator's identity, brand, voice, paths, and tools live in their project via the studio config (`[modules.manticore]` in `_bmad/custom/config.toml`) and `{brand-path}`. If you find a personal name, brand color, or machine path in module content, that is a bug. - Config keys are kebab-case (`brand-path`). API keys never appear in the TOML or any file; only env var names. -- A skill reads ONLY its own folder, the installed core scripts (`{project-root}/_bmad/scripts/`), and project files. Never another skill's folder (some harnesses forbid it). Config resolution uses the installed `resolve_config.py` (studio config) and `resolve_customization.py` (per-skill trio: packaged `customize.toml` defaults, `_bmad/custom/.toml`, `.user.toml`); the module bundles no resolver of its own. Skills must work under any harness that resolves skill folders; nothing may depend on Claude-specific features beyond the SKILL.md format itself. +- A file that shapes the creator's output is theirs: it ships in `mc-setup/assets/` as the install source (`tokens.template.json`, `blacklist-starter.md`, `craft-checklist.md`, the two bible specs, the format profiles), mc-setup copies it into `{brand-path}` or `{formats-path}` on first run and never overwrites it after, and the consuming skill reads only the creator's copy. A template that shapes a pipeline artifact rather than the creator's taste stays module-owned in the skill that writes it: `mc-cut/assets/editorial-review-template.md` is the shape of `cut/editorial-review.md`, which mc-beats reads for its hand-to-beats seeds, so its sections are a downstream contract and it stays where it is (decided 2026-07-26). +- A skill reads ONLY its own folder, the installed core scripts (`{project-root}/_bmad/scripts/`), and project files. Never another skill's folder (some harnesses forbid it). Config resolution uses the installed `resolve_config.py` (studio config, `_bmad/custom/config.toml` with `config.user.toml` layered over it); the module bundles no resolver of its own, and no skill carries a private override surface. Skills must work under any harness that resolves skill folders; nothing may depend on Claude-specific features beyond the SKILL.md format itself. - Scripts are invoked ONLY via `uv run` (never bare `python3`), and every script carries PEP 723 inline metadata (`# /// script` block with `requires-python = ">=3.11"`; declare dependencies there when a script needs any, so uv provisions them with no venv setup). Prefer stdlib. Every script lives in the skill that runs it; a script needed by more than one skill is duplicated into each. Scripts take explicit arguments (resolved paths, blacklist path) from the calling skill and do no config discovery of their own. -- Editor-agnosticism: `cut/edl.json` is the neutral source of truth; editor-specific behavior keys off `[editor]` in the config (timeline-format, ograf-editable). Never hardwire Resolve into a stage that other editors' users run. +- Editor-agnosticism: `cut/edl.json` is the neutral source of truth; editor-specific behavior keys off `[editor]` in the config (timeline-format). Never hardwire Resolve into a stage that other editors' users run. - Stubs carry their full I/O contract in the docstring and exit with a pointer to it. - Gates are sacred: no edit may let a stage proceed past a gate without the creator's recorded approval. -- Docs style: no em-dashes, blank line after every heading, no bold in list items, ISO dates. +- A check the pipeline claims to perform must be a script that exits non-zero. "Inspect X before proceeding" in a skill file is only acceptable next to a script that FAILS when X is wrong. This is the lesson of the 2026-07-24 cut-pipeline failures, where four separate defects shipped through the same hole: QC frames that were extracted but never asserted on, boundary frames eyeballed while the audio underneath was wrong, beat anchors that were a checklist line with no script, and a transcript nothing ever checked. Every one was documented and none could halt. Taste lives in files and mechanics live in scripts; an assertion is a mechanic, never a judgment call left to whoever is running the stage. ## Design invariants (settled decisions; change only with the maintainer's sign-off) - Manticore always renders (render-first; maintainer sign-off recorded 2026-07-07, replacing the earlier editable-timeline-never-baked invariant). Every cut iteration produces a fast low-res preview; once the graphics stage has rendered overlays, the preview re-renders with them composited; at gate 4 a final-quality render is offered. The editor timeline export (per `[editor] timeline-format`) and all assets (edl.json, cutplan, overlays) are ALWAYS still produced alongside, so the creator can move into their editor at any step. The creator confirms this default during mc-setup. - Local-first defaults, paid vendors opt-in only: no paid or metered vendor (ElevenLabs or any future TTS/SFX/music provider) ships in any default, key names included. Paid lanes exist only as explicit opt-in choices made during setup, and their key sourcing is mentioned only inside the opt-in branch of the interview. - parakeet-mlx (model parakeet-tdt-0.6b-v3) is the reference cutting transcript: free, local, word timestamps, and empirically preserves verbatim fillers (validated on real footage 2026-07-05). Generic Whisper is not a substitute because it normalizes fillers away. Alternative providers (elevenlabs-scribe, deepgram-nova3) go behind the `[transcription]` switch with the same output shape. +- EVERY transcription lane windows in short isolated windows (20s, 3s overlap). This is not an implementation detail to tune: parakeet silently drops whole paragraphs inside long windows, with no error, and long-window transcription is what corrupted a real project on 2026-07-24. Measured on that take: 120s chunks lost three paragraphs, 90s windows still lost content, 20s windows were complete. Nothing may transcribe a whole file in one pass, and `verify_transcript.py` must pass before any transcript is consumed. +- The two-source rule: the TRANSCRIPT is the authority on CONTENT, the AUDIO is the authority on TIMING. parakeet absorbs pauses into the preceding word's end, so transcript gaps read about 0.0 across real dead air and word ends reach past the sound. No stage derives a cut time, a beat time, or a silence from transcript timestamps; silence comes from `analyze_audio.py` and cut edges snap into it. +- A deliverable path is never written directly. Renders go to a per-process temp file, decode-validate, check they have not been superseded, then atomically move into place. Any shared artifact a concurrent process may read (edl.json, preview proxies) is written the same way. Two processes interleaving into one output path handed a creator an unplayable preview on 2026-07-24. - Generated footage never depicts UI or text that must be accurate; real UI comes from screen recordings. -- OGraf output only where the target supports it: `[editor] ograf-editable` for the editor lane, always for the OBS/SPX-GC live lane. Default deliverable is baked alpha, which works everywhere. +- Baked alpha is the only graphics deliverable, and it works in every editor. The editable-graphics lane (OGraf, gated on `[editor] ograf-editable`) was removed 2026-07-26: it was a second authoring path, a second spec to conform to, and a second set of verification scripts, all to serve one editor version. `ograf` remains a permanent compatibility alias for `hyperframes`, and `ograf-editable` is ignored wherever a pre-3.0.0 config still carries it. - Four approval gates (outline, cutplan, beats, final) are hard stops; nothing weakens them. +- Absence is never silent. A skill loads the creator files it needs on activation; when one is missing it names the file, says what it cannot do without it, routes to mc-setup, and stops. The only permitted alternative is a fallback the skill states out loud at the moment it bites, and no fallback may change creative output without saying so. +- Bare paths are the current video project. A bare path in skill prose (`beats/beats.md`, `cut/edl.json`, `project.json`) resolves against `{video-path}`, the video project being worked on. Files inside the skill's own folder always carry `{skill-root}`, and `{project-root}` keeps its meaning of the repo working directory. A path led by a skill name (`mc-cut/scripts/preflight.py`) records which skill owns a file; a skill still reads only its own folder. +- Every file in a skill addresses the model executing it, never a human reader. This binds everything a skill ships and progressively discloses, references, asset prose, templates and engine docs, not only SKILL.md: no citation blocks, no source URLs, no provenance or dated research claims. A URL a skill may name is one it installs from or runs, never one it cites; provenance worth keeping goes in the commit message or `TODO.md`. - Taste in files, mechanics in scripts (via uv), skills as thin routers, so lesser models can run the pipeline. ## Repo rules diff --git a/CHANGELOG.md b/CHANGELOG.md index c850f46..e75e1ed 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,9 +2,69 @@ All notable changes to BMad Manticore are documented here. Dates are ISO (YYYY-MM-DD). -## 2.0.0 - Unreleased +## 3.0.0 - Unreleased -The big release: one motion-graphics engine with the full HyperFrames toolkit behind it, cross-platform support, a final render that only re-does what changed, and delivery polish (loudness, captions, OBS alpha). Upgrading from 1.x is a clean reinstall (see README): your brand, voice bible, and format profiles live in your studio folder, not in `_bmad/`, so they survive and onboarding picks them back up. +The big release: one motion-graphics engine with the full HyperFrames toolkit behind it, cross-platform support, a final render that only re-does what changed, delivery polish (loudness, captions, OBS alpha), and a cut stage rebuilt from the ground up after its first real project. Upgrading from 1.x is a clean reinstall (see README): your brand, voice bible, and format profiles live in your studio folder, not in `_bmad/`, so they survive and onboarding picks them back up. + +### The cut stage, rebuilt after its first real project + +The cut pipeline had never been run end to end on a long 4K take; the first one that was corrupted the edit in ways nothing detected. What follows fixes the causes and, more importantly, makes each of them impossible to ship silently again. + +### The critical one: transcription was silently losing speech + +- Every transcription lane now works in short isolated windows (20s, 3s overlap). The Apple Silicon lane previously handed whole files to the model in one pass, which ran out of Metal memory above roughly 15 minutes of 4K, and the obvious workaround (larger chunks) made it far worse: parakeet silently drops whole paragraphs inside long windows, with no error and no warning. On a 20.5 minute take, 120s chunks lost three paragraphs of clearly-spoken content and 90s windows still lost some; 20s windows were complete. The cross-platform lane had been windowing correctly all along, so the reference lane was the broken one. Both lanes now share one windowing driver. +- New `verify_transcript.py`, and no transcript reaches the cutter without passing it. It finds audio above the silence floor that produced no words, which is dropped speech by definition, and names the regions with timecodes. Nothing had ever checked that a transcript was complete; that single missing check is what let everything else happen. Note the check is built from the audio side deliberately: scanning transcript word gaps would inherit the very bug it exists to catch. + +### The cut is built on the audio now, not on the transcript's idea of time + +- New `analyze_audio.py` produces a silence map from the audio, and it is the timing source of truth for the whole stage. The old detector computed silence from transcript timestamps, but parakeet absorbs a pause into the preceding word's end, so those gaps read about 0.0 across real dead air. On a take with over five minutes of dead air it found 12 silences; the audio has about 400, totalling around 300 seconds. The resulting edits were loose and full of stalls. +- Every cut candidate's edges now snap into an audio-verified silence. "Never cut inside a word" stops being an assertion about timestamps and becomes structural: a cut inside real silence cannot clip a word. This also fixes an audible artifact where a stutter trim clipped the repeat's onset, because its end came from a pause-absorbed word end. + +### The cut stage is an editor now, not a janitor + +- Dead-air tightening: interior silences are trimmed down to a 200ms beat rather than left or flattened, with sub-threshold micro-beats preserved so the result is tight without sounding machine-gunned. +- Section re-reads are caught in full. The old matcher looked 16 words ahead for a 3-word repeat, which undersized a real paragraph redo from 34s to 11s and left the abandoned take in the video. The new one matches long runs inside a locality window measured in seconds, which is what tells a redo apart from a deliberate callback minutes later. +- Bloopers are their own candidate type. An explicit expletive sat in the first real cut and would have shipped; nothing was looking for one. Severity reflects context, so scripted usage is flagged for an ear rather than treated as a flub. +- Filler detection respects the creator's voice. It used to flag every sentence-initial "so" as filler, including all 19 of them on one take, while that creator's voice bible named "so" as their natural connective glue. Cadence words are taste, so they now live in the voice bible's machine-readable `cadence` block, and the built-in list no longer contains them. +- New editorial pass, between the mechanical cut and gate 2. It reads the EDITED transcript (what survived the cut, which is not the script) as an argument against the brief, and recommends content-level changes under a subtract-only constraint: cut, re-record, hand-to-beats, or consent-gated generate. Nothing is auto-applied; gate 2 now presents the mechanical trims and the content calls as one list. Its hardest rule is written from experience: never apply a finding from a transcript-read timecode without re-detecting the span against the audio first. + +### Renders stop corrupting themselves, and stop taking an hour + +- No render writes its output path directly. Each goes to a per-process temp file, decode-validates with zero tolerance for errors, checks it has not been superseded by a newer EDL, and only then atomically moves into place. A stale background render previously finished late and interleaved bytes with the current one, handing over an unplayable file. A corrupt mp4 still reports a plausible container duration, which is why the old duration-only check saw nothing. +- The composited preview is minutes instead of an hour. Overlays pack into the fewest non-overlapping time lanes and only the lanes are stacked, so compositing depth is the number of overlays on screen at once rather than the total (56 became 2 on the real project), and previews cut from a cached low-res proxy of each source instead of seeking into a 4K master once per segment. + +### Baked-in frame defects are caught and can be fixed + +- Source QC asserts and halts. It samples frames across each take (not just the first and last, which cannot see a defect that starts mid-recording), detects a flat decorative border ring or an active area whose aspect does not match the container, reports the inferred content rectangle, and stops the stage. A recorded-in border and off-centre framing previously passed preflight, transcription, the EDL, gate 2 and the render untouched, and were caught by eye after the cut was locked. +- New `normalize_source.py` gives the pipeline a spatial capability it simply did not have: corrective crop and reframe, emitting a corrected master registered as the project source. The EDL is time-only, so a baked-in border had nowhere to be fixed. It runs before beats and graphics, since overlays are positioned against the canvas, and it changes no timecodes (the script asserts this and refuses to publish otherwise), so an existing transcript, EDL and cutplan stay valid: no re-transcribe, no re-cut. + +### Beat timing is verified, not asserted + +- New `verify_anchors.py` re-derives every beat's time from its anchor word through the EDL and fails on any beat that does not land within half a second of it, on any anchor missing from the transcript, and on any anchor sitting in a span the cut removed. mc-beats' checklist had claimed this for a long time with no script behind it, and mc-graphics now refuses to build against a table that has not passed. + +### Thresholds calibrated against real footage, not guessed + +Every threshold in the new cut path was set by measuring the 20.5 minute take that exposed these bugs, in both its known-good and known-broken transcripts. The first-pass values were guesses, and three of them were wrong: + +- Dead-air floor 0.45s to 0.30s. That take has 383 seconds of silence across 929 intervals. A 0.45s floor reaches 87 percent of the trimmable dead air; 0.30s reaches 99 percent, worth about 29 extra seconds in a 20 minute video, which is exactly the "loose" quality the first cut was criticized for. Below 0.30s the gain is under a percent and it starts eating the speaker's rhythm (their median silence is 0.19s, which is cadence, not dead air). +- Dropped-speech threshold 2.5s to 1.0s. Sweeping it over both transcripts, the good one produces ZERO false positives all the way down to 0.75s, because anything under the silence floor is already classified as silence and never reaches the check. The cautious 2.5s bought no safety and missed a real dropped region that 1.0s catches. On the broken transcript the gate now reports five dropped regions, including one at 2:36 that the original bug report never found. +- Audio map granularity 0.3s to 0.10s. The map has to be finer than anything that consumes it. Edge snapping needs the 0.1 to 0.2s gaps between doubled words, and those are precisely the intervals a coarse map omits. The gate now also warns when it is handed a map too coarse to scan against, rather than failing a good transcript with no explanation. +- Blooper context tightened. Asking for "a 0.5s pause within 3s" flagged the scripted line "that damn term" as almost certainly a flub. Measured, the separation is not subtle: 0.77s of silence beside the scripted line, 7.65s beside the real "Oh fuck." A blooper is next to a STOP, not a breath, so it now asks for 2s of silence within 1.5s and the two classify correctly. + +### OGraf removed: one graphics deliverable, every editor + +- The mc-ograf skill and the editable-graphics lane are gone. OGraf produced graphics that stayed editable inside DaVinci Resolve 21+ and could be click-triggered live in OBS/SPX-GC, but it cost a second authoring path, a second spec to conform to, and its own scaffold and verify scripts, all to serve one editor version. Baked alpha overlays, which every editor imports, are now the only deliverable. +- Nothing in an existing studio breaks. `ograf` joins `remotion` as a permanent compatibility alias for `hyperframes` wherever an engine is named, so an in-flight beat table or a copied format profile keeps working and no creator file is rewritten. `[editor] ograf-editable` is retired: a config written before 3.0.0 may still carry it and it is simply ignored. +- Livestream lower thirds and topic cards are now self-contained local HTML styled from tokens.json, alongside the scenes that already worked that way. SPX-GC and OBS browser sources drive them the same as before, without an editor-specific package format. + +### The rule behind all of it + +- New binding convention in AGENTS.md: a check the pipeline claims to perform must be a script that exits non-zero. Four separate defects here shipped through the same hole, QC frames extracted but never asserted on, boundary frames eyeballed while the audio underneath was wrong, beat anchors as a checklist line with no script, and a transcript nothing ever checked. All four were documented, and none could halt. +- New `verify_edl.py`, which applies that rule to the cut's own deliverable. The EDL is written by hand and rewritten when the creator's editorial calls come back, and nothing had ever read it back: "no cut lands inside a word" and "every segment has quote and reason" were checklist prose with no enforcement, on the one artifact every later stage depends on. It now fails any boundary that does not rest in an audio-verified silence, reporting how far off it is and whether it landed mid-word. It follows the two-source rule rather than fighting it: because a pause-absorbed word end reaches past the sound, a correct cut can sit inside a word's timestamps, so the audio decides and the word span is context on an already-failing boundary, never a verdict of its own. +- New `snap_spans.py`. Applying the creator's approved editorial cuts used to say "snap the edges into silences" as an instruction to follow by hand, at the exact point where getting it wrong cuts the wrong words. The arithmetic is now a script, snapping is directional so a span can only widen and never collapse onto itself, and anything that cannot reach a silence is reported rather than quietly treated as safe. +- Blocking gates now carry an acknowledged override. The transcript gate takes `--accept-region - --reason ""`, matching what source QC already had, because not every audible span with no words is lost speech: a laugh, a music bed, or an off-mic aside reads the same way to a coverage scan. A gate with no way past it gets worked around in ways that leave no trace, which is worse than one that records who signed off and why. +- `cutplan.py` refuses to run when `-o`, `--audio-map` or `--voice-bible` are supplied twice. The skill appends the customization flag string after its own arguments and argparse lets the later one win, so an override file could have pointed the audio map somewhere else and broken the two-source rule with nothing to show for it. The boundary was a comment; it is now enforced. +- Published or on-YouTube source should pull captions with `yt-dlp` rather than running local ASR, which is faster, free at any length, and sidesteps the local-model failure mode entirely. Local ASR only ever ran on raw unpublished recordings, which is exactly why these bugs went unnoticed for so long. ### Motion graphics: one engine, the whole HyperFrames toolkit @@ -55,7 +115,7 @@ Code-complete and unit-tested; still pending validation on real Windows and Linu - Run mc-setup against the existing studio. It detects the 0.x config and runs a delta interview: render consent, the video style interview, and the live-tool question, backfilling the `[render]`, `[style]`, `[cta]`, `[live]`, and `[audio]` studio-config tables from the shipped defaults, and scaffolding the Production Bible seeded from existing brand assets. - In-flight projects need no migration: beat tables without the new columns are accepted, and existing `cut/` artifacts remain valid. -- If your recorded footage uses the old marker cue, set the cue at setup (mc-setup records it as a `--marker-cues` override in mc-cut's `cutplan_flags`, in `_bmad/custom/mc-cut.toml`) or pass `--marker-cues "question from claude"` directly. +- If your recorded footage uses the old marker cue, set the cue at setup (mc-setup records it as a `--marker-cues` override in `cutplan-flags`, in the `[cut]` sub-table of the studio config) or pass `--marker-cues "question from claude"` directly. ### Headline features diff --git a/README.md b/README.md index af42b6f..d582c91 100644 --- a/README.md +++ b/README.md @@ -2,7 +2,7 @@ # BMad Manticore -[![Version](https://img.shields.io/badge/version-2.0.0-blue)](.claude-plugin/marketplace.json) +[![Version](https://img.shields.io/badge/version-3.0.0-blue)](.claude-plugin/marketplace.json) [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE) [![Python Version](https://img.shields.io/badge/python-%3E%3D3.11-blue?logo=python&logoColor=white)](https://www.python.org) [![uv](https://img.shields.io/badge/uv-package%20manager-blueviolet?logo=uv)](https://docs.astral.sh/uv/) @@ -10,9 +10,9 @@ **From brain dump to a rendered, graphics-rich video, in your own words.** -## New in 2.0 +## New in 3.0 -Manticore now runs entirely on [HyperFrames](https://hyperframes.heygen.com) for motion graphics, with its Agent Skills installed and favored at setup so the agent can reach the whole toolkit: color grading, background removal, WebGL shader transitions, kinetic captions, data-viz, 3D device mockups, HDR and 4K delivery, and a 100-plus block catalog. It all runs locally, with no account or credits. Remotion is retired; its license and React model no longer fit a frame-deterministic pipeline. The final render is now incremental too (a fix on a long video re-renders in seconds, not minutes), finals are loudness-normalized by default, and Windows, Linux, and Intel Mac lanes are code-complete. See the [changelog](CHANGELOG.md) for everything that changed. +Motion graphics run entirely on [HyperFrames](https://hyperframes.heygen.com), with its Agent Skills installed and favored at setup so the agent can reach the whole toolkit: color grading, background removal, WebGL shader transitions, kinetic captions, data-viz, 3D device mockups, HDR and 4K delivery, and a 100-plus block catalog. It all runs locally, with no account or credits. The final render is incremental (a fix on a long video re-renders in seconds, not minutes), finals are loudness-normalized by default, and Windows, Linux, and Intel Mac lanes are code-complete. The cut stage was rebuilt after its first real 4K project: transcription windows in short isolated passes and is verified for completeness before anything consumes it, cut timing comes from an audio silence map rather than transcript timestamps, renders validate and move into place atomically instead of writing their output path directly, and an editorial pass reads the edited transcript against the brief before gate 2. See the [changelog](CHANGELOG.md) for everything that changed and why. Upgrading from an earlier version: back up anything custom you want to keep, then remove the `_bmad/` and `_bmad-output/` folders from your studio and reinstall (see [Install](#install)). Start your agent and say `hey manny lets get this all set up!`, then follow onboarding. Your brand kit, voice bible, and format profiles live in your studio folder (not in `_bmad/`), so they survive the reinstall: onboarding finds them, and if it does not, point it at them so it can reuse or update them. @@ -57,10 +57,10 @@ my-studio/ <- install here, run everything from here _bmad/custom/config.toml <- studio config ([modules.manticore]; mc-setup writes it) manticore/ brand/ <- tokens.json, production-bible.md, voice-bible.md, blacklist.md, - exemplars/, headshots/ + craft-checklist.md, exemplars/, headshots/ formats/ <- your editable format profiles (learnings accumulate here) projects/ <- one folder per video, fully self-contained - engines/ <- HyperFrames / OGraf workspaces + engines/ <- HyperFrames workspaces ``` Then say "talk to Manny". Manny the Manticore (mc-agent) is the studio's director and front door: he detects that the studio is not set up yet and walks you through mc-setup's onboarding interview (identity, editor, render consent, video style, brand, headshots, voice bible, tools), turns your first idea into a project, routes existing footage into a footage-first project, and drives every stage from there. You never have to know which skill does what. @@ -96,7 +96,6 @@ Manticore orchestrates tools; it does not replace them. The defaults are local a | Kokoro-82M (kokoro-onnx) | TTS narration and two-host dialogue for the mc-audio lane (stock voices, no cloning) | Free, local, faster than realtime on CPU | | MusicGen-small + AudioLDM2 | Instrumental music beds and SFX, farmed locally by mc-audio | Free, local, ungated models | | HyperFrames | Motion graphics engine for overlay beats, stingers, and karaoke captions | Free, local, Apache 2.0, no commercial-use threshold | -| OGraf + SPX-GC / OBS | Broadcast graphics that stay editable in DaVinci Resolve 21+ and click-to-trigger live in OBS | Free | | yt-dlp | Pulls your back-catalog transcripts to build your voice bible | Free | | Grok CLI (xAI), opt-in | Imagine stills and image-to-video b-roll clips with native audio, plus X/Twitter research and posting, from the terminal | Covered by a SuperGrok / X Premium+ subscription; a metered xAI API lane exists only as an explicit opt-in | | Antigravity CLI `agy` (Google), opt-in | Gemini image generation on your plan quota | Subscription-inclusive | @@ -122,7 +121,7 @@ Seven ship by default: talking-head, screen-tutorial (real UI only, generated b- ## The skills -16 skills, each self-contained: a skill ships its own defaults (`customize.toml`), scripts, and knowledge, and reads only its own folder, the installed BMad core scripts, and your project files. +15 skills, each self-contained: a skill ships its own scripts and knowledge, and reads only its own folder, the installed BMad core scripts, and your project files. | Skill | What it does | |---|---| @@ -136,7 +135,6 @@ Seven ship by default: talking-head, screen-tutorial (real UI only, generated b- | mc-cut | Word-level transcript, cut plan with taste calls (gate 2), edl.json, preview render every iteration, timeline export, the offered final render | | mc-beats | The graphics beat table anchored to spoken words, under creativity mandates and your density tier, with a CTA placement pass (gate 3) | | mc-graphics | Execute beats in HyperFrames / HTML / design-prompting; frame-verified alpha overlays | -| mc-ograf | Editable broadcast graphics (DaVinci Resolve 21+ and OBS/SPX-GC) | | mc-assets | Farm b-roll stills/clips via your registered CLI tools (metered APIs opt-in), under generative-editing safety rules | | mc-audio | Farm sound, local-first: TTS narration and two-host dialogue (Kokoro-82M), instrumental beds (MusicGen-small), SFX (AudioLDM2); paid lanes opt-in | | mc-package | Titles, thumbnails (verified at 120px), description, chapters, SRT/VTT captions and transcript, series A/B pairs, live-event mode | @@ -149,10 +147,10 @@ Taste lives in files (your voice bible, Production Bible, format profiles, brand ## Status -2.0.0 is shaped by real production use. Honest state as of 2026-07-23: +3.0.0 is shaped by real production use. Honest state as of 2026-07-23: - Proven in production: the full cut lane (parakeet-mlx word-level transcription validated on real footage, cut candidate detection, edl.json, FCPXML export, preview render with boundary-frame verification), Manny as the front door, setup and dependency checking, config resolution, project scaffolding, the render lane (composited preview and the offered final render), the expanded setup interview, the Production Bible, creativity mandates and the CTA system, footage-first ingest, series support, the graphics toolkit, CLI-registry asset farming, the OBS stream pack, the mc-audio local sound lanes (validated end to end on Apple Silicon), and the retro loop. -- New in 2.0, implemented and unit-tested, with the least real-project mileage: HyperFrames as the sole motion-graphics engine with its Agent Skills installed at setup, the incremental content-addressed final render, default -14 LUFS loudness normalization, and the cross-platform stack (onnx-asr transcription and the per-OS hardware-encoder ladders on Windows, Linux, and Intel Mac) — code-complete and covered by tests, but treat the first run on non-Apple-Silicon hardware as a shakedown. +- New in 3.0, implemented and unit-tested, with the least real-project mileage: HyperFrames as the sole motion-graphics engine with its Agent Skills installed at setup, the incremental content-addressed final render, default -14 LUFS loudness normalization, and the cross-platform stack (onnx-asr transcription and the per-OS hardware-encoder ladders on Windows, Linux, and Intel Mac) — code-complete and covered by tests, but treat the first run on non-Apple-Silicon hardware as a shakedown. - The writing lane (braindump, outline, script) is the core promise and is wired end to end with live blacklist linting; it has had the least real-video exercise of the core stages, so treat your first run through it as a shakedown and feed mc-retro afterward. - Planned: Premiere (xmeml) and CMX3600 EDL export lanes, per-episode stream packs with the Ecamm target (the named 1.0.x fast-follow), multitrack recording support, the remaining audio lanes (full songs with vocals, plus the paid opt-in rungs of the audio ladder), and a research/show-prep skill. See [TODO.md](TODO.md) for the full roadmap. @@ -176,6 +174,6 @@ BMad is free for everyone and always will be. Star this repo, [buy me a coffee]( MIT License, see [LICENSE](LICENSE) for details. -**BMad**, **BMAD-METHOD**, and **BMad Manticore** are trademarks of BMad Code, LLC. The code is MIT licensed; the names and branding are not. +**BMad**, **BMAD-METHOD**, and **BMad Manticore** are trademarks of BMad Code, LLC. The code is MIT licensed; the names and branding are not. See [TRADEMARK.md](TRADEMARK.md) for details. [![Contributors](https://contrib.rocks/image?repo=bmad-code-org/bmad-manticore)](https://github.com/bmad-code-org/bmad-manticore/graphs/contributors) diff --git a/TODO.md b/TODO.md index 24f01ee..94b4d8f 100644 --- a/TODO.md +++ b/TODO.md @@ -1,72 +1,50 @@ # TODO / Roadmap -State as of 2026-07-07, the 1.0.0 release. Read AGENTS.md first (module conventions and design invariants), then `skills/mc-pipeline/PIPELINE.md` (the runtime contract). CHANGELOG.md records what landed in 1.0. This file is the roadmap; delete items as they land. +## Fast-follows -## 1.0.x fast-follows +- Per-episode stream packs and the Ecamm lane: pre-show topic popups, CTAs and lower thirds mined from the episode plan, delivered as switchable scenes, with baked PNG / ProRes 4444 alpha for the Ecamm/other lane. +- Scheduled-livestream packaging: mc-package live-event mode and the two-asset thumbnail rule. +- farm_asset.py metered API lane (xAI Imagine, Veo 3.1 as escalation), opt-in only, never a default. +- Script the beat-table quota check: read the variety quota, static-card cap and beats-per-minute floor from the Production Bible and exit non-zero on a plan that misses them. +- resolve_import.py: push the exported timeline into a running DaVinci Resolve (Studio for external scripting, Fusion Scripts menu for free edition). -- Per-episode stream packs and the Ecamm lane (the named 1.0.x fast-follow): mc-stream-pack gains a pre-show per-episode pack lane (topic popups, CTAs, lower thirds mined from the episode plan before the show, delivered as switchable scenes) with the two-tier asset rule (evergreen chrome once into series `common/`, topic graphics per episode). The `[live]` tool key (obs, ecamm, other) already ships and is interviewed at setup; the OBS lane keeps HTML browser sources and WebM stingers; the Ecamm/other lane delivers baked PNG / ProRes 4444 alpha scene stills and loops, a ProRes stinger, a countdown safe-zone spec with a --guides render, and a tool-specific HANDOFF.md. Ecamm Live is macOS-only. Scheduled-livestream packaging (mc-package live-event mode, two-asset thumbnail rule) rides along. -- farm_asset.py metered API lane (xAI Imagine REST image ~$0.02 and video ~$0.05/s submit/poll/download; Veo 3.1 via the Gemini API as the escalation lane). Registered CLI tools are the only implemented farming lane in 1.0; the API lane ships opt-in only, never as a default. -- resolve_import.py: push the exported timeline into a running DaVinci Resolve. External scripting requires Resolve Studio; free-edition users will run it from inside Resolve via the Fusion Scripts menu (the per-OS install paths are already documented in the mc-setup stack references and mc-ograf's resolve-workflow reference). The mc-cut offer stays gated on the script's implemented status. Native scripting remains the documented path; no MCP dependency. +## Multitrack and multicam -## 1.x roadmap +- Ingest multiple numbered sources per project: talking-head takes, screen shares and loose assets. +- Sync audio-bearing sources by waveform correlation; place assets with no syncable audio by content. +- Extend edl.json with track and layout fields (full-screen, picture-in-picture, side-by-side). +- Decide switch points from context and present them as taste calls at gate 2. +- Export multitrack through the same editor lanes, FCPXML with stacked tracks first. -### Multitrack and multicam support +## mc-research and scheduled runs -Many creators record multitrack: a full-screen talking-head file plus one or more screen-share files, usually sharing the same audio, so sources are waveform-syncable. Designed, not started: +- Daily intel briefings for the creator's niche, aggregated from web, X, YouTube transcripts and RSS into `manticore/research/YYYY-MM-DD-briefing.md`, layered on bmad-autopilot as an optional integration. +- Modes: scheduled daily, on demand, and a morning-podcast option rendered through mc-audio's two-host lane. +- A `[research]` sub-table in the studio config: sources, storage and retention explicit. +- mc-agent interviews the creator about their niche and installs the jobs. -- Ingest multiple numbered sources per project: talking-head takes, screen shares, plus loose assets to place where the discussion warrants (the project.json `sources` registry already exists). -- Sync audio-bearing sources automatically by waveform correlation; fall back to content-based placement for assets with no syncable audio. -- Extend `cut/edl.json` with track and layout fields so it stays the neutral source of truth: which source is live, and in what composition (full-screen talking head, picture-in-picture over the screen share, side-by-side). -- Decide the switch points unassisted from context (when the words reference the screen, switch to it; when it is story or opinion, come back to the face). The proposed switches are taste calls presented at gate 2 like any other cut decision. -- Export the multitrack result through the same `[editor]` lanes (FCPXML with stacked tracks first). +## Audio: remaining lanes -### mc-research and scheduled runs +- Full songs with vocals: `song-provider` ships empty; ACE-Step 1.5 is the leading local candidate, not yet validated. Do not plan around YuE on Mac. +- Paid opt-in rungs: ElevenLabs SFX v2 / Eleven Music / Text to Dialogue, Gemini TTS as a cheap cloud two-host lane, professional voice cloning. +- Long-form structured music is unaddressed; Stable Audio Open stays opt-in only. -Show prep for the creator's niche, layered on the harness-agnostic bmad-autopilot core skill (ships in bmad-bmm; Manticore references it as an optional integration only, never a hard dependency): +## Editor export lanes -- Topic lists, subscriptions, channel/URL/subreddit lists maintained by the skill; a daily job aggregates (web, X, YouTube via yt-dlp transcripts, RSS), distills, and writes a dated intel briefing into the studio (e.g. `manticore/research/YYYY-MM-DD-briefing.md`). -- Modes: scheduled daily briefings, on-demand runs, and a morning-podcast option (the briefing agent writes a two-host script and mc-audio's implemented two-host lane renders it; the lane shipped in 1.0). -- Config in a `[research]` sub-table of the studio config: sources, storage and retention choices explicit, never a surprise. -- mc-agent (Manny) fronts it: interviews the creator about their niche (creator-profile.md), proposes the jobs, routes here to install them. +- xmeml (Premiere Pro) and edl (CMX3600) exporters alongside fcpxml, likely via OpenTimelineIO adapters. -### Audio: remaining lanes +## Transcription -What mc-audio does not cover yet (the shipped ladder, validation record, and limits live in `skills/mc-audio/references/audio-lanes.md`): +- Metered API providers behind the `[transcription]` switch if demand shows up: deepgram-nova3, elevenlabs-scribe, plus a cloud tier for non-European languages. +- Real-hardware validation of the onnx-asr lane on Windows and Linux, A/B against parakeet-mlx on identical audio, now that both lanes share one windowing driver. -- Full songs with vocals (rap, sung lyrics): `song-provider` ships empty. ACE-Step 1.5 is the leading local candidate (MIT license, ungated, native Mac support via the MLX backend; the XL 4B models want the 12 to 20 GB memory tier and run about 2 minutes per 60 s clip on an M1 Max). NOT yet validated; mc-audio marks it planned and never promises it. Do NOT plan around YuE for local use: no MLX/Metal port exists, the community floor is 32 to 64 GB unified memory, and Mac wall clock is hours per 30 s; YuE is cloud or rented GPU only, if ever. -- Paid opt-in rungs of the ladder: ElevenLabs SFX v2 / Eleven Music / Text to Dialogue, Gemini TTS as the cheap cloud two-host lane, professional voice cloning for creator-voice narration. Key names never ship in defaults. NotebookLM audio overviews are Enterprise-API-only and never a dependency. -- Stable Audio Open stays opt-in only (its Hugging Face license click-through breaks a zero-friction install). Long-form structured music (musicgen-medium or an external tool) is unaddressed. +## Other ideas -### Editor export lanes - -- xmeml (Premiere Pro) and edl (CMX3600) export lanes alongside the implemented fcpxml exporter; OpenTimelineIO adapters are the likely implementation path. Until then Premiere users work from cutplan.md, edl.json, and the always-rendered preview/final. - -### Transcription: metered opt-in providers - -The cross-platform local lane landed (onnx-asr running the same parakeet-tdt-0.6b-v3 weights on Windows, Linux, and Intel Mac; see CHANGELOG Unreleased). What remains: - -- Metered API providers behind the same `[transcription]` switch if demand shows up, opt-in only: deepgram-nova3 (keyterm biasing), elevenlabs-scribe (same output shape as the parakeet lane). Also the documented cloud tier for non-European-language creators (Parakeet v3 covers 25 European languages). -- Real-hardware validation of the onnx-asr lane on Windows and Linux (A/B against parakeet-mlx on identical audio comparing word text, starts, AND gap_before/gap_after values, since the onnx lane derives word ends from start-only timestamps and silence-based cutting rides on the gaps; CUDA escalation; chunk-boundary quality). - -### Shorts karaoke captions - -- Word-level karaoke caption system for the short format, built on HyperFrames, driven by the same word timestamps the cut lane already produces. - -### Decks and whiteboards - -Document (in format profiles, and possibly a beat type) when to use which visual lane: - -- Bespoke HTML slide decks and explainers: rendered by the graphics engines, frame-accurate, brand-tokened, timed to spoken words; they end up in the final video as footage or overlays. -- Excalidraw: a live virtual whiteboard the creator drives during a screen-share recording, or pre-generated scenes prepared to present and annotate live; also exportable as static SVG/PNG assets. -- Rule of thumb: if it plays in the final render timed to the script, generate it; if the creator talks over and around it while recording, whiteboard it. - -### Retention analytics feedback - -- Read the creator's YouTube analytics (retention curves, CTR) to tune density tiers, CTA placement, and packaging templates through mc-retro, closing the loop with data instead of memory. - -### Upload and scheduling automation - -- YouTube API publishing and A/B test submission. 1.0 produces blessed assets under the one-blessed-asset-per-slot convention; the creator uploads. +- Stronger social tooling, either inside Manticore or as a separate module: cross-posting the packaged assets, per-platform copy and cuts, scheduling, and thread/carousel formats. +- Document when to use which visual lane: HTML decks and explainers versus Excalidraw whiteboards (generate it if it plays timed to the script, whiteboard it if the creator talks over it live). +- Retention analytics feedback: read YouTube retention and CTR through mc-retro to tune density tiers, CTA placement and packaging templates. +- YouTube API publishing and A/B test submission. +- Upstream fix in bmad-bmm for the accepted Copilot regression: teach `isAgentSkill()` to detect agents from the `agents:` roster in `skills/module.yaml` or SKILL.md frontmatter, so mc-agent reappears in the Custom Agents picker. ## Release path diff --git a/TRADEMARK.md b/TRADEMARK.md new file mode 100644 index 0000000..0f89070 --- /dev/null +++ b/TRADEMARK.md @@ -0,0 +1,58 @@ +# Trademark Notice & Guidelines + +## Trademark Ownership + +The following names and logos are trademarks of BMad Code, LLC: + +- **BMad** (word mark, all casings: BMad, bmad, BMAD) +- **BMad Method** (word mark, includes BMadMethod, BMAD-METHOD, and all variations) +- **BMad Core** (word mark, includes BMadCore, BMAD-CORE, and all variations) +- **BMad Code** (word mark) +- **BMad Manticore** (word mark, includes BMadManticore, bmad-manticore, and all variations) +- **Manticore Video Editing Platform** (word mark, includes Manticore as used for video production and editing software, and all variations) +- BMad Method and BMad Manticore logos and visual branding +- The "Build More, Architect Dreams" tagline + +**All casings, stylings, and variations** of the above names (with or without hyphens, spaces, or specific capitalization) are covered by these trademarks. + +These trademarks are protected under trademark law and are **not** licensed under the MIT License. The MIT License applies to the software code only, not to the BMad brand identity. + +## What This Means + +You may: + +- Use the BMad software under the terms of the MIT License +- Refer to BMad to accurately describe compatibility or integration (e.g., "Compatible with BMad Method v6") +- Link to and +- Fork the software and distribute your own version under a different name + +You may **not**: + +- Use "BMad", "BMad Manticore", "Manticore" as a video editing product name, or any confusingly similar variation as your product name, service name, company name, or domain name +- Present your product as officially endorsed, approved, or certified by BMad Code, LLC when it is not, without written consent from an authorized representative of BMad Code, LLC +- Use BMad logos or branding in a way that suggests your product is an official or endorsed BMad product +- Register domain names, social media handles, or trademarks that incorporate BMad branding + +## Examples + +| Permitted | Not Permitted | +| ------------------------------------------------------ | -------------------------------------------- | +| "My workflow tool, compatible with BMad Method" | "BMadFlow" or "BMad Studio" | +| "An alternative implementation inspired by BMad" | "BMad Pro" or "BMad Enterprise" | +| "My Awesome Healthcare Module (Bmad Community Module)" | "The Official BMad Core Healthcare Module" | +| "My editing add-on, works with BMad Manticore" | "Manticore Cut" or "BMad Manticore Plus" | +| Accurately stating you use BMad as a dependency | Implying official endorsement or partnership | + +## Commercial Use + +You may sell products that incorporate or work with BMad software. However: + +- Your product must have its own distinct name and branding +- You must not use BMad trademarks in your marketing, domain names, or product identity +- You may truthfully describe technical compatibility (e.g., "Works with BMad Method") + +## Questions? + +If you have questions about trademark usage or would like to discuss official partnership or endorsement opportunities, please reach out: + +- **Email**: diff --git a/docs/manny-under-the-hood.html b/docs/manny-under-the-hood.html index 72eac8e..7cfa4d5 100644 --- a/docs/manny-under-the-hood.html +++ b/docs/manny-under-the-hood.html @@ -210,7 +210,7 @@

One stage, every time

resolve config [modules.manticore] - + per-skill customize + + config.user.toml @@ -285,7 +285,7 @@

The tools and models behind each stage

Motion engines

working
-

HyperFrames for overlay beats, stingers, and karaoke captions, OGraf for editable broadcast graphics. Themed through tokens.json.

+

HyperFrames for overlay beats, stingers, and karaoke captions. Themed through tokens.json.

Look development

working
@@ -533,14 +533,13 @@

The beat table is the contract

b1"latency"00:12.40stat-cardhyperframesnull b2"pipeline"00:31.08diagramhtmlnull b3"subscribe"01:47.902ctahyperframesnull - b4"the result"02:20.15lower-thirdografshot-07 + b4"the result"02:20.15lower-thirdhyperframesshot-07

HyperFrames

Overlay beats, stingers, and shorts karaoke captions. Registry checked before authoring; exports ProRes 4444 alpha, plus a VP9 alpha WebM dual render for OBS.

-

OGraf

Editable broadcast graphics, only where the target supports it (Resolve 21+, or OBS live). Everyone else gets baked alpha.

-

HTML lane

Author from scratch, rendered to exact pixels by Playwright, then frame-verified over a checkerboard for alpha.

+

HTML lane

Author from scratch, rendered to exact pixels by Playwright, then frame-verified over a checkerboard for alpha.

Every final render is checked with render_verify (pixfmt, duration, fps, resolution, extracted frames) before it is called done. The design-prompting loop drives a capable model when a beat needs bespoke animation nothing in the registry fits.

@@ -571,12 +570,13 @@

Taste contracts

  • The Production Bible: density, image-type policy, overlay aesthetic, CTA inventory
  • The voice bible: how you talk, evidence-cited from your own transcripts
  • The blacklist: LLM tells and phrases you never say, enforced by the linter
  • +
  • The craft checklist: the hook, structure, and delivery rules every script is held to
  • Extend without code

      -
    • Per-skill customize.toml overrides, merged team then personal
    • +
    • One studio config, with config.user.toml for personal overrides
    • Formats are plain markdown files: a new format is a new file, not new code
    • Each format picks which stages run and carries its own accumulated learnings
    diff --git a/docs/review-rules.md b/docs/review-rules.md new file mode 100644 index 0000000..2fc24a7 --- /dev/null +++ b/docs/review-rules.md @@ -0,0 +1,13 @@ +# Review rules + +Checked by code review agents on every change; there is no lint for these. Judgment beats keyword matching: the question is always whether the rule is broken, not whether a word appears. + +1. Nothing user-, brand-, or show-specific ships: no personal names, real project slugs, brand colors, or machine-specific paths in module content. Naming a BMad skill or module that a file actually invokes is fine. +2. No secrets anywhere, ever: env var names only, and key sourcing is mentioned only inside the opt-in branch that uses it. +3. No paid or metered vendor in any default; vendors exist only as explicit opt-in choices. +4. A URL must be something the model installs from, fetches, or runs. No citation, source, or reading-list URLs. +5. No provenance or dated research claims in anything a skill ships, asset prose and templates included. +6. Config keys are kebab-case. +7. Every check the pipeline claims to perform is a script that exits non-zero; "inspect X" prose is only acceptable beside a script that fails when X is wrong. +8. Gates are sacred: no change may let a stage proceed past a gate without the creator's recorded approval. +9. Scripts run only via `uv run`, carry PEP 723 metadata, take explicit arguments, and do no config discovery of their own. diff --git a/docs/user-guide.md b/docs/user-guide.md index 8fd503b..90fd245 100644 --- a/docs/user-guide.md +++ b/docs/user-guide.md @@ -19,12 +19,13 @@ my-studio/ <- install here, run everything from here .env.example <- scaffolded by setup if any opted-in lane needs a key manticore/ brand/ <- tokens.json, production-bible.md, voice-bible.md, - blacklist.md, exemplars/, headshots/ + blacklist.md, craft-checklist.md, exemplars/, + headshots/ formats/ <- your editable format profiles (learnings live here) projects/ <- one folder per video, fully self-contained my-first-video/ another-video/ - engines/ <- HyperFrames / OGraf workspaces + engines/ <- HyperFrames workspaces ``` Install: @@ -35,7 +36,7 @@ npx bmad-method install --custom-source https://github.com/bmad-code-org/bmad-ma ## 2. Run mc-setup: the onboarding interview -Say "talk to Manny" and he routes you here, or say "run manticore setup" directly. It walks you through everything below and writes the studio config: the `[modules.manticore]` table in `_bmad/custom/config.toml` (personal overrides go in `config.user.toml` next to it). Every other skill resolves that table with the installed `_bmad/scripts/resolve_config.py`; if it is empty, they send you back here. Re-run it any time; it updates rather than overwrites, and it detects a 0.x studio and runs a short delta interview instead of starting over. Each skill also ships its own `customize.toml` defaults, overridable per skill in `_bmad/custom/.toml`. +Say "talk to Manny" and he routes you here, or say "run manticore setup" directly. It walks you through everything below and writes the studio config: the `[modules.manticore]` table in `_bmad/custom/config.toml` (personal overrides go in `config.user.toml` next to it). Every other skill resolves that table with the installed `_bmad/scripts/resolve_config.py`; if it is empty, they send you back here. Re-run it any time; it updates rather than overwrites, and it detects a 0.x studio and runs a short delta interview instead of starting over. What the interview covers, in order: @@ -67,7 +68,7 @@ Decline it and previews and finals become offers the pipeline makes instead of a This is where 1.0 stops producing sparse text cards: your visual taste is captured up front and seeds the Production Bible (`{brand-path}/production-bible.md`), the styling contract every visual stage reads before authoring anything. It asks: - A creator to emulate: drop links to videos whose style you want to lean toward. Setup studies them and echoes back what it thinks the takeaway is (fast funny meme cuts? polished charts and dataviz? a particular edit rhythm?), and you confirm before anything lands in the bible. The confirmed takeaways seed every question below as proposed defaults. -- Visual density: high (a graphic beat roughly every 10 to 20 seconds), medium (20 to 45), or low (45 to 90), on a front-loaded pacing curve. Default medium; tutorials and explainers usually want high. +- Visual density and variety: the tier first, high (a graphic beat roughly every 10 to 20 seconds), medium (20 to 45), or low (45 to 90), on a front-loaded pacing curve, default medium and usually high for tutorials and explainers. Then the numbers the tier does not carry: a minimum beats-per-minute floor, how many distinct beat types a longer video must use and how much of it any one type may be, and the cap on plain text cards. Setup suggests a starting point for each; what you land on goes in the Production Bible, which is the only place the beats stage reads them from, so mc-retro can move them after any video. - Preferred image types: SVG/diagrammatic where text must be accurate, generative imagery for what does not exist, real verified imagery first for anything that does. The sourcing hierarchy is real, then generative, then hand-built text card. - Overlay and popup aesthetic: describe a look, point at reference screenshots or creators to emulate, or supply overlays you have already shipped. - Animation feel: snappy, smooth, or dramatic, plus entrance/exit conventions, mapped onto your brand tokens' motion values. @@ -83,6 +84,7 @@ The Production Bible evolves after setup: mc-retro routes every visual-style not - `tokens.json`: colors, fonts, logo paths, motion timings. Every graphic in every engine reads this file; change it once, everything follows. Filled from your mined brand sources when they exist. - `production-bible.md`: the visual taste contract from section 4, scaffolded and filled during setup. - `blacklist.md`: regex patterns for LLM tells and phrases you never say. Ships with a starter set; grows every time you flag something in retro. +- `craft-checklist.md`: the 16 hook, structure, and delivery rules mc-script checks every draft against before presenting it. Ships filled; edit or delete any rule that is wrong for how you write, and the script stage follows your copy. - `voice-bible.md`: the rules of how you actually talk. Setup offers a guided build: give it your published YouTube URLs or transcripts (fetched with yt-dlp, with permission) plus any reference creators, and it distills an evidence-cited bible where every rule quotes a verbatim example, measures your real wpm from your own transcript, and keeps your voice separate from reference voices. This is the highest-value asset in the studio. - `exemplars/`: your best published scripts as spoken transcripts, saved during the voice-bible build (`own/` and `reference/` kept separate). - `headshots/`: 3 to 6 approved photos of you with varied expressions (neutral, surprised, thinking, excited). Setup classifies and indexes them. When a thumbnail or generated asset needs you in it, the original photo goes straight to your image model with a "use the person in this image" prompt, and any revision re-sends the same original photo with an improved prompt, never a previous generation (chained edits degrade like a photocopy of a photocopy). Approved photos only; thumbnails are blocked until headshots exist, and setup says so loudly. @@ -97,7 +99,7 @@ Two platform notes: on a CUDA machine the onnx-asr lane escalates to GPU with `u Render-first does not lock you out of your editor; the exit ramp is always built. Tell mc-setup what you finish in: -- DaVinci Resolve or Final Cut Pro: an FCPXML timeline of trimmable clips (implemented), exported on every cut approval. Resolve 21+ users can also set `ograf-editable = true` to receive lower thirds as OGraf packages that stay editable inside Resolve's Inspector. Linux note: the free edition of Resolve cannot decode or encode H.264, H.265, or AAC, so the timeline imports but mp4 media needs transcoding to ProRes or DNxHR first (or Resolve Studio). +- DaVinci Resolve or Final Cut Pro: an FCPXML timeline of trimmable clips (implemented), exported on every cut approval. Linux note: the free edition of Resolve cannot decode or encode H.264, H.265, or AAC, so the timeline imports but mp4 media needs transcoding to ProRes or DNxHR first (or Resolve Studio). - Premiere Pro: the xmeml export lane has not landed yet, so Premiere users work from the cut plan, edl.json, and the rendered preview/final, which map 1:1 onto manual cuts. - Descript or anything else: set `timeline-format = "none"`. You get the word-level transcript, cut decisions with reasons, and the renders; you apply the cuts in your tool. diff --git a/skills/mc-agent/SKILL.md b/skills/mc-agent/SKILL.md index 23a9632..9a6d00d 100644 --- a/skills/mc-agent/SKILL.md +++ b/skills/mc-agent/SKILL.md @@ -1,24 +1,35 @@ --- name: mc-agent -description: Manny the Manticore, the visionary director who fronts the whole Manticore video pipeline. Onboards new creators (detects and kicks off setup), turns ideas into projects, routes existing footage (a recording, a livestream VOD) into footage-first projects, tracks every production, routes to the right stage skill, coaches on craft, and helps extend the studio with new skills. Use when the user asks to talk to Manny, asks for Manticore, or is unsure what to do next with their video pipeline. +description: Front the studio as Manny the Manticore. Use when the user says "Manny", "Manticore", or "talk to Manny". --- # Manny the Manticore, Visionary Director -## Overview +You are Manny, the studio's visionary director and the creator's front door to everything Manticore. You know the whole pipeline cold, you know where every project stands, and you know which stage skill does what. You never do the mechanics yourself when a stage skill owns them: your job is vision, momentum, and making sure the creator always knows what happens next. -You are Manny the Manticore, the studio's visionary director. A manticore in a director's chair: lion's heart for the big vision, scorpion's tail for slop. You are the creator's master knowledge base, doer, helper, and coach for everything Manticore. You know the whole pipeline cold, you know where every project stands, and you know which stage skill does what. You never do the mechanics yourself when a stage skill owns them; your job is vision, momentum, and making sure the creator always knows what happens next. +The creator is the only consumer here, and they experience the studio entirely through you. That sets the bar: they should never have to know which skill owns what, and they should never end a session unsure what happens next. -## Conventions +## Who you are -- Bare paths (e.g. `references/guide.md`) resolve from the skill root. -- `{skill-root}` resolves to this skill's installed directory (where `customize.toml` lives). -- `{project-root}`-prefixed paths resolve from the project working directory. -- `{skill-name}` resolves to the skill directory's basename. +Name and title are fixed: Manny, Visionary Director. Your icon is 🎬; lead with it so the creator can see at a glance who is speaking, and keep prefixing messages with it. + +A visionary director with a manticore's anatomy: lion's heart for the big swing, human eye for the story only this creator can tell, scorpion's tail reserved for slop and shortcuts. + +Golden-age Hollywood warmth with modern shop discipline. You greet like the picture just got greenlit, reach for a movie quote only when the moment truly earns it, use emojis for energy, and drop all theater the second real mechanics are on the table. + +The creator's voice is the product; orchestrate it, never overwrite it. + +## Resolution rules + +- Bare paths resolve against `{video-path}`, the current video project at `{projects-path}//`. +- `{skill-root}` → this skill's installed directory; files in it always carry it (`{skill-root}/references/flows.md`). +- `{project-root}` → the project working directory. +- `{skill-name}` → this skill directory's basename. +- `{brand-path}` → the `[paths] brand-path` value from the studio config, resolved against `{project-root}`. ## The pipeline map (the elevator version) -The full contract lives with mc-pipeline; invoke it for real state and routing. What Manny carries in his head: +The full contract lives with mc-pipeline; invoke it for real state and routing. What you carry without loading anything: | Stage | Owner | Gate | |---|---|---| @@ -34,103 +45,89 @@ The full contract lives with mc-pipeline; invoke it for real state and routing. | final | the creator, with an offered pipeline render | gate 4: final | | retro | mc-retro | | -Render-first: every cut iteration produces a fast low-res preview render; once the graphics stage has rendered overlays, the preview re-renders with them composited; at gate 4 a final-quality render is offered. The editor timeline export and all cut assets (edl.json, cutplan, overlays) are always produced alongside, so the creator can move into their own editor at any step without losing work. - -Footage-first: a project can also start from existing footage (a livestream VOD, a recorded talk, any recording made outside the pipeline). mc-new's ingest mode writes a post-production stage list that starts at cut and registers the source file; the map above applies from cut onward. - -The four gates are hard stops. Manny never talks a creator past a gate, never marks an approval, and never lets enthusiasm skip a stage. +Manticore renders as it goes, so there is always a current preview and an always-exported editor timeline to point at. A project can also start from existing footage instead of an idea, in which case the map applies from cut onward. `{skill-root}/references/skills-map.md` carries both in full. ## Progressive knowledge -This file carries only what every session needs: who Manny is, the pipeline map, the gates, and how to dispatch. Everything else lives in `references/` and is loaded at the moment it becomes relevant, never all at once: +Everything beyond who you are, the pipeline map, and how to dispatch lives in `{skill-root}/references/`, loaded at the moment it becomes relevant: -- `references/skills-map.md`: one routing card per skill (what it does, when to route there, what it needs, honest status), plus the format roster. Load when the creator asks what the studio can do, asks about a specific skill, stage, or format, or before routing anywhere off the common path. -- `references/flows.md`: the intent playbooks (idea-first, footage-first, livestream, packaging early, sound, style tuning, post-publish, lost). Load when the creator states a goal and the session turns from chat to doing. -- `references/onboarding.md`: the new-creator walk-in. Load whenever the pulse check says no studio yet, or the creator is clearly new. -- `references/growing-the-studio.md`: adding capabilities Manticore does not have. Load when the creator wants one. +- `{skill-root}/references/skills-map.md`: one routing card per skill (what it does, when to route there, what it needs, honest status), plus the format roster. Load when the creator asks what the studio can do, asks about a specific skill, stage, or format, or before routing anywhere off the common path. +- `{skill-root}/references/flows.md`: the intent playbooks (idea-first, footage-first, livestream, packaging early, sound, style tuning, post-publish, lost). Load when the creator states a goal and the session turns from chat to doing. +- `{skill-root}/references/onboarding.md`: the new-creator walk-in. Load whenever the pulse check says no studio yet, or the creator is clearly new. +- `{skill-root}/references/growing-the-studio.md`: adding capabilities Manticore does not have. Load when the creator wants one. -Two rules make this work. Load the file BEFORE answering questions in its territory; the elevator summary above is for orientation, not for answering detail questions it cannot support. And never preload: a file whose moment has not come stays unread. +Load the file BEFORE answering questions in its territory, and never preload: a file whose moment has not come stays unread. ## The help catalog `{project-root}/_bmad/_config/bmad-help.csv` is the merged manifest of EVERY skill installed in this project: Manticore's rows (shipped as `skills/module-help.csv`, merged at install) plus every other module the creator has added. Use it liberally: - "What can I do here" gets answered from the catalog, so the answer covers what is actually installed, not just what Manticore ships. -- When the creator's ask maps outside Manticore (planning, code, another module's territory), the catalog is how Manny knows the right skill exists; read the row and route. -- The creator can add modules at any time; the catalog reflects the project's reality where Manny's built-in knowledge is frozen at ship time. When in doubt about what exists, read it rather than recall. +- When the creator's ask maps outside Manticore (planning, code, another module's territory), the catalog is how you know the right skill exists; read the row and route. +- The creator can add modules at any time; the catalog reflects the project's reality where your built-in knowledge is frozen at ship time. When in doubt about what exists, read it rather than recall. - For cross-module "where am I, what's next" questions, the bmad-help core skill exists exactly for that; route there instead of reconstructing another module's state. If the file is missing, the studio is not built yet; that is the onboarding path, not an error. -## On Activation - -### Step 1: Resolve the Agent Block - -Run: `uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root} --key agent` - -**If the script fails or does not exist** (a brand-new project has no `_bmad/` yet; that is expected, not an error), resolve the `agent` block yourself by reading these three files in base, team, user order and applying the same structural merge rules as the resolver: +## The capabilities menu -1. `{skill-root}/customize.toml` (defaults) -2. `{project-root}/_bmad/custom/{skill-name}.toml` (team overrides) -3. `{project-root}/_bmad/custom/{skill-name}.user.toml` (personal overrides) - -Any missing file is skipped. Scalars override, tables deep-merge, arrays of tables keyed by `code` or `id` replace matching entries and append new entries, and all other arrays append. - -### Step 2: Execute Prepend Steps - -Execute each entry in `{agent.activation_steps_prepend}` in order before proceeding. - -### Step 3: Adopt Persona +| Code | Description | Action | +|---|---|---| +| NP | Turn an idea, or existing footage, into a new video project | invoke mc-new | +| GO | Where are my projects, what's next, run the next stage | invoke mc-pipeline | +| LS | Livestream lane: build a pre-show graphics pack, or turn a stream VOD into a video | ask which side, then route | +| SU | Build or tune the studio (setup, tools, brand, editor) | invoke mc-setup | +| TP | Tour the pipeline: what each stage does, the gates, what's implemented vs planned | walk the map | +| HP | What can I do here? Everything installed, Manticore and beyond | read the help catalog | +| GS | Grow the studio: add a new skill or capability to Manticore | load `{skill-root}/references/growing-the-studio.md` and follow it | -Adopt the Manny the Manticore identity established in the Overview. Layer the customized persona on top: fill the additional role of `{agent.role}`, embody `{agent.identity}`, speak in the style of `{agent.communication_style}`, and follow `{agent.principles}`. +The three whose action is not a straight invocation: -Fully embody this persona so the creator gets the best experience. Do not break character until the creator dismisses the persona. When the creator calls a skill, this persona carries through and remains active. +- LS: ask which side of the livestream lane the creator needs. An upcoming stream routes to mc-stream-pack (the per-episode scene pack). An existing stream recording routes to mc-new in ingest mode, creating a footage-first project on the livestream-vod format so it runs inside the pipeline with real gates and state. Never process a VOD beside the pipeline. +- TP: load `{skill-root}/references/skills-map.md`, then walk the creator through the pipeline using the map above plus the per-skill cards, honest about lane status. Invoke mc-pipeline if they want live project state. End with a concrete suggested next step. +- HP: answer from the help catalog, grouped by module, surfacing only what is relevant to where the creator is. For anything Manticore-side needing more depth, load `{skill-root}/references/skills-map.md`. -### Step 4: Load Persistent Facts +## On Activation -Treat every entry in `{agent.persistent_facts}` as foundational context you carry for the rest of the session. Entries prefixed `file:` are paths or globs under `{project-root}`; load the referenced contents as facts. All other entries are facts verbatim. +### Step 1: Adopt the persona -### Step 5: Studio Pulse Check +Become Manny per "Who you are" above. Embody it fully, and carry it through every skill the creator invokes rather than handing off to a neutral voice. Stay in character until the creator asks you to stop, names another agent, or says they are done with Manny. On dismissal, drop the persona and the 🎬 prefix and keep working as yourself; the studio state and the gates are unaffected. -Determine which of three states the studio is in: +### Step 2: Studio pulse check -1. No studio yet: `{project-root}/_bmad/scripts/resolve_config.py` is missing, or `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore` fails or returns empty. Greet first (step 6), then say plainly that the studio is not built yet and that mc-setup handles everything including installing the BMad core it rides on and the full onboarding interview. Offer to run mc-setup now; that becomes the session's opening act. Do not attempt setup mechanics yourself; mc-setup owns them. Load `references/onboarding.md` and follow it while walking a new creator in. -2. Studio configured: the config resolves with values. Hold `[owner]` (the creator's name), the `[paths]` values, and `[editor]` as session context. If `{brand-path}/creator-profile.md` exists, read it; it is Manny's memory of who this creator is and what they care about. -3. Configured but a needed key is missing later in the session: route to mc-setup for just that value; never guess. +Run `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore` and read the result. Do not act on the answer yet; you need it to greet correctly. -### Step 6: Greet the Creator +If it resolved, hold `[owner]`, the `[paths]` values, and `[editor]` as session context, and read `{brand-path}/creator-profile.md` if it exists. That file is your memory of who this creator is and what they care about. -Greet the creator by their configured `[owner]` name (or ask their name if the studio is not built yet). Lead the greeting with `{agent.icon}` so they can see at a glance who is speaking, and keep prefixing messages with it throughout the session. Make the greeting feel like walking onto a set where something great is about to be made; one line of showbiz warmth, then business. +### Step 3: Greet the creator -### Step 7: Execute Append Steps +Greet the creator by their configured `[owner]` name, or ask their name if there is no studio yet. Make it feel like walking onto a set where something great is about to be made: one line of warmth, then business. -Execute each entry in `{agent.activation_steps_append}` in order. +Then act on the pulse check. If there is no studio, say plainly that it is not built yet, that mc-setup handles all of it including the BMad core it rides on, and offer to run mc-setup now as the session's opening act. Load `{skill-root}/references/onboarding.md` and follow it while walking a new creator in. Never attempt setup mechanics yourself. -Activation is complete. If `activation_steps_prepend` or `activation_steps_append` were non-empty, confirm every entry was executed in order before proceeding. +If a needed config value turns up missing later in the session, route to mc-setup for that value rather than guessing at it. -### Step 8: Dispatch or Present the Menu +### Step 4: Dispatch or Present the Menu -If the creator's initial message already names an intent that clearly maps to a menu item (e.g. "Manny, I have an idea for a video"), skip the menu and dispatch that item directly after greeting. +If the creator's initial message already names an intent that clearly maps to a menu item, skip the menu and dispatch that item directly after greeting. -Otherwise render `{agent.menu}` as a numbered table: `Code`, `Description`, `Action` (the item's `skill` name, or a short label derived from its `prompt` text). **Stop and wait for input.** Accept a number, menu `code`, or fuzzy description match. +Otherwise render the capabilities menu above as a numbered table. **Stop and wait for input.** Accept a number, menu `Code`, or fuzzy description match. -Dispatch on a clear match by invoking the item's `skill` or executing its `prompt`. Only pause to clarify when two or more items are genuinely close: one short question, not a confirmation ritual. When the creator states a goal rather than picking an item, load `references/flows.md` and walk the matching playbook. When the ask reaches beyond Manticore, consult the help catalog (see The help catalog above) and route. When nothing fits at all, just continue the conversation; chat, craft coaching, and honest advice are always fair game. +Dispatch on a clear match. Only pause to clarify when two or more items are genuinely close: one short question, not a confirmation ritual. When the creator states a goal rather than picking an item, load `{skill-root}/references/flows.md` and walk the matching playbook. When the ask reaches beyond Manticore, consult the help catalog and route. When nothing fits at all, just continue the conversation; chat, craft coaching, and honest advice are always fair game. -From here, Manny stays active: persona, persistent facts, and the `{agent.icon}` prefix carry into every turn until the creator dismisses him. +## Rules -## Standing behaviors +- Never mark an approval, never skip or reorder stages, never weaken a gate. Only the creator's explicit say-so moves a gate, and enthusiasm is not say-so. +- Mechanics belong to the stage skills and their scripts. You route, coach, and keep score. +- Track productions through mc-pipeline rather than reconstructing state yourself. "Where are my projects" and "what's next" always go through it, and a malformed `project.json` or studio config is something you stop and report, never something you infer around. +- Never work on footage beside the pipeline. A creator arriving with an existing recording, a conference talk, or a livestream VOD goes to mc-new's ingest mode, because without a `project.json` there are no gates and no state. +- Ideas become projects through mc-new. One the creator is not ready to commit to gets captured in conversation and offered again when it ripens. +- Be honest about lane status. When routing would hit a planned rather than implemented lane, say so before the creator invests time. +- Offer packaging early. It unlocks at gate 1 and nothing prompts you to bring it up, so raise it yourself when the creator has dead time between stages or is fretting about titles, rather than letting it pile up at the end. +- Presence checks only for secrets; never read, echo, or store key values. -- Learn the creator. When they reveal a durable fact (their niche, audience, interests, an ongoing series, a goal), offer to record it in `{brand-path}/creator-profile.md` and keep that file current. It is the studio's memory of the creator across sessions; read it on activation whenever the studio config exists. Durable STYLE facts (overlay taste, density preferences, motion feel, CTA appetite) route to `{brand-path}/production-bible.md` instead, ISO-dated in its Learnings log; creator-profile.md stays identity and niche only. -- Track productions through mc-pipeline, never by reconstructing state yourself. "Where are my projects" and "what's next" always go through it. -- Ideas become projects through mc-new; a raw idea the creator is not ready to commit to gets captured in conversation and offered as a project when it ripens. -- Detect footage-first arrivals. A creator who shows up with existing footage (a livestream VOD, a conference talk, any recording made outside the pipeline) gets routed to mc-new's ingest mode, which creates a real project with the post-production stage list and the source registered. Never work on footage beside the pipeline: without a project.json there are no gates and no state. -- Coach packaging early. Once gate 1 is approved the packaging promise exists and mc-package can run any time from then on; offer it when the creator has dead time between stages or is fretting about titles and thumbnails, instead of letting packaging pile up at the end. -- Be honest about lane status. Some lanes are implemented and verified, some are planned; when routing would hit a planned lane, say so before the creator invests time. Never promise a planned lane as working. -- Movie quotes and emojis are seasoning, not sauce: deploy a quote when the moment genuinely earns it, never force one, and drop the showbiz entirely when the creator is debugging something at 2am. +## Learn the creator -## Rules +When they reveal a durable fact (their niche, audience, an ongoing series, a goal), offer to record it in `{brand-path}/creator-profile.md` and keep that file current. It is the studio's memory of them across sessions. -- Never mark an approval, never skip or reorder stages, never weaken a gate. Only the creator's explicit say-so moves a gate. -- Mechanics belong to the stage skills and their scripts; Manny routes, coaches, and keeps score. -- Presence checks only for secrets; never read, echo, or store key values. -- If `project.json` or the studio config is malformed, stop and report; do not reconstruct state by guessing. +Durable STYLE facts (overlay taste, density preferences, motion feel, CTA appetite) go to `{brand-path}/production-bible.md` instead, ISO-dated in its Learnings log. Keep creator-profile.md to identity and niche, so the two never compete to describe the same thing. diff --git a/skills/mc-agent/customize.toml b/skills/mc-agent/customize.toml deleted file mode 100644 index d720eb0..0000000 --- a/skills/mc-agent/customize.toml +++ /dev/null @@ -1,82 +0,0 @@ -# DO NOT EDIT -- overwritten on every update. -# -# Manny the Manticore, the Visionary Director, is the hardcoded identity -# of this agent. Customize the persona and menu below to shape behavior -# without changing who the agent is. -# Override files (not edited here): -# {project-root}/_bmad/custom/mc-agent.toml (team) -# {project-root}/_bmad/custom/mc-agent.user.toml (personal) - -[agent] -# non-configurable skill frontmatter, create a custom agent if you need a new name/title -name = "Manny" -title = "Visionary Director" - -# --- Configurable below. Overrides merge per BMad structural rules: --- -# scalars: override wins - arrays (persistent_facts, principles, activation_steps_*): append -# arrays-of-tables with `code`/`id`: replace matching items, append new ones. - -icon = "🎬" - -# Steps to run before the standard activation (persona, pulse check, greet). -activation_steps_prepend = [] - -# Steps to run after greet but before presenting the menu. -activation_steps_append = [] - -# Persistent facts the agent keeps in mind for the whole session: literal -# sentences, or "file:{project-root}/..." paths (globs supported) whose -# contents are loaded as facts. Overrides append. -persistent_facts = [] - -role = "Front the whole Manticore studio: onboard the creator, turn ideas into projects, track every production through the pipeline, route to the right stage skill, coach on craft, and grow the studio with new skills." -identity = "A visionary director with a manticore's anatomy: lion's heart for the big swing, human eye for the story only this creator can tell, scorpion's tail reserved for slop and shortcuts." -communication_style = "Golden-age Hollywood warmth with modern shop discipline: greets like the picture just got greenlit, reaches for a movie quote only when the moment truly earns it, uses emojis for energy, and drops all theater the second real mechanics are on the table." - -# The agent's value system. Overrides append to defaults. -principles = [ - "The creator's voice is the product; orchestrate it, never overwrite it.", - "The four gates (outline, cutplan, beats, final) are hard stops; nothing and no one talks past them.", - "Honesty about what is implemented beats hype about what is planned.", - "Mechanics belong to stage skills and their scripts; the director's job is vision, momentum, and what comes next.", - "Every session ends with the creator knowing exactly what happens next and who owes what.", -] - -# Capabilities menu. Overrides merge by `code`: matching codes replace the -# item in place, new codes append. Each item has exactly one of `skill` -# (invokes a registered skill by name) or `prompt` (executes the prompt text). - -[[agent.menu]] -code = "NP" -description = "Turn an idea, or existing footage, into a new video project" -skill = "mc-new" - -[[agent.menu]] -code = "GO" -description = "Where are my projects, what's next, run the next stage" -skill = "mc-pipeline" - -[[agent.menu]] -code = "LS" -description = "Livestream lane: build a pre-show graphics pack, or turn a stream VOD into a video" -prompt = "Ask which side of the livestream lane the creator needs. An upcoming stream routes to mc-stream-pack (the per-episode scene pack). An existing stream recording routes to mc-new in ingest mode, creating a footage-first project on the livestream-vod format so it runs inside the pipeline with real gates and state. Never process a VOD beside the pipeline." - -[[agent.menu]] -code = "SU" -description = "Build or tune the studio (setup, tools, brand, editor)" -skill = "mc-setup" - -[[agent.menu]] -code = "TP" -description = "Tour the pipeline: what each stage does, the gates, what's implemented vs planned" -prompt = "Load references/skills-map.md, then walk the creator through the Manticore pipeline using the pipeline map in this skill plus the per-skill cards, honest about lane status. Invoke mc-pipeline if they want live project state. End with a concrete suggested next step." - -[[agent.menu]] -code = "HP" -description = "What can I do here? Everything installed, Manticore and beyond" -prompt = "Read {project-root}/_bmad/_config/bmad-help.csv (the merged catalog of every installed skill across all modules) and present what is actually available, grouped by module, surfacing only what is relevant to where the creator is. For anything Manticore-side needing more depth, load references/skills-map.md. If the catalog is missing, the studio is not built yet: route to onboarding." - -[[agent.menu]] -code = "GS" -description = "Grow the studio: add a new skill or capability to Manticore" -prompt = "Load references/growing-the-studio.md and follow it: check the help catalog for an existing skill first, prefer the BMB builder skills if available, suggest installing BMB if not, and otherwise apply the linked best practices plus the house rules. Scope the new capability with the creator before any file is written." diff --git a/skills/mc-agent/references/flows.md b/skills/mc-agent/references/flows.md index c5d7159..f0dbf7f 100644 --- a/skills/mc-agent/references/flows.md +++ b/skills/mc-agent/references/flows.md @@ -45,4 +45,4 @@ mc-pipeline for real state, always. Then recommend the single next step with rea ## "Can Manticore do X?" -Load `references/skills-map.md` and answer from it. If X exists but is planned, say planned. If X does not exist, the growing-the-studio flow (`references/growing-the-studio.md`) is the honest offer. +Load `{skill-root}/references/skills-map.md` and answer from it. If X exists but is planned, say planned. If X does not exist, the growing-the-studio flow (`{skill-root}/references/growing-the-studio.md`) is the honest offer. diff --git a/skills/mc-agent/references/growing-the-studio.md b/skills/mc-agent/references/growing-the-studio.md index 615add3..021bf16 100644 --- a/skills/mc-agent/references/growing-the-studio.md +++ b/skills/mc-agent/references/growing-the-studio.md @@ -6,11 +6,11 @@ Load this file when the creator wants a capability Manticore does not have. Scop 1. If the BMB builder skills are available in this harness (bmad-workflow-builder, bmad-agent-builder), use them; they are the canonical factory. 2. If not, suggest installing the BMB module (`npx bmad-method install` and add bmb), and offer to proceed without it meanwhile. -3. Without BMB, follow the skill best practices at https://agentskills.io/skill-creation/best-practices and the house rules: taste in files, mechanics in scripts run via `uv run` with PEP 723 metadata, skills as thin routers, config through the studio config and a `customize.toml`, nothing user-specific inside the skill itself. +3. Without BMB, follow the skill best practices at https://agentskills.io/skill-creation/best-practices and the house rules: taste in files, mechanics in scripts run via `uv run` with PEP 723 metadata, skills as thin routers, config through the studio config, nothing user-specific inside the skill itself. ## Joining the pipeline -New stage skills that join the pipeline must conform to the mc-pipeline contract (stage table, project.json, gates); route the creator through mc-pipeline's docs for that contract rather than improvising one. A capability that serves other skills without owning a stage (the mc-audio and mc-ograf shape) needs no stage-table entry: no gate, no project.json state, called by name from whatever needs it. +New stage skills that join the pipeline must conform to the mc-pipeline contract (stage table, project.json, gates); route the creator through mc-pipeline's docs for that contract rather than improvising one. A capability that serves other skills without owning a stage (the mc-audio shape) needs no stage-table entry: no gate, no project.json state, called by name from whatever needs it. ## Check what already exists first diff --git a/skills/mc-agent/references/onboarding.md b/skills/mc-agent/references/onboarding.md index a5784a9..11617fa 100644 --- a/skills/mc-agent/references/onboarding.md +++ b/skills/mc-agent/references/onboarding.md @@ -18,7 +18,7 @@ In this order, conversationally, not as a lecture: ## First project advice -Suggest talking-head as the first format: shortest path through every stage, and the format the pipeline has the most mileage on. A creator arriving with existing footage skips all of this and goes straight to the footage-first flow (`references/flows.md`); setup still has to exist first. +Suggest talking-head as the first format: shortest path through every stage, and the format the pipeline has the most mileage on. A creator arriving with existing footage skips all of this and goes straight to the footage-first flow (`{skill-root}/references/flows.md`); setup still has to exist first. ## Automation questions diff --git a/skills/mc-agent/references/skills-map.md b/skills/mc-agent/references/skills-map.md index a728204..ca3b049 100644 --- a/skills/mc-agent/references/skills-map.md +++ b/skills/mc-agent/references/skills-map.md @@ -54,10 +54,6 @@ Farms the stills and b-roll the beat table calls for through the creator's regis Service skill, no stage or gate: farms sound the way mc-assets farms pictures. Local-first: Kokoro-82M narration and two-host dialogue (stock voices, no cloning), MusicGen-small instrumental beds, AudioLDM2 SFX (16 kHz). Song-with-vocals lane is planned, not implemented; paid lanes opt-in. Called from mc-graphics, mc-stream-pack, and voiceover narration, or directly when the creator asks for sound. First use may build the engine workspace (large downloads, always consented). -### mc-ograf - -Service skill: OGraf broadcast graphics that stay editable, only where the target supports them (DaVinci Resolve 21+ via `[editor] ograf-editable`, or the OBS/SPX-GC live lane). Everyone else gets baked alpha, which works everywhere. - ## Packaging and live ### mc-package diff --git a/skills/mc-assets/SKILL.md b/skills/mc-assets/SKILL.md index d2f3c60..ef43595 100644 --- a/skills/mc-assets/SKILL.md +++ b/skills/mc-assets/SKILL.md @@ -1,31 +1,60 @@ --- name: mc-assets -description: Source and farm the stills and b-roll the beat table calls for, driving the creator's registered CLI tools by name, with real verified imagery preferred over generation. Use at the assets stage. Never generates UI or text that must be accurate. +description: Farm the stills and b-roll the beats need. Use at the assets stage, or when the user says "farm the assets", "find the images", or "get the b-roll". --- # mc-assets -## Steps +Farm every still and clip the approved beat table calls for. The outcome is `assets/` holding exactly one blessed file per beat-row `asset` slot, plus `assets/manifest.json` recording where each one came from. mc-graphics composes overlays against these files and the final render puts them on screen at full size, so that is the bar: the right thing depicted, sourced as high up the Production Bible's hierarchy as the shot allows, and clean at zoom. A generation that misrepresents something real is worse than no asset at all. -1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there) and this skill's own surface (`uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}`; run `{workflow.activation_steps_prepend}` now, `{workflow.activation_steps_append}` after this step, and hold `{workflow.persistent_facts}` as standing context). Resolve `paths` values against `{project-root}`. Read `project.json` (stage `assets`), the beat rows in `beats/beats.md` whose `asset` column names a farmed asset (tolerate 0.x tables without the column: a missing `asset` is `null`, nothing to farm for that row), the format profile, `{brand-path}/production-bible.md` (its image-type policy and sourcing hierarchy govern every choice below), and `{skill-root}/references/generative-editing-rules.md` (hard rules for every generative lane; the checklist mirrors them). If the profile says `generated_broll: banned`, stop and report; something upstream is wrong. -2. For each needed asset, pick the source per the Production Bible's image-type policy and the sourcing hierarchy: real verified imagery first (the creator's own libraries and their locations per the bible, screen recordings, verified photos), generative only for what does not exist, a hand-built text card last. Claim-bearing text accuracy belongs to the SVG/diagrammatic lane (route to mc-graphics; not a farming job). When an asset needs the creator (or any person) in it, pass the approved original photo from `{brand-path}/headshots/` as `--ref` and say it in the prompt: "use the person in this image to {what the asset needs}"; the image models handle the likeness from there. -3. Resolve the generative lane per `[assets]` in the config (`image-provider`, `video-provider`, `escalation-provider`). Each value names either a registered `[[tools]]` CLI (the working default in 1.0) or a metered API lane (`xai-api`, `veo-api`: not implemented in 1.0, planned for 1.0.x). If the lane an asset needs is empty or names nothing registered, STOP and ask the creator which registered tool to use (route to mc-setup step 5 if none exists); never fall back to a metered lane the creator did not explicitly choose. -4. Farm each asset by tool NAME through the farming script, so no session ever has to remember how a tool is driven: first save the resolved config as JSON (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore > `), then write the prompt per the generative editing rules (concrete subject, camera and framing, lighting, mood, brand-adjacent palette where it fits; quote exact strings for any short text that must appear; spell out physics for wardrobe and object edits; always list what must NOT change; ask for margins near canvas edges; expression variants from one reference use "use this person but have them {expression}"), then run `uv run {skill-root}/scripts/farm_asset.py --kind image|video --prompt "..." --provider --config --out-dir [--seconds 8] [--ref ]`. The script resolves the provider against `[[tools]]`, surfaces the tool's `notes` (the persistent memory for driving it), substitutes the prompt into its `headless` invocation (parsed with POSIX shell quoting on every OS; the tool name is resolved via PATH lookup, so Windows npm shims launch by bare name), runs it in `--out-dir` with the environment passed through, and appends provenance rows to `assets/work/manifest.json`. Escalate to `escalation-provider` only for hero shots where realism must not wobble. Long jobs (video generation, large batches) run in the background with proactive progress reports; never leave the creator staring at a silent stage. -5. Revisions regenerate from the ORIGINAL source assets with every accumulated fix expressed in one prompt; never feed a generated output back in as the base for the next edit (a revision of a revision degrades like a photocopy of a photocopy; re-send all the originals with the improved prompt instead). Small deterministic fixes (a logo swap, one wrong text line, a color correction) are composited programmatically (rsvg, ffmpeg), never regenerated. -6. Self-inspect every output at zoom against the request BEFORE the creator sees it: the gesture points at the right target, the expression matches, no anatomical or rendering artifacts, any text is exactly the requested string. An output that fails inspection is retried, not shown. Standing rule: generated footage never depicts UI or text that must be accurate; real UI comes from screen recordings. -7. Blessed slots: candidates, drafts, and retries stay in `assets/work/`. When the creator picks, copy exactly one blessed file per beat-row slot into `assets/`, named by its asset id, and append its row from the work manifest to `assets/manifest.json` (file, kind, prompt, provider, model, cost, date). Report total spend where a lane reports cost (registered CLI tools draw on the creator's subscription and report none). -8. Deadline mode: when project.json carries an event deadline (set at mc-new), order the remaining assets by their hard external gates and cap iteration loops in favor of good-enough delivery; a shipped asset at the deadline beats a perfect asset after it. -9. Update project.json: append `assets` to `stages_done` and set `stage` to the next stage in its `stages` list. +## Resolution rules + +- Bare paths resolve against `{video-path}`, the current video project at `{projects-path}//`. Every bare `assets/` below is that project's folder, never a skill folder. +- `{skill-root}` → this skill's installed directory; files in it always carry it (`{skill-root}/references/generative-editing-rules.md`). +- `{project-root}` → the project working directory. +- `{skill-name}` → the skill directory's basename. + +## On Activation + +1. Load the studio config: `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`. Empty means mc-setup has not run: stop and route the creator there. Resolve `paths` values against `{project-root}`. +2. Read `project.json` (stage `assets`), `beats/beats.md`, and the format profile. The rows to farm are the ones whose `asset` column names an id; a 0.x table with no `asset` column has nothing to farm. If the profile says `generated_broll: banned`, stop and report, because something upstream is wrong. +3. Read `{brand-path}/production-bible.md`. If it does not exist, tell the creator it is missing and that the image-type policy and sourcing hierarchy governing every choice here cannot happen without it, then route to mc-setup and stop. +4. Read `{brand-path}/headshots/`. If it does not exist, tell the creator it is missing and that any asset with the creator or another person in it cannot happen without it, then route to mc-setup and stop. + +Before any generative farm or revision, load `{skill-root}/references/generative-editing-rules.md`. Its rules on chaining, compositing, self-inspection, people, and prompting bind every lane and every provider. + +## Sourcing + +Real verified imagery first: the creator's own libraries at the locations the bible names, screen recordings, verified photos. Generative only for what does not exist. A hand-built text card last. + +Generated footage never depicts UI or text that has to be accurate; real UI comes from screen recordings, and claim-bearing text belongs to mc-graphics' SVG/diagrammatic lane rather than to farming. Any asset with the creator or another person in it starts from an approved original photo in `{brand-path}/headshots/` passed as `--ref`. + +## Lanes + +`[assets]` names the lane per kind: `image-provider`, `video-provider`, `escalation-provider`. Each value is either the `name` of a registered `[[tools]]` CLI (the working default in 1.0) or a metered API lane (`xai-api`, `veo-api`, both unimplemented until 1.0.x). If the lane an asset needs is empty or names nothing registered, STOP and ask the creator which registered tool to use, routing to mc-setup's tool registration if none exists. Never fall back to a metered lane the creator did not explicitly choose. + +## Farming + +Farm by tool NAME through the script, so no session has to remember how a tool is driven. Save the resolved config as JSON once: + +`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore > ` + +Then, per asset, write the prompt per the generative editing rules and run: + +`uv run {skill-root}/scripts/farm_asset.py --kind image|video --prompt "..." --provider --config --out-dir [--seconds 8] [--ref ]` + +The script prints the tool's `notes` first, which are the persistent memory for driving that tool, and appends a provenance row per new file to `assets/work/manifest.json`. Escalate to `escalation-provider` only for hero shots where realism must not wobble. + +Video generation and large batches run in the background with proactive progress; never leave the creator staring at a silent stage. When project.json carries an event deadline (set at mc-new), order the remaining assets by their hard external gates and cap iteration loops: a shipped asset at the deadline beats a perfect asset after it. + +## Blessing + +Candidates, drafts, and retries stay in `assets/work/`. When the creator picks, copy exactly one blessed file per beat-row slot into `assets/`, named by its asset id, and append its work-manifest row to `assets/manifest.json` (file, kind, prompt, provider, model, cost, date). Report total spend for the lanes that report cost; registered CLI tools draw on the creator's own subscription and report none. + +Then update project.json: append `assets` to `stages_done` and set `stage` to the next stage in its `stages` list. ## Checklist -- Every blessed asset in `assets/` maps to a beat-row slot, exactly one per slot; alternates and retries live in `assets/work/`. -- Sourcing hierarchy honored per the Production Bible: real verified imagery first, generative second, hand-built text card last. -- No chained generative edits: every revision regenerated from the original sources; every `--ref` was real photography or the original source, never a prior generation. -- Small deterministic fixes were composited, never regenerated. -- Every asset featuring a person used an approved original photo as the reference, with the "use the person in this image" prompt pattern. -- Every presented output passed zoom self-inspection (gesture target, expression, artifacts, exact text strings). -- Nothing in `assets/` contains readable UI or text meant to be accurate; short quoted strings verified character-exact. -- `assets/manifest.json` complete, costs summed where the lane reports them. -- Long generative jobs ran in the background with progress reported; deadline mode applied when the project carries one. -- No metered lane was used without the creator's explicit configuration or consent. +- Exactly one blessed file in `assets/` per beat-row `asset` slot; every alternate and retry stayed in `assets/work/`. +- `assets/manifest.json` carries a row per blessed file, with spend summed for the lanes that report it. +- Nothing in `assets/` shows readable UI or misrepresents anything real, and every short quoted string is character-exact. diff --git a/skills/mc-assets/customize.toml b/skills/mc-assets/customize.toml deleted file mode 100644 index 94e8026..0000000 --- a/skills/mc-assets/customize.toml +++ /dev/null @@ -1,20 +0,0 @@ -# DO NOT EDIT -- overwritten on every update. -# -# Customization surface for mc-assets. -# Override files (not edited here): -# {project-root}/_bmad/custom/mc-assets.toml (team) -# {project-root}/_bmad/custom/mc-assets.user.toml (personal) -# -# Studio-wide config (owner, paths, editor, transcription, assets, tools) -# lives in [modules.manticore] in {project-root}/_bmad/custom/config.toml, -# maintained by mc-setup. This file holds only this skill's own surface. - -[workflow] - -# Steps to run before / after the standard activation. -activation_steps_prepend = [] -activation_steps_append = [] - -# Persistent facts held for the whole run: literal sentences, or -# "file:{project-root}/..." paths whose contents are loaded as facts. -persistent_facts = [] diff --git a/skills/mc-assets/references/generative-editing-rules.md b/skills/mc-assets/references/generative-editing-rules.md index 9b93a2d..a2985aa 100644 --- a/skills/mc-assets/references/generative-editing-rules.md +++ b/skills/mc-assets/references/generative-editing-rules.md @@ -1,31 +1,34 @@ # Generative Editing Rules -Safety rules for every generative asset lane (image and video, any provider). mc-assets applies them on every farm and every revision; they are mirrored in the mc-assets checklist and in the `farm_asset.py` docstring. Violating any of them produces the classic generative failure modes: compounding artifacts, mutated subjects, and wasted iteration loops. +Hard rules for every generative asset lane, image or video, any provider. mc-assets applies them on every farm and every revision, and `farm_asset.py` restates them in its docstring. Violating them produces the classic generative failure modes: compounding artifacts, mutated subjects, and wasted iteration loops. ## Rule 1: never chain generative edits -Every revision regenerates from the ORIGINAL source assets with all accumulated fixes expressed in one prompt. Never feed a generated output back in as the base for the next edit: passing the revision of a revision of a revision degrades like a photocopy of a photocopy. When a tweak is needed (say a thumbnail built from some assets and a prompt), start fresh: send ALL the original assets again with one improved prompt, not the first version plus a delta. Reference inputs (`--ref`) take real photography or the original source only, never a prior generation. +Every revision regenerates from the ORIGINAL source assets with all accumulated fixes expressed in one prompt. A generated output is never the base for the next edit: a revision of a revision degrades like a photocopy of a photocopy. When a tweak is needed, send ALL the original assets again with one improved prompt, not the last version plus a delta. `--ref` takes real photography or the original source only, never a prior generation. -## Rule 2: small deterministic fixes are composited, never regenerated +## Rule 2: composite small deterministic fixes, never regenerate them -A logo swap, one wrong text line, a color correction: these are programmatic composites (rsvg, ffmpeg), not regeneration jobs. Regenerating a whole asset to fix one deterministic element risks everything else that was already right. +A logo swap, one wrong text line, a color correction: these are programmatic composites (rsvg, ffmpeg). Regenerating a whole asset to fix one deterministic element risks everything else that was already right. ## Rule 3: self-inspect before the creator sees anything -Inspect every output against the request at zoom BEFORE presenting it: the gesture points at the right target, the expression matches, no anatomical or rendering artifacts, all text is exactly the requested string. An output that fails inspection is retried, not shown. +Inspect every output against the request at zoom: the gesture points at the right target, the expression matches, no anatomical or rendering artifacts, all text is exactly the requested string. An output that fails inspection is retried, not shown. ## Rule 4: people come from their original photos -To put the creator (or anyone) in an asset, pass the approved original photo as a reference input and say it in the prompt: "use the person in this image to {whatever the asset needs}". Current image models handle the likeness from there; no pre-processing, masking, or cutout step is required. What breaks likeness is not the model, it is chaining (rule 1): every revision re-sends the same original photo with the revised prompt, never a prior generation of the person. +To put someone in an asset, pass their approved original photo as `--ref` and say so in the prompt: "use the person in this image to {whatever the asset needs}". Current models handle the likeness from there, with no masking or cutout step. What breaks likeness is chaining, so every revision re-sends that same original photo with the revised prompt. -## Rule 5: prompting rules +## Rule 5: prompting -- Short text that must appear in the asset is quoted as an exact string in the prompt ("the badge reads exactly: \"1.0\""). This replaces the stale blanket warning that generated text is always gibberish; current models render short quoted strings reliably, and exactness comes from quoting. +- Concrete subject, camera and framing, lighting, mood, and a brand-adjacent palette where it fits. +- Short text that must appear in the asset is quoted as an exact string ("the badge reads exactly: \"1.0\""). Current models render short quoted strings reliably and exactness comes from the quoting, so the old blanket warning that generated text is gibberish no longer holds. - Wardrobe and object edits spell out the physics (how fabric hangs, what the object rests on, what occludes what) instead of naming the item alone. -- Always list what must NOT change, explicitly, every time (face, pose, lighting, background, framing). +- List what must NOT change, explicitly, every time: face, pose, lighting, background, framing. - Ask for margins when content sits near canvas edges; generation crops and drifts at borders. -- Expression variants from one reference image use the pattern: "use this person but have them {expression}". +- Expression variants from one reference image use the pattern "use this person but have them {expression}". -## Rule 6: long tasks run in background; deadlines cap iteration +## Rule 6: long jobs run in background, deadlines cap iteration -Long generative jobs (video generation, large batches) run in the background with proactive progress reporting; the creator is never left staring at a silent stage. When the project carries an event deadline (set at mc-new), deadline mode orders deliverables by their hard external gates and caps iteration loops in favor of good-enough delivery: a shipped asset at the deadline beats a perfect asset after it. +Enforced in the farming flow rather than here; the mc-assets SKILL.md carries it. +Numbered so the rule numbering stays aligned with `farm_asset.py`, which cites +these rules by number in its docstring and in its error text. diff --git a/skills/mc-audio/SKILL.md b/skills/mc-audio/SKILL.md index b73ee2e..88d16cd 100644 --- a/skills/mc-audio/SKILL.md +++ b/skills/mc-audio/SKILL.md @@ -1,20 +1,46 @@ --- name: mc-audio -description: Farm sound for the studio, local-first: single-voice TTS narration and multi-host dialogue (Kokoro-82M), instrumental music beds (MusicGen-small), and SFX (AudioLDM2), with paid lanes strictly opt-in. A service skill like mc-ograf, not a pipeline stage: no gate, no project.json state; call it from mc-graphics (whooshes, stingers), mc-stream-pack (beds, stinger audio), the voiceover-explainer format (narration), or whenever the creator asks for sound. +description: Farm narration, music beds, and SFX locally. Use when another skill needs sound, or when the user says "add narration", "music bed", or "sound effect". --- # mc-audio -mc-assets farms pictures; this skill farms sound. It is a service skill: it owns no stage, stops at no gate, and writes no project state. The caller (another skill or the creator directly) says what sound is needed and where it lands; this skill resolves the lane, runs the engine, and hands back files with provenance. Read `references/audio-lanes.md` before farming anything; the ladder, the limits, and the honesty rules there are binding. +mc-assets farms pictures; this skill farms sound. A caller (another skill, or the creator directly) says what sound is needed and where it lands; you resolve the lane, run the engine, and hand back files with provenance. It is a service skill: it owns no stage, stops at no gate, and writes no project state. The caller mixes what you hand back without hearing it first, so the bar is honesty about what came back. -## Steps +Read `{skill-root}/references/audio-lanes.md` before farming anything; the ladder, the limits, and the honesty rules there are binding. -1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there) and this skill's own surface (`uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}`; run `{workflow.activation_steps_prepend}` now, `{workflow.activation_steps_append}` after this step, and hold `{workflow.persistent_facts}` as standing context). Resolve `paths` values against `{project-root}`. From `[audio]` take the lane values (`tts-provider`, `music-provider`, `sfx-provider`, `song-provider`) and `workspace`; the engine workspace is `{engines-path}/{audio.workspace}`. -2. Resolve the lane for the requested kind. The implemented 1.0 lanes are the local defaults: `kokoro-local` (tts and podcast), `musicgen-local` (music), `audioldm2-local` (sfx). A paid or planned value (`gemini-tts`, `elevenlabs-*`, `stable-audio-open`, `ace-step-local`) means the creator opted into a lane that has not landed: say so plainly and stop; never substitute a paid lane the creator did not choose, and never pretend an unvalidated lane works. An empty `song-provider` is the shipped state: full songs with vocals have no validated local lane yet (ACE-Step is the planned candidate; see the reference). -3. Workspace check: `uv run {skill-root}/scripts/ensure_workspace.py --workspace --check`. Not ready: tell the creator what a bootstrap downloads (venv wheels of several GB; on Windows with an NVIDIA GPU, torch installs CUDA wheels from the PyTorch cu126 index, adding roughly 2.5 to 3 GB more; ~340 MB of Kokoro models now, ~5 GB of Hugging Face cache on the first music/sfx run), get their go-ahead, then run it without `--check`. The script's `--dry-run` JSON includes a `torch` field stating which wheel source this machine will use; relay it verbatim during consent. Idempotent: an existing validated workspace (a lab the creator built by hand counts) is used as-is, never rebuilt or duplicated. -4. Farm through the entry script, one call per asset: `uv run {skill-root}/scripts/farm_audio.py --kind tts|podcast|music|sfx --provider --workspace --out-dir [--name ]` plus the kind's arguments (`--text/--voice/--speed`, `--script lines.json`, `--prompt/--seconds/--seed`). For podcast dialogue, write the script JSON per the shape in the reference and apply the realism recipe knobs (speed variation, gaps, backchannels) rather than uniform lines. The script appends provenance to `/manifest.json` (same row shape as mc-assets; cost is null on local lanes). -5. Listen before presenting: play or inspect every output (duration matches the request, no silent or truncated file, dialogue lines land in order). Deliver with the honest caveats from the reference where they apply: SFX are 16 kHz (fine under a mix, thin exposed solo), music is instrumental only, TTS voices are stock (no cloning, no "your voice" claims), crosstalk is simulated. -6. First-run model downloads are long: run them in the background with proactive progress reports; never leave the creator staring at a silent stage. Report where every file landed and what was appended to the manifest. +## Resolution rules + +- Bare paths resolve against `{video-path}`, the current video project at `{projects-path}//`. +- `{skill-root}` → this skill's installed directory; files in it always carry it (`{skill-root}/references/audio-lanes.md`). +- `{project-root}` → the project working directory. + +## On Activation + +1. Load the studio config: `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`. Empty means mc-setup has not run: stop and route the creator there. +2. Resolve `paths` values against `{project-root}`. From `[audio]` take the lane values (`tts-provider`, `music-provider`, `sfx-provider`, `song-provider`) and `workspace`; the engine workspace is `{engines-path}/{audio.workspace}`. + +## The lane + +The implemented 1.0 lanes are the local defaults: `kokoro-local` (tts and podcast), `musicgen-local` (music), `audioldm2-local` (sfx). A paid or planned value (`gemini-tts`, `elevenlabs-*`, `stable-audio-open`, `ace-step-local`) means the creator opted into a lane that has not landed: say so plainly and stop. Never substitute a paid lane the creator did not choose, and never pretend an unvalidated lane works. An empty `song-provider` is the shipped state, not a misconfiguration: full songs with vocals have no validated local lane yet. + +## Workspace + +`uv run {skill-root}/scripts/ensure_workspace.py --workspace --check`. Not ready: tell the creator what a bootstrap downloads (venv wheels of several GB; on Windows with an NVIDIA GPU, torch installs CUDA wheels from the PyTorch cu126 index, adding roughly 2.5 to 3 GB more; ~340 MB of Kokoro models now, ~5 GB of Hugging Face cache on the first music/sfx run), relay the `torch` field of the script's `--dry-run` JSON verbatim so they know which wheel source this machine will use, get their go-ahead, then run it without `--check`. An existing validated workspace is used as-is, a lab the creator built by hand included: never rebuilt, never duplicated. + +## Farming + +One call per asset: + +`uv run {skill-root}/scripts/farm_audio.py --kind tts|podcast|music|sfx --provider --workspace --out-dir [--name ]` plus the kind's arguments (`--text/--voice/--speed`, `--script lines.json`, `--prompt/--seconds/--seed`). The script appends provenance to `/manifest.json` (same row shape as mc-assets; cost is null on local lanes). + +Podcast dialogue takes a script JSON in the shape the reference gives, with its realism knobs applied (speed variation, gaps, backchannels) rather than uniform lines. + +First-run model downloads are long: run them in the background with proactive progress reports. + +## Delivery + +Listen to or inspect every output before presenting it: duration matches the request, nothing silent or truncated, dialogue lines in order. Deliver with the caveats from the reference that apply: SFX are 16 kHz (fine under a mix, thin exposed solo), music is instrumental only, TTS voices are stock (no cloning, no "your voice" claims), crosstalk is simulated. Report where every file landed. ## Checklist @@ -22,5 +48,4 @@ mc-assets farms pictures; this skill farms sound. It is a service skill: it owns - Workspace bootstrap (and its downloads) was consented to before anything was fetched; an existing workspace was reused, not rebuilt. - Every delivered clip was listened to or duration-verified against the request; nothing silent, truncated, or out of order shipped. - Caveats delivered with the assets they apply to: 16 kHz SFX, instrumental-only music, stock TTS voices, simulated crosstalk. -- Podcast scripts used the realism recipe (distinct voices, varied speed and gaps, backchannels under the other speaker), not uniform stitched lines. -- `manifest.json` in the destination gained one row per file. +- Podcast scripts varied voices, speed, and gaps and put backchannels under the other speaker, rather than stitching uniform lines. diff --git a/skills/mc-audio/customize.toml b/skills/mc-audio/customize.toml deleted file mode 100644 index 19e6ff5..0000000 --- a/skills/mc-audio/customize.toml +++ /dev/null @@ -1,21 +0,0 @@ -# DO NOT EDIT -- overwritten on every update. -# -# Customization surface for mc-audio. -# Override files (not edited here): -# {project-root}/_bmad/custom/mc-audio.toml (team) -# {project-root}/_bmad/custom/mc-audio.user.toml (personal) -# -# Studio-wide config (owner, paths, audio lanes, tools) lives in -# [modules.manticore] in {project-root}/_bmad/custom/config.toml, -# maintained by mc-setup ([audio] carries the lane providers and the -# workspace name). This file holds only this skill's own surface. - -[workflow] - -# Steps to run before / after the standard activation. -activation_steps_prepend = [] -activation_steps_append = [] - -# Persistent facts held for the whole run: literal sentences, or -# "file:{project-root}/..." paths whose contents are loaded as facts. -persistent_facts = [] diff --git a/skills/mc-audio/references/audio-lanes.md b/skills/mc-audio/references/audio-lanes.md index 448d47a..5af09d4 100644 --- a/skills/mc-audio/references/audio-lanes.md +++ b/skills/mc-audio/references/audio-lanes.md @@ -1,19 +1,19 @@ # Audio Lanes -The provider ladder for every kind of generated sound, following the settled module pattern: a free local provider is the default, paid vendors are explicit opt-ins whose key names never ship in defaults. The three local lanes below were validated end to end on Apple Silicon (M4 Pro, 24 GB) on 2026-07-07: fully local, no cloud, no API keys, no gated Hugging Face models. +The provider ladder for every kind of generated sound: a free local default, paid vendors as explicit opt-ins. The three local lanes below are validated end to end on Apple Silicon: fully local, no cloud, no API keys, no gated Hugging Face models. ## TTS: narration and multi-host dialogue -Default `kokoro-local`: Kokoro-82M via kokoro-onnx 0.5.0, model files kokoro-v1.0.onnx (~310 MB) plus voices-v1.0.bin (~27 MB) from the kokoro-onnx GitHub releases, stored in the workspace `models/`. Validated at roughly 6.6x realtime on CPU (79 s of finished two-host audio in 12 s wall clock). This settles the earlier open provider choice (Kokoro over Chatterbox). +Default `kokoro-local`: Kokoro-82M via kokoro-onnx 0.5.0, model files kokoro-v1.0.onnx (~310 MB) plus voices-v1.0.bin (~27 MB) from the kokoro-onnx GitHub releases, stored in the workspace `models/`. Roughly 6.6x realtime on CPU: 79 s of finished two-host audio in 12 s wall clock. -The two-host realism recipe (implemented in the tts_kokoro.py payload; what makes it sound like a conversation instead of two stitched TTS files): +Two hosts have to sound like a conversation, not two stitched TTS files. Write the script JSON with: -- Two distinct voices, one per host (for example am_michael and af_heart). +- Two distinct voices, one per host (for example am_michael and af_heart), panned apart. - Per-line speed variation so the pacing breathes. - Variable inter-line gaps; humans do not leave uniform silence. A negative gap overlaps the previous line. -- Backchannels ("Mm-hm.", "Right.") rendered at 0.4x gain and overlapped under the other speaker WITHOUT advancing the timeline cursor. -- Constant-power stereo panning per host, like a two-mic room. -- Soft-limited master normalized to about -1 dBFS. +- Backchannels ("Mm-hm.", "Right.") flagged `backchannel: true`; tts_kokoro.py renders them at 0.4x gain under the other speaker without advancing the timeline cursor. + +tts_kokoro.py applies a constant-power pan law per host and soft-limits the master to about -1 dBFS. Script JSON shape for `--kind podcast`: @@ -35,15 +35,15 @@ Paid/cloud rungs, opt-in only, planned: `gemini-tts` (cheap cloud two-host), `el ## Music beds -Default `musicgen-local`: facebook/musicgen-small via transformers, ungated (no token, no license click-through). Validated: 10.2 s of usable intro-theme music in 55 s on MPS. Instrumentals only: beds, stingers, intro themes; no vocals or lyrics. Fine for beds and stingers; long-form structured music needs musicgen-medium (slower) or an external tool. +Default `musicgen-local`: facebook/musicgen-small via transformers, ungated (no token, no license click-through). 10.2 s of usable intro-theme music in 55 s on MPS. Instrumentals only: beds, stingers, intro themes; no vocals or lyrics. Long-form structured music needs musicgen-medium (slower) or an external tool. -Stable Audio Open produces better output but is gated behind a Hugging Face license click-through, which breaks a zero-friction install; it is demoted to opt-in, never the default. `eleven-music` is the paid opt-in rung (planned). +Stable Audio Open produces better output but is gated behind a Hugging Face license click-through, which breaks a zero-friction install; it is opt-in, never the default. `eleven-music` is the paid opt-in rung (planned). ## SFX -Default `audioldm2-local`: cvssp/audioldm2 via diffusers, ungated. Validated: 7 to 14 s per 4 s effect on MPS once cached. Output is 16 kHz: fine for whooshes, chimes, and ambience under a mix, thin when exposed solo; upsample/EQ or layer it, and say so when delivering an exposed effect. +Default `audioldm2-local`: cvssp/audioldm2 via diffusers, ungated. 7 to 14 s per 4 s effect on MPS once cached. Output is 16 kHz: fine for whooshes, chimes, and ambience under a mix, thin when exposed solo; upsample/EQ or layer it, and say so when delivering an exposed effect. -CRITICAL DEPENDENCY PIN: AudioLDM2's diffusers pipeline breaks with transformers >= 4.44. The validated pair is `diffusers==0.31.0` + `transformers==4.43.4`. Kokoro and MusicGen are unaffected by the pin, so one workspace venv serves all three engines; ensure_workspace.py installs exactly this pair and verifies it. +CRITICAL DEPENDENCY PIN: AudioLDM2's diffusers pipeline breaks with transformers >= 4.44. The validated pair is `diffusers==0.31.0` + `transformers==4.43.4`, which Kokoro and MusicGen tolerate, so one workspace venv serves all three engines; ensure_workspace.py installs exactly this pair and verifies it. `elevenlabs-sfx` (SFX v2) is the paid opt-in rung (planned). @@ -52,9 +52,3 @@ CRITICAL DEPENDENCY PIN: AudioLDM2's diffusers pipeline breaks with transformers `song-provider` ships empty; no local lane is implemented or promised. The leading candidate is ACE-Step 1.5 (MIT license, ungated, native Mac support via an MLX backend; the XL 4B models want the 12 to 20 GB memory tier and run about 2 minutes per 60 s clip on an M1 Max). It is NOT yet validated; mark it planned wherever it comes up and never promise it works. Do NOT plan around YuE for local use: no MLX/Metal port exists, the community floor is 32 to 64 GB unified memory, and Mac wall clock is hours per 30 s of audio. YuE is cloud or rented GPU only, if ever. - -## First-run costs (state before bootstrapping) - -- Workspace venv: torch-class wheels, several GB of disk. -- Kokoro model files: ~340 MB download at workspace build. -- MusicGen + AudioLDM2: roughly 5 GB into the workspace HF cache on the first music/sfx run (shared between the two). diff --git a/skills/mc-beats/SKILL.md b/skills/mc-beats/SKILL.md index e41e331..fe6fc6b 100644 --- a/skills/mc-beats/SKILL.md +++ b/skills/mc-beats/SKILL.md @@ -1,54 +1,82 @@ --- name: mc-beats -description: Riff graphic and motion ideas with the creator, then build the graphics beat table (id, timing, anchor words, type, engine, asset, composition per beat) from the approved cut, plan CTA beats, and STOP for gate 3 approval. Use at the beats stage. Never writes graphics code. +description: Riff visuals, then build the graphics beat table. Use at the beats stage after gate 2, or when the user says "plan the graphics", "beat table", or "what visuals go here". --- # mc-beats -Gate 3. The beat table is the engine-neutral contract between the script and the graphics engines; no graphics code exists until the creator approves it. Read `references/density-and-creativity.md` and `references/cta-placement.md` in full before planning any beats; the Creativity Mandates below are binding on every plan. +Act as the creator's graphics planner. The outcome is an approved beat table: `beats/beats.md` plus `beats/STORYBOARD.md`. -## Steps +The table is the engine-neutral contract between the script and the graphics engines, and no graphics code exists until the creator approves it at gate 3. Three consumers set the bar. The creator must be able to picture every beat from its storyboard paragraph alone. mc-assets farms from the `asset` column. mc-graphics renders from `type`, `engine`, and `composition` with nobody in the room to ask what was meant. -1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there) and this skill's own surface (`uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}`; run `{workflow.activation_steps_prepend}` now, `{workflow.activation_steps_append}` after this step, and hold `{workflow.persistent_facts}` as standing context). Resolve `paths` values against `{project-root}`. Read `project.json` (confirm `approvals.cutplan` is a date, stage is `beats`), `script.md`, `cut/edl.json`, `transcript/`, the format profile at `{formats-path}/.md`, `{brand-path}/production-bible.md`, and `{brand-path}/tokens.json`. From the format profile frontmatter take `beat-types` (the beat types this format allows) and `density` (the map of high/medium/low tiers to seconds-per-beat budgets, front-loaded). The density tier is `graphics-frequency` in `[style]` of the studio config (`medium` when unset) unless the format profile frontmatter overrides it; the profile's `density` map turns the chosen tier into this plan's seconds-per-beat budget. -2. Riff with the creator BEFORE writing the table. Walk the edited timeline once, then bring your strongest ideas to them in plain words: the moments you would put a graphic on and the treatment you would give each, as a short pitch, not a table. Ask what they were picturing: anything specific they already imagined for this video, moments they know they want a visual on, references they have been chewing on. A few minutes of riffing here beats a revision cycle at gate 3. Carry every answer into the plan. And never assume a medium monoculture: beats are not all static text, all SVG, all images, or all clips, gifs, and memes. The mix comes from the Production Bible (the creator's style, learned over time) plus this conversation, never from habit. -3. Walk the EDITED timeline (times derive from edl.json, not the raw take). Scan the transcript with the trigger heuristics in `references/density-and-creativity.md` and, for every moment that earns a graphic, add a row: id, start, dur, end, anchor word with its transcript timestamp, the spoken phrase it rides on, type (one of the profile's `beat-types`), engine, asset, and the composition (named registry block or a one-line description). Apply the Creativity Mandates below to every row and to the plan as a whole. -4. Mark each row's engine per the format profile defaults and PIPELINE.md's engine policy; a profile that still names `remotion` in its `engine_overlays`/`engine_stingers` frontmatter (a studio configured before 2.0.0) is written into the table as `hyperframes` per the engine policy's compatibility alias, never as `remotion`. Rows needing farmed assets carry the asset id in the `asset` column (they become the mc-assets shopping list), all other rows carry `null`. -5. Run the CTA placement pass per `references/cta-placement.md`: read `[cta]` (inventory and appetite) from the studio config, scan the transcript for verbal CTAs and payoff seams, and plan `cta` beats within the reference's zones, caps, and spacing. CTA rows go into the same table with timestamps, anchors, and rationale, approved at gate 3 like any other beat. End-screen rule: no overlay beats in the final 20 seconds unless they ARE the end card. When the inventory includes a next-video or end-card item, optionally add an end-card beat themed from `{brand-path}/tokens.json`. -6. Write `beats/beats.md` (the table) and `beats/STORYBOARD.md`. Each STORYBOARD.md beat gets one short paragraph that doubles as a design brief: what the viewer sees, the motion character (how it enters, moves, and exits), and the anchor phrase it rides on, in plain words a design tool could execute from. -7. Update `artifacts` in project.json (`"beats": "beats/beats.md"`, `"storyboard": "beats/STORYBOARD.md"`), set `approvals.beats = "pending"`, present the table, and STOP for gate 3. -8. On approval: record the ISO date, append `beats` to `stages_done`, and set `stage` to the next entry in project.json's `stages` array. +Read `{skill-root}/references/density-and-creativity.md` (overlay taxonomy, transcript triggers, tier character, pacing curve) and `{skill-root}/references/cta-placement.md` (zones, caps, spacing) in full before planning any beats. -## Creativity Mandates +## The rule that is not inferable -- For every moment, propose the most visually ambitious composition the Production Bible allows before settling for less. The creator can downgrade a diagram to a card in seconds; they cannot upgrade a card to a diagram without doing the planner's job for it. -- A static text card is the composition of last resort. Cap them at roughly a third of rows (aim for the tighter 25% target in the reference) and never place two in a row. -- Vary across the overlay taxonomy in `references/density-and-creativity.md`: at least 6 distinct types in any video over 5 minutes, no single type over 40% of rows. Popups, staged infographic builds, lower thirds, framed real imagery, animated elements, dataviz, and screen zooms all exist for a reason; use them where their triggers fire. -- Meet the minimum beat count for the runtime at the configured tier: edited runtime in minutes times the tier's beats-per-minute floor (high 3, medium 1.5, low 0.7), with roughly double density in the first 30-60 seconds per the pacing curve. A plan below the floor is a failed plan. -- STORYBOARD.md must justify any stretch that exceeds the tier's seconds-per-beat budget; unexplained flat stretches fail the checklist. -- Escalate the treatment to the content: a number gets a stat treatment, a process gets a staged diagram, a comparison gets a split or table build, not a sentence on a card. +NEVER ESTIMATE A BEAT TIME. Every `start` is DERIVED: take the anchor word's timestamp from `transcript/words.json` (original source time) and remap it through `cut/edl.json` onto the clean edited timeline. An eyeballed time that looks right in the table is a graphic that lands off its phrase in the render, discovered after the graphics work is already paid for. + +## Resolution rules + +- Bare paths resolve against `{video-path}`, the current video project at `{projects-path}//`. +- `{skill-root}` → this skill's installed directory; files in it always carry it (`{skill-root}/references/cta-placement.md`). +- `{project-root}` → the project working directory. + +## On Activation + +1. Load the studio config: `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`. Empty means mc-setup has not run; stop and route the creator there. Resolve `paths` values against `{project-root}`. +2. Read `project.json` (confirm `approvals.cutplan` is a date and stage is `beats`), `script.md`, `cut/edl.json`, `transcript/`, `cut/editorial-review.md`, and the format profile at `{formats-path}/.md`. Gate 2 has passed, so the editorial review exists; if it does not, hand back to mc-cut. +3. Read `{brand-path}/production-bible.md`. If it does not exist, tell the creator it is missing and that the medium mix, the composition ambition below, and the density and variety numbers this plan is held to cannot happen without it, then route to mc-setup and stop. +4. Read `{brand-path}/tokens.json`. If it does not exist, tell the creator it is missing and that brand-themed beats cannot happen without it, then route to mc-setup and stop. +5. Fix this plan's vocabulary and budget from the format profile frontmatter: `beat-types` is the whole type vocabulary for the format, and `density` maps tiers to seconds-per-beat budgets. The tier is `graphics-frequency` in `[style]` of the studio config (`medium` when unset), unless the profile frontmatter overrides it. + +## Riff before you plan + +Pitch your strongest ideas before writing any table, and ask what the creator already pictured. The hand-to-beats items in `cut/editorial-review.md` are moments gate 2 already settled need a visual rather than a cut, so lead with them: they are decided work, not pitches. The medium mix comes from the Production Bible and this conversation, never from habit. + +## Build the table + +Walk the EDITED timeline (times derive from `cut/edl.json`, not the raw take). Scan the transcript with the trigger heuristics in `{skill-root}/references/density-and-creativity.md`, and for every moment that earns a graphic add a row. + +Propose the most visually ambitious composition the Production Bible allows before settling for less: the creator can downgrade a diagram to a card in seconds, but cannot upgrade a card to a diagram without doing the planner's job for it. Escalate the treatment to the content, so a number gets a stat treatment, a process gets a staged diagram, and a comparison gets a split or table build. + +Across the plan as a whole, hold the numbers in the Production Bible's visual density and variety section: type variety, the cap on static text cards, and the beats-per-minute floor for the resolved tier, which times the edited runtime in minutes gives the minimum beat count. Resolve them for this format first, since a per-project-type section in the bible overrides the global one. Nothing scripted checks them, so they are yours to hold, and the first 30-60 seconds run at roughly double density per the pacing curve. + +`engine` comes from the format profile defaults and PIPELINE.md's engine policy. A profile that still names `remotion` in its `engine_overlays`/`engine_stingers` frontmatter (a studio configured before 3.0.0) is written into the table as `hyperframes` per the engine policy's compatibility alias, never as `remotion`. Rows needing farmed assets carry the asset id in `asset` and become the mc-assets shopping list; all other rows carry `null`. + +## The CTA pass + +Read `[cta]` (inventory and appetite) from the studio config, scan the transcript for verbal CTAs and payoff seams, and plan `cta` beats within the zones, caps, and spacing in `{skill-root}/references/cta-placement.md`. CTA rows join the same table with timestamps, anchors, and rationale, approved at gate 3 like any other beat. No overlay beats in the final 20 seconds unless they ARE the end card. When the inventory includes a next-video or end-card item, optionally add an end-card beat themed from `{brand-path}/tokens.json`. ## Beat table format -The table follows PIPELINE.md's engine-neutral contract, one row per beat: +One row per beat, per PIPELINE.md's engine-neutral contract: | id | start | dur | end | anchor word | anchor ts | spoken phrase | type | engine | asset | composition | |---|---|---|---|---|---|---|---|---|---|---| -- `type` is one of the format profile's `beat-types` (the frontmatter list is the whole vocabulary for the format). The reserved placeholder `overlay` exists only for READING legacy tables per PIPELINE.md's tolerance rule; mc-beats never writes it. -- `engine` names the rendering engine per the engine policy (e.g. `hyperframes`, `ograf`, `html`). -- `asset` is `null` or a farmed-asset id for mc-assets. -- mc-beats always writes every column. When revising a legacy 0.x table that lacks the extended columns, apply PIPELINE.md's tolerance rule to read it (missing `type` reads as the reserved `overlay` placeholder, missing `engine` is the engine-policy default, missing `asset` is `null`), then write the revised table with all columns filled: every `overlay` placeholder is replaced with a type from the profile's `beat-types`. +`composition` is a named registry block or a one-line description. mc-beats always writes every column. The reserved placeholder `overlay` exists only for READING a legacy 0.x table under PIPELINE.md's tolerance rule (missing `type` reads as `overlay`, missing `engine` is the engine-policy default, missing `asset` is `null`); the revised table is written with all columns filled and every `overlay` replaced by a type from the profile's `beat-types`. + +## Write and verify + +`beats/beats.md` carries the table. `beats/STORYBOARD.md` gives each beat one short paragraph that doubles as a design brief: what the viewer sees, the motion character (how it enters, moves, and exits), and the anchor phrase it rides on, in plain words a design tool could execute from. It also carries what the table has no column for, and what the creator would otherwise have to ask about at the gate: any stretch exceeding the tier's seconds-per-beat budget, any specific ask from the riff or hand-to-beats item that did not become a row, any missing CTA. + +Then the anchor gate: -## Checklist +``` +uv run {skill-root}/scripts/verify_anchors.py beats/beats.md --edl cut/edl.json --words transcript/words.json -o beats/anchor-check.json +``` + +It independently re-derives every beat's time from its anchor word and fails on any beat that does not land within 0.5s of it, on any anchor that is not in the transcript, and on any anchor sitting in a span the cut removed. A non-zero exit is a hard stop: fix the rows it names and re-run. Do not present the table, and do not let graphics be rendered, against a table that has not passed. + +## Before you present + +verify_anchors.py covers anchor placement and nothing else. These are the checks nobody else makes: -- The riff happened before the table: the creator heard the pitched treatments and was asked what they were picturing, and every specific ask from that conversation is in the table or its absence is explained in STORYBOARD.md. -- Every beat has an anchor word that exists in the transcript at that timestamp. - No overlapping beats unless the composition is explicitly layered. -- Every row's `type` is in the format profile's `beat-types`, and `type`, `engine`, and `asset` are filled on every row. -- Beat count meets the tier's minimum for the runtime, front-loaded per the pacing curve; STORYBOARD.md justifies any stretch exceeding the seconds-per-beat budget. -- Variety quota holds: at least 6 distinct types (videos over 5 minutes), no type over 40% of rows, static text cards at or under roughly a third and never consecutive. -- Composition consistency: every composition conforms to the Production Bible's overlay style and animation language; one visual system across the whole plan. -- Image-type policy: every asset-bearing row respects the Production Bible's image-type policy (diagrammatic vs generative vs real) for its purpose. -- CTA beats are present per the configured `[cta]` inventory, or their absence is justified in STORYBOARD.md. -- No overlay beats in the final 20 seconds unless they are the end card. -- Generated-asset rows are absent in formats whose profile bans generated b-roll. +- One visual system across the whole plan: every composition conforms to the Production Bible's overlay style and animation language. +- Every asset-bearing row respects the Production Bible's image-type policy (diagrammatic vs generative vs real) for its purpose. +- No generated-asset rows in formats whose profile bans generated b-roll. + +## Gate 3 + +Update `artifacts` in project.json (`"beats": "beats/beats.md"`, `"storyboard": "beats/STORYBOARD.md"`), set `approvals.beats = "pending"`, present the table, and STOP. Gate 3 is a hard stop; only the creator's explicit approval moves it. On approval, record the ISO date, append `beats` to `stages_done`, and set `stage` to the next entry in project.json's `stages` array. diff --git a/skills/mc-beats/customize.toml b/skills/mc-beats/customize.toml deleted file mode 100644 index 2cae6c2..0000000 --- a/skills/mc-beats/customize.toml +++ /dev/null @@ -1,20 +0,0 @@ -# DO NOT EDIT -- overwritten on every update. -# -# Customization surface for mc-beats. -# Override files (not edited here): -# {project-root}/_bmad/custom/mc-beats.toml (team) -# {project-root}/_bmad/custom/mc-beats.user.toml (personal) -# -# Studio-wide config (owner, paths, editor, transcription, assets, tools) -# lives in [modules.manticore] in {project-root}/_bmad/custom/config.toml, -# maintained by mc-setup. This file holds only this skill's own surface. - -[workflow] - -# Steps to run before / after the standard activation. -activation_steps_prepend = [] -activation_steps_append = [] - -# Persistent facts held for the whole run: literal sentences, or -# "file:{project-root}/..." paths whose contents are loaded as facts. -persistent_facts = [] diff --git a/skills/mc-beats/references/cta-placement.md b/skills/mc-beats/references/cta-placement.md index a6d847a..b94494f 100644 --- a/skills/mc-beats/references/cta-placement.md +++ b/skills/mc-beats/references/cta-placement.md @@ -1,15 +1,13 @@ # CTA Placement Reference -Research-backed rules (2024-2026 era) for deciding when, where, and how to place CTA beats in a video, using the transcript and position-based retention logic. mc-beats reads this during its CTA placement pass; mc-package reads it for description lines, the pinned-comment suggestion, and end-screen guidance. Brand-agnostic; the creator's actual inventory comes from `[cta]` in the studio config and the Production Bible's CTA section. +Rules for deciding when, where, and how to place CTA beats in a video, using the transcript and position-based retention logic. mc-beats reads this during its CTA placement pass; mc-package reads it for description lines, the pinned-comment suggestion, and end-screen guidance. Brand-agnostic; the creator's actual inventory comes from `[cta]` in the studio config and the Production Bible's CTA section. ## Core principles 1. Earn before asking. CTAs convert best immediately after a moment of delivered value: a payoff, insight, demo result, or completed segment. Asking before value is delivered depresses both conversion and retention. 2. One primary CTA per video. Multiple CTAs are fine only when spaced apart, serving different purposes, with a clear hierarchy. Competing asks in the same window create choice paralysis and read as noise. -3. Verbal plus on-screen beats either alone. A spoken ask reinforced by a synchronized graphic outperforms voice-only or graphic-only CTAs. When the transcript contains a verbal CTA, always pair it with a graphic, synced to start within about half a second of the spoken words. A silent graphic is acceptable only for low-friction asks (subscribe bug, link-in-description lower third). +3. Verbal plus on-screen beats either alone, and beats no ask at all: a clear, non-pushy verbal ask lifts subscribe conversion by a few percent up to 30-40% relative. When the transcript contains a verbal CTA, always pair it with a graphic synced to start within about half a second of the spoken words. A silent graphic is acceptable only for low-friction asks (subscribe bug, link-in-description lower third). 4. Continue the journey, do not interrupt it. Retention graphs commonly dip at CTA moments; the dips come from jarring, disconnected asks, not from CTAs per se. A CTA framed as the natural next step holds retention; a hard sales pivot does not. -5. Never interrupt tension. No CTAs mid-explanation, mid-demo, during a build-up, or in high-information-density passages. Place them at natural seams: topic transitions, post-payoff moments, chapter boundaries. -6. Asking works. Controlled creator tests consistently show meaningful lift (a few percent up to 30-40% relative increase in subscribe conversion, in some tests a doubling) when a clear, non-pushy verbal ask is present versus absent. The graphic supports the spoken ask; it does not replace it. ## Placement zones by video position @@ -25,7 +23,7 @@ Avoid engagement CTAs; retention is still settling. Permitted: a brief content-r ### Zone C: mid-video peak zone (~25% to ~60%), the primary CTA zone -The primary engagement CTA belongs here, at a post-payoff seam near the 40-50% mark. Retention is typically at its healthiest and viewers have received tangible value; end-of-video placement wastes the ask because a large share of viewers never reaches it (retention collapses in the final 30 seconds). Trigger on payoff moments, not the clock: find the strongest completed value moment nearest 40-50% (a problem just solved, a demo that just worked, a section just wrapped) and anchor the CTA there. Reason-based asks outperform bare asks: prefer copy that states a benefit ("Subscribe for weekly deep dives") over a bare "Subscribe". Platform cards (the info teaser) also belong here, at 50-75% of runtime, tied to the moment a related topic is mentioned; use 1-2 at most. +The primary engagement CTA belongs here, at a post-payoff seam near the 40-50% mark. Trigger on payoff moments, not the clock: find the strongest completed value moment nearest 40-50% (a problem just solved, a demo that just worked, a section just wrapped) and anchor the CTA there. Reason-based asks outperform bare asks, so prefer copy that states a benefit ("Subscribe for weekly deep dives") over a bare "Subscribe". Platform cards (the info teaser) also belong here, at 50-75% of runtime, tied to the moment a related topic is mentioned; use 1-2 at most. ### Zone D: valleys and interior dips (anywhere in the body) @@ -46,7 +44,7 @@ Reserve the last 10-20 seconds as a deliberate outro runway: a talking-head or h - Minimum spacing: no two CTA graphics within 2 minutes or 20% of runtime of each other, whichever is larger. - Never stack: no two different asks within the same 30-second window. Rapid-fire "like AND subscribe AND join" is the canonical failure; one well-placed prompt outperforms three. - Duplicate suppression: if the speaker verbally asks for the same action more than once, graphic-support only the strongest instance (best zone per the rules above); leave the others voice-only, or flag them for trimming when editing is in scope. -- Forbidden zones: the first 30 seconds; mid-sentence; mid-demo or mid-tension; retention valleys; under the end screen. +- Forbidden zones: the first 30 seconds; mid-sentence; mid-demo, mid-explanation, information-dense passages, or any other build-up of tension; retention valleys; under the end screen. ## On-screen treatment @@ -62,7 +60,7 @@ Reserve the last 10-20 seconds as a deliberate outro runway: a talking-head or h | Goal | Best position | Why | |---|---|---| -| Subscribe / like (channel growth) | Mid-video, at the strongest payoff near 40-50% | Retention is highest; end-of-video asks miss most viewers | +| Subscribe / like (channel growth) | Mid-video, at the strongest payoff near 40-50% | Retention is highest; end-of-video asks miss most viewers, since retention collapses in the final 30 seconds | | Comment prompt | Mid-video, phrased as a specific question tied to the content | Specific questions drive meaningful comments; generic "comment below" does not | | Watch next / session continuation | End screen (final 10-20 s), optional card at 50-75% | Natural next step at content end | | Conversion (community, site, newsletter, product) | Pre-outro (~85-95%) as the spoken pitch; description and pinned comment as the click surface | Remaining viewers are most invested; the full video earns the ask | @@ -80,12 +78,11 @@ When the source is an edited livestream VOD: 4. Always add an end-screen runway. Raw stream endings have none; hold or extend the final shot to create the 10-20 second zone and point at a related VOD or highlight video. 5. Clip-to-full-video CTAs: any short or clip cut from the VOD ends with an on-screen CTA pointing to the full video. This is the highest-leverage CTA in a clipping pipeline. -## Confidence notes +## Overrides and trade-offs -- High confidence (multiple independent sources or platform mechanics): end screens limited to the final 5-20 s; simple 2-element end screens beat cluttered ones; verbal plus visual pairing beats either alone; no CTAs in the opening seconds; retention collapses in the final 30 seconds, making end-only subscribe asks weak. -- Medium confidence (creator experiments, tool-vendor studies): the size of the ask-vs-no-ask lift; animated CTAs outperforming static; benefit-framed asks strongly outperforming bare commands. -- Directional (practitioner consensus, no controlled data): exact zone percentages, the 2-minute spacing rule, the 3-CTA cap. Treat these as sane defaults; per-channel retention data overrides them when available. -- Known trade-off: even well-executed CTAs produce small retention dips. A small dip at a well-placed ask is an acceptable cost for conversion; a large dip means the ask was jarring, mistimed, or too long. +The zone percentages, the 2-minute spacing rule, and the 3-CTA cap are defaults; per-channel retention data overrides them wherever the creator has it. + +Even well-executed CTAs produce small retention dips. A small dip at a well-placed ask is an acceptable cost for conversion; a large dip means the ask was jarring, mistimed, or too long. ## Config schema @@ -112,14 +109,3 @@ priority = 1 4. Fill CTA beats from the configured inventory by priority: sync graphics to kept verbal CTAs first, then place silent-eligible low-friction CTAs at the best remaining seams; enforce spacing, caps, and one ask per window. 5. Reserve and validate the end-screen runway: the final shot must tolerate overlays, and the narration should verbally bridge to the watch-next target (flag it when it does not). 6. Emit CTA rows in the beat table with timestamps, anchors, transcript evidence, and rationale, for gate-3 approval like any other beat. - -## Sources - -- vidIQ, What Is a YouTube CTA? Definition, Examples, and How to Write One. https://vidiq.com/blog/post/youtube-cta/ -- Ventress, YouTube CTA Strategy 2025: Convert Viewers to Subscribers. https://ventress.app/blog/youtube-call-to-action-strategy-convert-viewers-subscribers/ -- Mark Brinker, The Real Reason YouTubers Obsess Over Likes and Subscribes (TubeBuddy ask-vs-no-ask experiment). https://www.markbrinker.com/youtube-engagement -- TubeAnalytics, YouTube Cards and End Screens Checklist. https://www.tubeanalytics.net/blog/youtube-cards-end-screens-checklist-for-retention -- Humble & Brag, YouTube End Screens: How to Set Them Up and Optimise Them. https://humbleandbrag.com/blog/youtube-end-screens -- OverseerOS, YouTube Retention Curve Audit. https://www.overseeros.com/blog/youtube-retention-curve-audit -- Viral Idea Marketing, YouTube Video Editing for Livestream Replays. https://www.viralideamarketing.com/post/youtube-video-editing-for-livestream-replays-how-to-cut-and-repurpose-content -- Restream, 9 Ways to Repurpose Your Live Video Content. https://restream.io/blog/repurpose-live-videos/ diff --git a/skills/mc-beats/references/density-and-creativity.md b/skills/mc-beats/references/density-and-creativity.md index 1ef8ba6..4a9ae74 100644 --- a/skills/mc-beats/references/density-and-creativity.md +++ b/skills/mc-beats/references/density-and-creativity.md @@ -1,30 +1,31 @@ # Density and Creativity Reference -Research-backed rules for proposing the graphics beat table from a transcript of talking-head, tutorial, or livestream-VOD content. mc-beats reads this before planning any beats. Goal: maximize retention through purposeful visual variety without clutter. +Rules for proposing the graphics beat table from a transcript of talking-head, tutorial, or livestream-VOD content. Goal: maximize retention through purposeful visual variety without clutter. -## The creativity mandate (read this first) +## The creativity mandate -The single most common failure mode of automated graphics planning is a sparse sequence of plain text cards: a few bullet slides scattered through the video. That is a failed plan. In high-retention channels, plain talking head is the minority of screen time in the opening, and the supporting visuals are diverse: b-roll, animated diagrams, screenshots with motion, annotated zooms, stat treatments, not just text. +The failure this reference exists to prevent is a sparse sequence of plain text cards: a handful of unanimated bullet slides scattered through a long video. In high-retention channels, plain talking head is the minority of screen time in the opening and the supporting visuals are diverse: b-roll, animated diagrams, screenshots with motion, annotated zooms, stat treatments. -Hard rules for every plan: +How much variety a plan owes, how few plain cards it may carry, and how many beats a minute must hold are the creator's numbers, in the Production Bible's visual density and variety section. This is the craft that makes them reachable: -- Variety quota: use at least 6 distinct overlay types (from the taxonomy below) in any video over 5 minutes. No single type may account for more than 40% of proposed beats. -- Plain text card cap: static text-only cards may be at most 25% of beats, and never more than 2 consecutive beats of the same type. Every text moment should first be considered for an upgrade: can it be an icon plus text callout, an animated list build, a stat counter, a diagram, or a screenshot instead? +- A static text-only card is the composition of last resort. Every text moment is first considered for an upgrade to an icon plus text callout, an animated list build, a stat counter, a diagram, or a screenshot. - Escalate the treatment to the content: a number deserves a big animated stat, not a sentence on a card. A process deserves a diagram, not a paragraph. A comparison deserves a split-screen or table build, not two bullets. - Default to motion: every element enters and exits with simple animation (fade, slide, pop, word-by-word build). Static frames read as unfinished. - When in doubt, propose the richer option and mark it optional. The creator can downgrade a diagram to a card in seconds; they cannot upgrade a card to a diagram without doing the planner's job for it. -## Density tiers +The opposite failure is real too: the creator's floor is a floor, not a target to overshoot. Endless zooms, whooshes, and effects on routine sentences fatigue viewers, especially audiences 25 and up. Space elements out, keep one graphic at a time, and save the biggest treatments for genuine peaks. Past roughly 6 beats a minute at the high tier, 3 at medium, and 1.3 at low, the density is itself the clutter. -The graphics-frequency tier comes from `[style]` in the studio config, with per-format overrides recorded in the Production Bible. Two budgets matter: visual changes (any pattern interrupt: cut, zoom, b-roll, graphic) and graphic beats (what this plan proposes). These targets govern the beats: +## Tier character -| Tier | Seconds per beat | Beats per minute | Character | Typical mix | -|---|---|---|---|---| -| high | ~10-20 s | 3-6 | Retention-editing style; something new on screen most of the time; layered sound cues | Heavy b-roll, keyword pops, animated diagrams, zoom annotations; talking head rarely bare for more than 10 s | -| medium | ~20-45 s | 1.5-3 | Polished educational channel; every key point visualized, breathing room between | B-roll, lower thirds, list builds, stat cards, screenshots; bare talking head fine for 15-20 s stretches | -| low | ~45-90 s | 0.7-1.3 | Minimal, authoritative; graphics only where they genuinely clarify | Chapter cards plus the occasional stat, diagram, or screenshot at the most important moments only | +The graphics-frequency tier comes from `graphics-frequency` in `[style]` of the studio config, overridden for one format by that format profile's frontmatter. Its numbers come from elsewhere: seconds per beat from the format profile's `density` frontmatter, and the beats-per-minute floor and the variety and card-cap quotas from the Production Bible's visual density and variety section, which carries any per-project-type override of those three. What each tier feels like on screen: -Benchmarks behind the tiers: high-energy content changes something visually every 5-7 seconds; top creators average roughly 19 shot changes in the first 30 seconds with bare talking head under ~20% of those shots; high-production channels introduce a new stimulus every 20-30 seconds; minimal-touch guidance for long talking segments is one well-placed graphic every 30-60 seconds. +| Tier | Character | Typical mix | +|---|---|---| +| high | Retention-editing style; something new on screen most of the time; layered sound cues | Heavy b-roll, keyword pops, animated diagrams, zoom annotations; talking head rarely bare for more than 10 s | +| medium | Polished educational channel; every key point visualized, breathing room between | B-roll, lower thirds, list builds, stat cards, screenshots; bare talking head fine for 15-20 s stretches | +| low | Minimal, authoritative; graphics only where they genuinely clarify | Chapter cards plus the occasional stat, diagram, or screenshot at the most important moments only | + +Pacing reference points for visual changes of every kind (cut, zoom, b-roll, graphic), which are a wider budget than the graphic beats this plan proposes: fast-paced editing changes something visually every 5-7 seconds, with roughly 19 shot changes in the first 30 seconds and bare talking head under 20% of those shots; a new visual stimulus every 20-30 seconds reads as high production; through long talking stretches, one well-placed graphic every 30-60 seconds is the sparsest pacing that still holds attention. ## Pacing curve (apply at every tier, front-loaded) @@ -32,7 +33,7 @@ Benchmarks behind the tiers: high-energy content changes something visually ever - Mid-video: settle to the tier baseline; favor context-adding b-roll and diagrams over decorative pops. - Reset every 1-2 minutes: insert a deliberate pattern interrupt (chapter card, big diagram, full-screen b-roll) so no long stretch is visually flat. - Chapter and topic changes always get a visual event, regardless of tier. -- Livestream VODs: treat the trimmed VOD like a normal video; density targets apply to the edited runtime. Additionally add context overlays (what is happening, who is speaking, what was just asked) since replay viewers lack the live chat context. +- Livestream VODs: density targets apply to the edited runtime like any other video. Additionally add context overlays (what is happening, who is speaking, what was just asked), since replay viewers lack the live chat context. STORYBOARD.md must justify any stretch that exceeds the tier's seconds-per-beat budget. @@ -61,7 +62,7 @@ STORYBOARD.md must justify any stretch that exceeds the tier's seconds-per-beat ## Transcript-trigger heuristics -Scan the transcript sentence by sentence. Each pattern below is a trigger; the mapped treatment is the default proposal. (Adobe's B-Script research confirmed that transcript-anchored b-roll recommendation produces measurably more engaging edits than unaided placement.) +Scan the transcript sentence by sentence. Each pattern below is a trigger; the mapped treatment is the default proposal. | Transcript trigger | Detect by | Default treatment | |---|---|---| @@ -89,36 +90,9 @@ Priority when triggers collide or the budget is tight: chapter changes > process - Sync to speech: an overlay appears on the exact word it supports and leaves when the point is done. Late or lingering graphics feel broken. - Durations: keyword pops 1-2 s; callouts and lower thirds 3-6 s; b-roll clips 2-5 s; diagrams and list builds as long as the explanation, animating in stages. -- One graphic at a time: never stack two competing overlays (persistent progress bars and captions excepted). +- One graphic at a time: never stack two competing overlays (persistent progress bars and captions excepted), and never overlap beats unless the composition is explicitly layered. - Layout: keep a 3-5% margin from screen edges; never cover the speaker's face or the most informative region of the frame; respect caption space when captions are on. - Consistency: one type scale, one color system, one animation language across the whole video (the Production Bible is that contract). Variety of type, consistency of style. - Readable on a phone: big, high-contrast text; roughly 8 words maximum per text element. - Sound cues: a subtle whoosh or pop on important beats signals attention, but not on every element. - Match the register of the content: no memes in a corporate explainer, no neon gaming pops in a finance channel. When style is unknown, default to clean and neutral and flag tone-dependent beats as swappable. - -## The two anti-patterns - -1. Sparse static text cards. The baseline failure this reference exists to prevent: a handful of unanimated bullet slides across a long video. Fixed by the creativity mandate, the variety quota, and the tier's density floor. -2. Overedited chaos. Endless zooms, whooshes, and effects on routine sentences fatigue viewers, especially audiences 25 and up; even the most-watched hyper-edited channels publicly slowed their editing in 2024 because hyper-stimulus was hurting watch time. Density targets are ceilings as well as floors: space elements out, keep one graphic at a time, and save the biggest treatments for genuine peaks. - -## Plan self-check - -Before presenting the beat table, verify: - -- Beats per minute match the configured tier, with roughly 2x density in the first 30-60 seconds. -- No gap longer than the tier's seconds-per-beat ceiling without any visual event (or a STORYBOARD.md justification). -- At least 6 overlay types used; no type over 40% of beats; plain text cards at or under 25%. -- Every list, number, comparison, process, and chapter change in the transcript has a treatment. -- Every beat is anchored to a specific transcript timestamp and phrase. -- The biggest treatments land on the video's genuine peak moments. -- Nothing overlaps, covers faces, or hugs screen edges. - -## Sources - -- AIR Media-Tech, Advanced retention editing: cutting patterns that keep viewers past minute 8. https://air.io/en/youtube-hacks/advanced-retention-editing-cutting-patterns-that-keep-viewers-past-minute-8 -- Edicion Video Pro, Audience Retention: How to Edit Videos That Keep Viewers Hooked. https://edicionvideopro.com/en/video-workflow-tutorials/audience-retention-how-to-edit-videos-that-keep-viewers-hooked/ -- Huber et al. (Adobe Research, CHI 2019), B-Script: Transcript-based B-roll Video Editing with Recommendations. https://arxiv.org/abs/1902.11216 -- Uppbeat, How to Increase Audience Retention on YouTube. https://uppbeat.io/blog/youtube-growth/youtube-analytics/youtube-audience-retention -- Washington Post (2024), MrBeast calls for slowing down video editing styles. https://www.washingtonpost.com/technology/2024/03/30/video-editing-mrbeast-retention/ -- Increditors, Guide to Hormozi, Abdaal, and MrBeast editing styles. https://increditors.com/an-ultimate-guide-to-alex-hormozi-ali-abdaal-and-mr-beast-video-editing-style-and-methods/ -- FilterGrade, How to Edit Livestreams for YouTube Highlight Videos. https://filtergrade.com/how-to-edit-livestreams-youtube-highlight/ diff --git a/skills/mc-beats/scripts/tests/test-verify_anchors.py b/skills/mc-beats/scripts/tests/test-verify_anchors.py new file mode 100644 index 0000000..2a2f06a --- /dev/null +++ b/skills/mc-beats/scripts/tests/test-verify_anchors.py @@ -0,0 +1,298 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.11" +# /// +"""Tests for verify_anchors.py: a beat whose time was estimated rather than +derived from its anchor word must fail the gate.""" +import importlib.util +import json +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +SCRIPT = Path(__file__).resolve().parent.parent / "verify_anchors.py" +spec = importlib.util.spec_from_file_location("verify_anchors", SCRIPT) +va = importlib.util.module_from_spec(spec) +spec.loader.exec_module(va) + + +def words(pairs): + """Word dicts from (text, start) pairs, each 0.4s long.""" + return [{"word": t, "start": s, "end": round(s + 0.4, 2), + "confidence": 1.0, "i": i, "gap_before": 0.0, "gap_after": 0.0} + for i, (t, s) in enumerate(pairs)] + + +EDL = {"source": "raw/cam.mp4", "segments": [ + # Keeps 0-10 and 20-30 of the source; 10-20 is cut away. + {"source": "raw/cam.mp4", "start": 0.0, "end": 10.0}, + {"source": "raw/cam.mp4", "start": 20.0, "end": 30.0}, +]} +MAPPING = va.build_map(EDL) + + +def table(rows, header="| id | start | dur | end | anchor word | anchor ts | " + "spoken phrase | type |"): + sep = "|" + "|".join(["---"] * (header.count("|") - 1)) + "|" + return "\n".join([header, sep, *rows]) + "\n" + + +class TestBuildMap(unittest.TestCase): + def test_offsets_accumulate_kept_durations(self): + self.assertEqual([s["offset"] for s in MAPPING], [0.0, 10.0]) + + def test_empty_edl_is_an_empty_map(self): + self.assertEqual(va.build_map({"segments": []}), []) + + +class TestOrigToClean(unittest.TestCase): + def test_time_in_the_first_segment(self): + self.assertAlmostEqual(va.orig_to_clean(MAPPING, 4.0), 4.0) + + def test_time_in_a_later_segment_is_shifted(self): + # Source 25.0 sits 5s into the second kept span, which starts at 10. + self.assertAlmostEqual(va.orig_to_clean(MAPPING, 25.0), 15.0) + + def test_time_in_a_removed_span_is_none(self): + # Deliberately does NOT snap: "this anchor was cut away" is the + # answer the gate needs, not a nearby guess. + self.assertIsNone(va.orig_to_clean(MAPPING, 15.0)) + + def test_segment_boundaries_are_inclusive(self): + self.assertIsNotNone(va.orig_to_clean(MAPPING, 0.0)) + self.assertIsNotNone(va.orig_to_clean(MAPPING, 10.0)) + + def test_source_filter_is_honoured(self): + self.assertIsNone(va.orig_to_clean(MAPPING, 4.0, source="raw/b.mp4")) + + +class TestParseBeats(unittest.TestCase): + def test_reads_id_start_and_anchor_columns(self): + rows, skipped = va.parse_beats(table([ + "| b01 | 4.0 | 2.0 | 6.0 | markdown | 4.0 | the markdown is | " + "diagram |"])) + self.assertEqual(skipped, []) + self.assertEqual(rows[0]["id"], "b01") + self.assertEqual(rows[0]["start"], 4.0) + self.assertEqual(rows[0]["anchor_word"], "markdown") + self.assertEqual(rows[0]["anchor_ts"], 4.0) + + def test_column_order_does_not_matter(self): + rows, _ = va.parse_beats(table( + ["| 4.0 | b01 | markdown |"], + header="| start | id | anchor word |")) + self.assertEqual(rows[0]["id"], "b01") + self.assertEqual(rows[0]["anchor_word"], "markdown") + + def test_missing_anchor_column_parses_as_empty(self): + rows, _ = va.parse_beats(table(["| b01 | 4.0 |"], + header="| id | start |")) + self.assertEqual(rows[0]["anchor_word"], "") + + def test_separator_row_is_not_a_beat(self): + rows, _ = va.parse_beats(table([ + "| b01 | 4.0 | 2.0 | 6.0 | markdown | 4.0 | x | diagram |"])) + self.assertEqual(len(rows), 1) + + def test_non_numeric_start_is_skipped_loudly(self): + rows, skipped = va.parse_beats(table([ + "| b01 | soon | 2.0 | 6.0 | markdown | 4.0 | x | diagram |"])) + self.assertEqual(rows, []) + self.assertEqual(len(skipped), 1) + + def test_null_anchor_ts_reads_as_absent(self): + rows, _ = va.parse_beats(table([ + "| b01 | 4.0 | 2.0 | 6.0 | markdown | null | x | diagram |"])) + self.assertIsNone(rows[0]["anchor_ts"]) + + +class TestFindAnchorOccurrences(unittest.TestCase): + W = words([("The", 1.0), ("markdown", 2.0), ("is", 3.0), + ("the", 4.0), ("markdown", 5.0)]) + + def test_finds_every_occurrence(self): + self.assertEqual( + [w["start"] for w in va.find_anchor_occurrences(self.W, "markdown")], + [2.0, 5.0]) + + def test_match_is_case_and_punctuation_insensitive(self): + w = words([("Markdown,", 2.0)]) + self.assertEqual(len(va.find_anchor_occurrences(w, "markdown")), 1) + + def test_multi_word_anchor_matches_a_run(self): + found = va.find_anchor_occurrences(self.W, "markdown is") + self.assertEqual([w["start"] for w in found], [2.0]) + + def test_absent_word_finds_nothing(self): + self.assertEqual(va.find_anchor_occurrences(self.W, "nonsense"), []) + + def test_empty_anchor_finds_nothing(self): + self.assertEqual(va.find_anchor_occurrences(self.W, ""), []) + + +class TestVerifyBeat(unittest.TestCase): + W = words([("The", 1.0), ("markdown", 4.0), ("is", 5.0), + ("code", 25.0), ("cut", 15.0)]) + + def beat(self, **kw): + base = {"id": "b01", "start": 4.0, "anchor_word": "markdown", + "anchor_ts": None, "phrase": ""} + base.update(kw) + return base + + def test_correctly_derived_beat_passes(self): + r = va.verify_beat(self.beat(), self.W, MAPPING) + self.assertEqual(r["status"], "ok") + self.assertAlmostEqual(r["derived_clean"], 4.0) + + def test_beat_in_a_later_segment_uses_the_remapped_time(self): + # "code" is at source 25.0, which is clean 15.0. + r = va.verify_beat(self.beat(anchor_word="code", start=15.0), + self.W, MAPPING) + self.assertEqual(r["status"], "ok") + self.assertAlmostEqual(r["derived_clean"], 15.0) + + def test_estimated_beat_time_fails(self): + # The B1 class: the time was guessed, not derived. Source 25.0 maps + # to clean 15.0, but the planner wrote the SOURCE time into the table. + r = va.verify_beat(self.beat(anchor_word="code", start=25.0), + self.W, MAPPING) + self.assertEqual(r["status"], "violation") + self.assertIn("remapping", r["reason"]) + + def test_small_drift_inside_tolerance_passes(self): + r = va.verify_beat(self.beat(start=4.3), self.W, MAPPING) + self.assertEqual(r["status"], "ok") + + def test_drift_beyond_tolerance_fails(self): + r = va.verify_beat(self.beat(start=5.2), self.W, MAPPING) + self.assertEqual(r["status"], "violation") + + def test_tolerance_is_configurable(self): + self.assertEqual( + va.verify_beat(self.beat(start=5.2), self.W, MAPPING, + tolerance=2.0)["status"], "ok") + + def test_anchor_absent_from_the_transcript_fails(self): + r = va.verify_beat(self.beat(anchor_word="unicorn"), self.W, MAPPING) + self.assertEqual(r["status"], "violation") + self.assertIn("does not appear", r["reason"]) + + def test_anchor_in_a_removed_span_fails(self): + # "cut" sits at source 15.0, inside the span the EDL removed. + r = va.verify_beat(self.beat(anchor_word="cut", start=5.0), + self.W, MAPPING) + self.assertEqual(r["status"], "violation") + self.assertIn("never hears", r["reason"]) + + def test_nearest_occurrence_is_used(self): + w = words([("go", 1.0), ("go", 8.0)]) + r = va.verify_beat(self.beat(anchor_word="go", start=8.0), w, MAPPING) + self.assertEqual(r["status"], "ok") + self.assertAlmostEqual(r["derived_clean"], 8.0) + self.assertEqual(r["occurrences"], 2) + + def test_no_anchor_word_is_unverifiable_not_a_failure(self): + # PIPELINE.md's tolerance rule for legacy 0.x beat tables. + r = va.verify_beat(self.beat(anchor_word=""), self.W, MAPPING) + self.assertEqual(r["status"], "unverifiable") + + def test_anchor_ts_disagreeing_with_the_derived_time_fails(self): + r = va.verify_beat(self.beat(anchor_ts=25.0), self.W, MAPPING) + self.assertEqual(r["status"], "violation") + self.assertIn("EDITED timeline", r["reason"]) + + def test_anchor_ts_agreeing_passes(self): + r = va.verify_beat(self.beat(anchor_ts=4.0), self.W, MAPPING) + self.assertEqual(r["status"], "ok") + + def test_drift_sign_is_reported(self): + r = va.verify_beat(self.beat(start=3.0), self.W, MAPPING) + self.assertAlmostEqual(r["drift"], 1.0) + + +class TestBuildReport(unittest.TestCase): + W = words([("The", 1.0), ("markdown", 4.0), ("code", 25.0)]) + + def test_all_good_beats_report_ok(self): + beats = [{"id": "b01", "start": 4.0, "anchor_word": "markdown", + "anchor_ts": None, "phrase": ""}] + r = va.build_report(beats, self.W, MAPPING) + self.assertTrue(r["ok"]) + self.assertEqual(r["checked"], 1) + + def test_one_bad_beat_fails_the_whole_report(self): + beats = [{"id": "b01", "start": 4.0, "anchor_word": "markdown", + "anchor_ts": None, "phrase": ""}, + {"id": "b02", "start": 25.0, "anchor_word": "code", + "anchor_ts": None, "phrase": ""}] + r = va.build_report(beats, self.W, MAPPING) + self.assertFalse(r["ok"]) + self.assertEqual([v["id"] for v in r["violations"]], ["b02"]) + + def test_unverifiable_rows_do_not_fail_the_report(self): + beats = [{"id": "b01", "start": 4.0, "anchor_word": "", + "anchor_ts": None, "phrase": ""}] + r = va.build_report(beats, self.W, MAPPING) + self.assertTrue(r["ok"]) + self.assertEqual(r["checked"], 0) + self.assertEqual(len(r["unverifiable"]), 1) + + +class TestCli(unittest.TestCase): + def _run(self, rows, extra=None): + with tempfile.TemporaryDirectory() as tmp: + d = Path(tmp) + (d / "beats.md").write_text(table(rows)) + (d / "edl.json").write_text(json.dumps(EDL)) + (d / "words.json").write_text(json.dumps({"words": words([ + ("The", 1.0), ("markdown", 4.0), ("code", 25.0)])})) + return subprocess.run( + [sys.executable, str(SCRIPT), str(d / "beats.md"), + "--edl", str(d / "edl.json"), + "--words", str(d / "words.json"), *(extra or [])], + capture_output=True, text=True) + + def test_good_table_exits_zero(self): + r = self._run(["| b01 | 4.0 | 2.0 | 6.0 | markdown | 4.0 | x | d |"]) + self.assertEqual(r.returncode, 0, r.stderr) + self.assertTrue(json.loads(r.stdout)["ok"]) + + def test_mistimed_beat_exits_one_and_names_the_row(self): + r = self._run(["| b07 | 25.0 | 2.0 | 27.0 | code | 25.0 | x | d |"]) + self.assertEqual(r.returncode, 1) + self.assertIn("ANCHOR PLACEMENT FAILED", r.stderr) + self.assertIn("b07", r.stderr) + + def test_failure_warns_against_rendering(self): + r = self._run(["| b07 | 25.0 | 2.0 | 27.0 | code | 25.0 | x | d |"]) + self.assertIn("Do NOT render graphics", r.stderr) + + def test_empty_table_is_a_usage_error(self): + r = self._run([]) + self.assertEqual(r.returncode, 2) + + def test_missing_edl_is_a_usage_error(self): + with tempfile.TemporaryDirectory() as tmp: + d = Path(tmp) + (d / "beats.md").write_text(table( + ["| b01 | 4.0 | 2.0 | 6.0 | markdown | 4.0 | x | d |"])) + (d / "words.json").write_text(json.dumps({"words": []})) + r = subprocess.run( + [sys.executable, str(SCRIPT), str(d / "beats.md"), + "--edl", str(d / "nope.json"), + "--words", str(d / "words.json")], + capture_output=True, text=True) + self.assertEqual(r.returncode, 2) + + def test_help_exits_zero(self): + r = subprocess.run([sys.executable, str(SCRIPT), "--help"], + capture_output=True, text=True) + self.assertEqual(r.returncode, 0) + self.assertIn("--tolerance", r.stdout) + + +if __name__ == "__main__": + unittest.main() diff --git a/skills/mc-beats/scripts/verify_anchors.py b/skills/mc-beats/scripts/verify_anchors.py new file mode 100644 index 0000000..6865784 --- /dev/null +++ b/skills/mc-beats/scripts/verify_anchors.py @@ -0,0 +1,337 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.11" +# /// +"""Beat anchor placement gate: prove every graphic lands on the words it rides. + +Usage: + uv run {skill-root}/scripts/verify_anchors.py {projects-path}//beats/beats.md \ + --edl {projects-path}//cut/edl.json \ + --words {projects-path}//transcript/words.json \ + [--tolerance 0.5] [-o {projects-path}//beats/anchor-check.json] + +Why this exists: + mc-beats' checklist has always said "every beat has an anchor word that + exists in the transcript at that timestamp". Nothing enforced it. Beat + times were written by eye against the edited timeline, and an overlay + landing a second off its anchor phrase is invisible in a beat table and + obvious in the render, after the graphics work is already done. + + It is the same defect shape as the QC frames that were extracted but + never asserted on, and the boundary frames that were eyeballed while the + underlying audio was wrong: a documented check with no mechanical + assertion behind it. An assertion is a mechanic, not a taste call, so it + belongs in a script that exits non-zero. + +How a beat time SHOULD be derived (and what this checks): + Never estimate. Take the anchor word's timestamp from the transcript, + which is in ORIGINAL source time, and remap it through the EDL onto the + clean edited timeline. That mapping is exactly what this script redoes + independently, so a beat whose time was estimated rather than derived + fails here. + +Checks, per beat row: + 1. The anchor word appears in the transcript at all. + 2. Its source time falls inside a KEPT EDL segment. An anchor sitting in + a span the cut removed is a hard failure: the graphic rides on words + the viewer never hears. + 3. Its remapped clean time is within --tolerance of the beat's start. + When several occurrences of the word exist, the NEAREST one is used, + so the failure means "no occurrence of this word is anywhere near + where you placed this beat". + 4. When the table carries an `anchor ts` column, that value agrees with + the derived clean time too (anchors are measured against the edited + timeline, per PIPELINE.md). + +Rows with no anchor word are reported as unverifiable rather than failed, +per PIPELINE.md's tolerance rule for legacy 0.x beat tables. + +Contract: + input beats.md (the engine-neutral beat table), cut/edl.json, and + transcript/words.json. + output optional -o report JSON: {"ok", "checked", "violations": [...], + "unverifiable": [...]} + summary json.dumps on stdout either way. + +Exit codes: 0 every anchor verified, 1 one or more violations (do not render +graphics against this table), 2 usage error. + +Note on duplication: the orig-to-clean mapping below is a deliberate copy of +the same logic in mc-cut/scripts/remap_timecode.py, per the module's +script-duplication convention (a skill reads only its own folder). + +STATUS: implemented (pure logic covered by +scripts/tests/test-verify_anchors.py). +""" + +import argparse +import json +import string +import sys +from pathlib import Path + +DEFAULT_TOLERANCE_S = 0.5 +_STRIP = string.punctuation + + +def norm(word): + """Lowercase, strip surrounding punctuation (pure).""" + return str(word).strip(_STRIP).lower() + + +def build_map(edl): + """EDL segments as [{source, start, end, offset}] in timeline order (pure).""" + out = [] + offset = 0.0 + for seg in edl["segments"]: + out.append({"source": seg["source"], "start": seg["start"], + "end": seg["end"], "offset": offset}) + offset += seg["end"] - seg["start"] + return out + + +def orig_to_clean(mapping, t, source=None): + """Map an original-source time onto the clean timeline, or None (pure). + + None means the time falls in a span the cut removed. That is a real + answer here, not an error: it is exactly the "anchor was cut away" case + check 2 exists to catch, so this deliberately does NOT snap to the + nearest kept segment the way the chapter remapper does. + """ + for s in mapping: + if source is not None and s["source"] != source: + continue + if s["start"] <= t <= s["end"]: + return s["offset"] + (t - s["start"]) + return None + + +def parse_beats(text): + """Parse the beat table, keeping the anchor columns (pure). + + Returns (rows, skipped). Columns are matched by header name so extra + columns and any order are fine, matching composite_core's parser. + """ + header = None + col = {} + rows, skipped = [], [] + for line in text.splitlines(): + if "|" not in line: + continue + cells = [c.strip() for c in line.strip().strip("|").split("|")] + low = [c.lower() for c in cells] + if header is None: + if "id" in low and "start" in low: + header = low + col = {name: i for i, name in enumerate(low)} + continue + if all(set(c) <= set("-: ") for c in cells if c): + continue # the ---|--- separator row + if len(cells) < len(header): + skipped.append(f"row has {len(cells)} cells, expected " + f"{len(header)}: {cells[:2]}") + continue + + def cell(name): + i = col.get(name) + if i is None or i >= len(cells): + return "" + return cells[i] + + bid = cell("id") + if not bid: + skipped.append(f"row has no id: {cells[:3]}") + continue + try: + start = float(cell("start")) + except ValueError: + skipped.append(f"beat {bid}: start {cell('start')!r} is not a number") + continue + anchor_ts = None + raw_ts = cell("anchor ts") + if raw_ts and raw_ts.lower() not in ("null", "none", "-"): + try: + anchor_ts = float(raw_ts) + except ValueError: + anchor_ts = None + rows.append({ + "id": bid, + "start": start, + "anchor_word": cell("anchor word"), + "anchor_ts": anchor_ts, + "phrase": cell("spoken phrase"), + }) + return rows, skipped + + +def find_anchor_occurrences(words, anchor): + """Every transcript word matching the anchor text (pure). + + A multi-word anchor matches a consecutive run; the occurrence's time is + the first word's start. + """ + tokens = [norm(t) for t in str(anchor).split() if norm(t)] + if not tokens: + return [] + nwords = [norm(w["word"]) for w in words] + out = [] + for i in range(len(nwords) - len(tokens) + 1): + if nwords[i:i + len(tokens)] == tokens: + out.append(words[i]) + return out + + +def verify_beat(beat, words, mapping, tolerance=DEFAULT_TOLERANCE_S): + """Verify one beat row. Returns a result dict (pure). + + status is "ok", "violation", or "unverifiable". + """ + result = {"id": beat["id"], "start": beat["start"], + "anchor_word": beat["anchor_word"]} + if not beat["anchor_word"]: + result["status"] = "unverifiable" + result["reason"] = "row carries no anchor word (legacy 0.x table)" + return result + + occurrences = find_anchor_occurrences(words, beat["anchor_word"]) + if not occurrences: + result["status"] = "violation" + result["reason"] = (f'anchor word "{beat["anchor_word"]}" does not ' + "appear in the transcript at all") + return result + + mapped = [(w, orig_to_clean(mapping, w["start"])) for w in occurrences] + kept = [(w, c) for w, c in mapped if c is not None] + if not kept: + result["status"] = "violation" + result["reason"] = (f'every occurrence of "{beat["anchor_word"]}" ' + "falls in a span the cut removed; this graphic " + "rides on words the viewer never hears") + result["source_times"] = [round(w["start"], 2) for w, _ in mapped] + return result + + word, clean = min(kept, key=lambda wc: abs(wc[1] - beat["start"])) + drift = clean - beat["start"] + result["derived_clean"] = round(clean, 3) + result["source_time"] = round(word["start"], 3) + result["drift"] = round(drift, 3) + result["occurrences"] = len(occurrences) + + if abs(drift) > tolerance: + result["status"] = "violation" + result["reason"] = ( + f'beat starts at {beat["start"]:.2f}s but the nearest ' + f'"{beat["anchor_word"]}" lands at {clean:.2f}s on the clean ' + f"timeline ({drift:+.2f}s off, tolerance {tolerance}s). Derive " + "the time by remapping the anchor's transcript timestamp through " + "the EDL instead of estimating it.") + return result + + if beat["anchor_ts"] is not None and \ + abs(beat["anchor_ts"] - clean) > tolerance: + result["status"] = "violation" + result["reason"] = ( + f'anchor ts column says {beat["anchor_ts"]:.2f}s but the anchor ' + f"lands at {clean:.2f}s on the clean timeline. Anchor times are " + "measured against the EDITED timeline (PIPELINE.md).") + return result + + result["status"] = "ok" + return result + + +def build_report(beats, words, mapping, tolerance=DEFAULT_TOLERANCE_S): + """Verify every beat and assemble the verdict (pure).""" + results = [verify_beat(b, words, mapping, tolerance) for b in beats] + violations = [r for r in results if r["status"] == "violation"] + unverifiable = [r for r in results if r["status"] == "unverifiable"] + return { + "ok": not violations, + "beats": len(results), + "checked": len(results) - len(unverifiable), + "tolerance": tolerance, + "violations": violations, + "unverifiable": unverifiable, + "results": results, + } + + +def _load(path, label): + try: + with open(path, encoding="utf-8") as f: + return json.load(f) + except (OSError, json.JSONDecodeError) as e: + print(f"verify_anchors: cannot read {label} {path}: {e}", + file=sys.stderr) + return None + + +def main(argv=None): + p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + p.add_argument("beats", help="path to beats/beats.md") + p.add_argument("--edl", required=True, help="path to cut/edl.json") + p.add_argument("--words", required=True, + help="path to transcript/words.json") + p.add_argument("-o", "--output", default=None, + help="optional path for the full report JSON") + p.add_argument("--tolerance", type=float, default=DEFAULT_TOLERANCE_S, + help=f"seconds a beat may sit from its anchor (default " + f"{DEFAULT_TOLERANCE_S})") + args = p.parse_args(argv) + + try: + table = Path(args.beats).read_text(encoding="utf-8") + except OSError as e: + print(f"verify_anchors: cannot read beat table {args.beats}: {e}", + file=sys.stderr) + return 2 + edl = _load(args.edl, "edl") + if edl is None: + return 2 + transcript = _load(args.words, "transcript") + if transcript is None: + return 2 + if not edl.get("segments"): + print("verify_anchors: edl has no segments", file=sys.stderr) + return 2 + if "words" not in transcript: + print("verify_anchors: transcript has no 'words' key", file=sys.stderr) + return 2 + + beats, skipped = parse_beats(table) + for reason in skipped: + print(f"beat row skipped: {reason}", file=sys.stderr) + if not beats: + print("verify_anchors: no beat rows found in the table", + file=sys.stderr) + return 2 + + report = build_report(beats, transcript["words"], build_map(edl), + args.tolerance) + + if args.output: + out = Path(args.output) + out.parent.mkdir(parents=True, exist_ok=True) + out.write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n", + encoding="utf-8") + + print(json.dumps({k: report[k] for k in + ("ok", "beats", "checked", "tolerance")}, indent=2)) + + for r in report["unverifiable"]: + print(f"note: beat {r['id']} unverifiable: {r['reason']}", + file=sys.stderr) + + if report["ok"]: + return 0 + + print(f"\nANCHOR PLACEMENT FAILED: {len(report['violations'])} beat(s) do " + "not land on their anchor words. Do NOT render graphics against " + "this table.", file=sys.stderr) + for r in report["violations"]: + print(f" {r['id']}: {r['reason']}", file=sys.stderr) + return 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/mc-braindump/SKILL.md b/skills/mc-braindump/SKILL.md index e37d506..33d36b4 100644 --- a/skills/mc-braindump/SKILL.md +++ b/skills/mc-braindump/SKILL.md @@ -1,31 +1,42 @@ --- name: mc-braindump -description: Interview the creator about a video idea and capture their exact words verbatim. The braindump is the raw material every script sentence must trace back to. Use at the braindump stage or when the creator wants to talk through an idea for a project. +description: Capture the creator's idea in their exact words. Use at the braindump stage, or when the user says "braindump", "let me talk this through", or "here is my idea". --- # mc-braindump The single most important input to the whole pipeline: the script stage may only use words that exist in this file (quote-or-cut). Capture generously. -## Steps - -1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there) and this skill's own surface (`uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}`; run `{workflow.activation_steps_prepend}` now, `{workflow.activation_steps_append}` after this step, and hold `{workflow.persistent_facts}` as standing context). Resolve `paths` values against `{project-root}`. Read `project.json` and `brief.md`. Confirm stage is `braindump`. Read the format profile at `{formats-path}/.md` for the project's `format` and any taste files it names. -2. ALWAYS offer camera-rolling capture (interview-recording mode) before the first question: if the creator records this session (constant frame rate), their spoken answers double as potential takes. The convention: they read each question aloud to the lens as "Question from the interviewer: " before answering; that spoken marker cue lets mc-cut segment the recording mechanically (cut the question reads, keep the answers). The cue is configurable (setup interview; cutplan.py `--marker-cues`); the older "question from claude" phrasing remains a documented alternative for studios that recorded with it. Whether or not the camera rolls, then interview, one question at a time, conversationally. Goal: get the creator talking at length in their own phrasing. Capture each answer into `braindump.md` verbatim as it is given; never rely on conversation memory for their exact words. Cover, in whatever order the conversation goes: - - the core claim (what do they actually believe here), - - who it is for and what they get, - - the story or example they would tell a friend, - - the strongest objection and their answer to it, - - what everyone else gets wrong, - - the demo/proof they can show, - - what the viewer should SEE, asked verbatim: "what should the viewer see: demos, screens, drawings, motion, or moments you picture as a graphic?" (capture the answer word for word like every other; mc-outline and mc-beats read it as candidate visual moments), - - how they would say the payoff in one breath. -3. Keep interviewing until their phrasing repeats (saturation) or they call it. Do not stop at a fixed question count. -4. Finalize `braindump.md`: VERBATIM capture of their answers, lightly grouped under the question headings. Their words untouched: keep their fragments, their slang, their fillers. No paraphrasing, no cleanup, no summarizing. -5. If the session was recorded: have the creator drop the file into the project's `raw/` and register it in project.json `sources` with role `"interview"` (see the PIPELINE.md contract). It is both braindump corpus and candidate takes. -6. Update project.json: set `artifacts.braindump`, append `braindump` to `stages_done`, and set `stage` to the entry after `braindump` in this project's `stages` array. Stop. - -## Rules - -- Never polish their phrasing. The point is a corpus of their real language. -- If the session arrives as an external recording/transcript instead of live conversation, save the raw transcript into the project's `raw/` first, then quote from it in braindump.md. -- Your questions are scaffolding; their answers are the artifact. +## Resolution rules + +- Bare paths resolve against `{video-path}`, the current video project at `{projects-path}//`. +- `{skill-root}` → this skill's installed directory; files in it always carry it. +- `{project-root}` → the project working directory. + +## On Activation + +Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there). Resolve `paths` values against `{project-root}`. Read `project.json` and `brief.md`. Confirm stage is `braindump`. Read the format profile at `{formats-path}/.md` for the project's `format` and any taste files it names. + +## Offer the camera first + +Before the first question, always offer interview-recording mode: if the creator records this session at constant frame rate, their spoken answers double as potential takes. The convention is that they read each question to the lens as "Question from the interviewer: " before answering; that spoken marker cue is what lets mc-cut segment the recording mechanically. The cue is configurable (the setup interview; cutplan.py `--marker-cues`), and the older "question from claude" phrasing stays supported for studios that recorded with it. + +## The interview + +Whether or not the camera rolls, interview one question at a time, conversationally, until their phrasing starts repeating or they call it; there is no question count to hit. Goal: get the creator talking at length in their own phrasing. Capture each answer into `braindump.md` verbatim as it is given, never from conversation memory afterwards. + +Cover, in whatever order the conversation goes: the core claim they actually believe, who it is for and what they get, the story they would tell a friend, the strongest objection and their answer to it, what everyone else gets wrong, the demo or proof they can show, and the payoff in one breath. + +Ask the visual question verbatim: "what should the viewer see: demos, screens, drawings, motion, or moments you picture as a graphic?" mc-outline and mc-beats read that answer as candidate visual moments. + +## The artifact + +`braindump.md` is their answers lightly grouped under the question headings, their words untouched: fragments, slang and fillers all intact. Polishing destroys the corpus the script stage draws from. + +If the session arrives as an external recording or transcript instead of a live conversation, save the raw transcript into the project's `raw/` first, then quote from it in `braindump.md`. + +## Close out + +If the session was recorded, have the creator drop the file into the project's `raw/` and register it in project.json `sources` with role `"interview"` (see the PIPELINE.md contract). + +Update project.json: set `artifacts.braindump`, append `braindump` to `stages_done`, and set `stage` to the entry after `braindump` in this project's `stages` array. Stop. diff --git a/skills/mc-braindump/customize.toml b/skills/mc-braindump/customize.toml deleted file mode 100644 index 8b21096..0000000 --- a/skills/mc-braindump/customize.toml +++ /dev/null @@ -1,20 +0,0 @@ -# DO NOT EDIT -- overwritten on every update. -# -# Customization surface for mc-braindump. -# Override files (not edited here): -# {project-root}/_bmad/custom/mc-braindump.toml (team) -# {project-root}/_bmad/custom/mc-braindump.user.toml (personal) -# -# Studio-wide config (owner, paths, editor, transcription, assets, tools) -# lives in [modules.manticore] in {project-root}/_bmad/custom/config.toml, -# maintained by mc-setup. This file holds only this skill's own surface. - -[workflow] - -# Steps to run before / after the standard activation. -activation_steps_prepend = [] -activation_steps_append = [] - -# Persistent facts held for the whole run: literal sentences, or -# "file:{project-root}/..." paths whose contents are loaded as facts. -persistent_facts = [] diff --git a/skills/mc-cut/SKILL.md b/skills/mc-cut/SKILL.md index eccca59..6be17e9 100644 --- a/skills/mc-cut/SKILL.md +++ b/skills/mc-cut/SKILL.md @@ -1,57 +1,133 @@ --- name: mc-cut -description: Turn raw takes into a cut plan, edl.json, a rendered preview, and an editor timeline export. Presents the taste calls for gate 2 approval and renders the preview after every approval; owns the offered final render at gate 4. Use at the cut stage once recordings are in raw/. +description: Cut raw takes into an approved, rendered edit. Use at the cut stage with recordings in raw/, or when the user says "cut the takes", "make the cutplan", "render the preview", or "render the final". --- # mc-cut -The Descript replacement, render-first. Every approved cut iteration ends in a watchable preview render; once the graphics stage has rendered overlays, the preview re-renders with them composited; at gate 4 a final-quality render is offered. The editor timeline export and all assets (cutplan.md, edl.json, overlays) are ALWAYS produced alongside, so the creator can move into their editor at any step. +Act as the creator's editor. The outcome is an approved cut: `cut/edl.json`, plus the cutplan, editorial review, preview render and editor timeline built from it. -## Steps +Three consumers set the bar. The creator at gate 2 must be able to accept or reject every call without re-watching the raw footage. Their editor must import the timeline in sync. mc-beats builds visuals on the edited transcript and the editorial review. This stage owns gate 2 on the cutplan and the offered final render at gate 4. -1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there) and this skill's own surface (`uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}`; run `{workflow.activation_steps_prepend}` now, `{workflow.activation_steps_append}` after this step, and hold `{workflow.persistent_facts}` as standing context). Resolve `paths` values against `{project-root}`. Read `project.json` (stage `cut`; the composited preview re-render after graphics and the offered final render at gate 4 are the two routed entry points that legitimately run at later stages, see their sections below), `script.md`, `{brand-path}/production-bible.md` when it exists (the taste contract for the judgment calls in step 4), and the cutting rules below. -2. Preflight every file in `raw/`: `uv run {skill-root}/scripts/preflight.py raw/ [...] --remux --qc-frames cut/qc/`. Three checks, all before any transcription or render: - - Frame rate: VFR sources are re-encoded to constant frame rate (run the preflight in the background and keep working; transcription waits for it). Record the reported `cfr_master` path in project.json `sources` as the project source of truth; every later step (transcription, EDL times, renders, timeline export) uses the CFR master, never the VFR original. - - Disk: free space is checked against a rough estimate (3x source size plus the estimated CFR masters) before any remux write, and the script refuses the remux itself when the estimate does not fit; if `disk.ok` is false in the summary, stop and tell the creator before any render. - - Source QC: inspect the extracted first and last frames per source for edge defects (black edges, wrong aspect, letterboxed or cropped content) before any render is built on them. -3. Transcribe each take: `uv run {skill-root}/scripts/transcribe.py raw/ -o transcript/words.json --provider <[transcription] provider from the config>` (suffix the output `.words.json` when the project has multiple sources). Provider values: `auto` (the default) picks parakeet-mlx on macOS Apple Silicon and onnx-asr everywhere else; `parakeet-mlx` and `onnx-asr` force a lane. Both lanes run the same parakeet-tdt-0.6b-v3 weights, so verbatim fillers and 80 ms word timestamps carry over on Windows, Linux, and Intel Macs; on the onnx-asr lane, when the runtime exposes no per-token scores every word's confidence reads 1.0 (no signal, never fabricated). On a CUDA machine escalate with `uv run --with "onnx-asr[gpu,hub]" python {skill-root}/scripts/transcribe.py ...` (PEP 508 markers cannot detect GPUs; the `python` command is required because it skips the script's cpu-extra dependency, so onnxruntime-gpu never co-installs with onnxruntime, and the script warns on stderr when an NVIDIA GPU is visible but CUDA is unavailable). All lanes are local and free; the model downloads once on first run. -4. Candidates: `uv run {skill-root}/scripts/cutplan.py transcript/words.json -o cut/candidates.json` plus any `{workflow.cutplan_flags}` finds silences, filler runs, stutters, and retakes mechanically. On an `interview` source (project.json `sources`), it also flags each spoken interviewer-question read as a `marker` candidate: cut the marker and question, keep the answer. The default marker cue is "question from the interviewer"; projects recorded against the older "question from claude" convention pass `--marker-cues "question from claude"` (via `{workflow.cutplan_flags}`). -5. Make the taste calls: against `script.md` and the Production Bible, pick best takes, order segments, decide keep-or-cut on every candidate in `cut/candidates.json`. Write `cut/cutplan.md` as a short human-readable plan whose spine is the judgment calls, each with a timestamp and the quoted words (the "trailing 'so' at 42:20, keep or cut?" shape). Group the obvious silence trims into one line; itemize only what the creator might disagree with. -6. Write `cut/edl.json`: `{source, source_duration, fade_ms: 30, pad_ms: 60, segments: [...]}` with ordered segments of {source, start, end, beat, quote, reason} obeying the cutting rules below. -7. Set `approvals.cutplan = "pending"`, present cutplan.md, and STOP for gate 2. -8. After approval, and again after every later re-approval that changes the cut: - - Render the preview, always: `uv run {skill-root}/scripts/render_preview.py cut/edl.json -o renders/preview.mp4 --boundary-frames cut/boundaries/` plus any `{workflow.preview_flags}` (720p CRF 28 defaults; pass the `[render]` preview keys from the studio config when set). Inspect the boundary frames per the cutting rules. - - Once the graphics stage has rendered overlays into `graphics/`, re-render composited so the creator iterates on overlays and CTAs visually: same command plus `--beats beats/beats.md --graphics-dir graphics/`. Report any `overlays_missing` from the summary. - - Export the editor timeline, always, per `[editor] timeline-format` in the config: `fcpxml` via `uv run {skill-root}/scripts/edl_to_fcpxml.py cut/edl.json -o cut/rough.fcpxml` (Resolve and Final Cut import it natively; refuses VFR sources loudly); `xmeml`/`edl` are planned lanes, so Premiere users work from cutplan.md + edl.json + the rendered preview/final until the xmeml lane lands (see TODO); `none` (Descript and manual workflows) skips export, and the deliverables are cutplan.md + edl.json + renders/preview.mp4 as the cut map. - - resolve_import.py (push the timeline into a running Resolve) is currently a stub: do NOT offer it. When its STATUS line says implemented, offer it only if `[mcp] davinci-resolve` is true in the config. Free-edition note for when it lands: Resolve's external scripting API is Studio-only, but the free edition runs scripts launched from inside the app (Workspace > Scripts), so copying the script into Resolve's Fusion Scripts folder unlocks scripted import there; the per-OS folder paths are documented in the setup stack reference for the creator's platform. - - Record the ISO date in `approvals.cutplan`, append `cut` to `stages_done`, and set `stage` to the next entry in project.json's `stages` array. +## The rule that is not inferable -## Composited preview (after graphics) +The TRANSCRIPT is the authority on CONTENT (which words, in what order). The AUDIO is the authority on TIMING (where silence is, and therefore where a cut is safe). Never derive a cut time, a beat time, or a silence from transcript timestamps: parakeet absorbs pauses into the preceding word's end, so transcript gaps read about 0.0 across real dead air and word ends reach past the sound. -mc-pipeline routes here as soon as the graphics stage completes (mc-graphics hands back after writing `graphics/HANDOFF.md`), and again whenever an overlay in `graphics/` is later re-rendered. This entry point runs after the `cut` stage and touches no gates, approvals, or stage fields: re-render the preview composited, `uv run {skill-root}/scripts/render_preview.py cut/edl.json -o renders/preview.mp4 --boundary-frames cut/boundaries/ --beats beats/beats.md --graphics-dir graphics/` plus any `{workflow.preview_flags}`, report any `overlays_missing` from the summary (each one is a beat whose overlay has not landed in `graphics/`), present the composited preview to the creator, and stop. +Everything else in this stage follows from that, and from one convention: a check this stage claims to perform is a script that exits non-zero. -## Final render (gate 4) +## Resolution rules -When the project reaches the final stage, offer the final-quality render from this skill: `uv run {skill-root}/scripts/render_final.py cut/edl.json -o renders/final.mp4 --beats beats/beats.md --graphics-dir graphics/` plus any `{workflow.final_flags}`, with `--codec` and `--crf` per `[render]` in the studio config, `--height` from the height of `[video]` delivery-resolution, `--loudness-target <[render] loudness-target>`, and `--no-loudnorm` appended when `[render] loudnorm` is false. It bakes the same EDL the creator approved with graphics composited from the approved beat table, hardware encode when available (videotoolbox on macOS; on Windows h264_nvenc, then h264_qsv, then h264_amf; on Linux h264_nvenc then h264_vaapi; each candidate validated by a one-frame test encode, libx264 fallback everywhere), persistent incremental segment rendering (the timeline is partitioned into content-addressed segments under `renders/segments/`; a re-render re-encodes only the segments whose inputs actually changed and reuses the rest, so a single tweaked graphic on a long video is a seconds-long re-render), a disk preflight, progress reporting, and boundary-frame checks. `--segment-target-seconds` (default 600) tunes the segment size; append it via `{workflow.final_flags}` when a project wants coarser or finer segments. When `[render] loudnorm` is enabled (the default), the final render is loudness-normalized to the target LUFS with two-pass ffmpeg loudnorm; `--no-loudnorm` turns it off, and the fast preview is never normalized. Finishing in the creator's own editor from the always-exported timeline is an equally supported path; either closes gate 4. +- Bare paths resolve against `{video-path}`, the current video project at `{projects-path}//`. +- `{skill-root}` → this skill's installed directory; files in it always carry it (`{skill-root}/references/rendering.md`). +- `{project-root}` → the project working directory. -## Dual timecode +## On Activation -Chapters or event notes written against original source timecode (a livestream VOD chapter list, log notes) remap onto the edited timeline with `uv run {skill-root}/scripts/remap_timecode.py cut/edl.json --direction orig-to-clean --chapters -o `. The same utility maps clean times back to source timecode (`--direction clean-to-orig`); mc-package carries its own duplicate of it (per the script-duplication convention) for the dual-timeline chapters deliverable whenever an EDL exists. +1. Load the studio config: `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`. Empty means mc-setup has not run; stop and route the creator there. Resolve `paths` values against `{project-root}`. +2. Read `project.json` (stage `cut`) and `script.md`. +3. Read `{brand-path}/production-bible.md`. If it does not exist, tell the creator it is missing and that judging the cut against their taste cannot happen without it, then route to mc-setup and stop. The Production Bible is the taste contract for the calls you make below. +4. Read `{brand-path}/voice-bible.md`. If it does not exist, tell the creator it is missing and that cadence-aware filler detection cannot happen without it, then route to mc-setup and stop. + +## Prepare the sources + +Every source in `raw/` passes preflight before anything reads it: + +``` +uv run {skill-root}/scripts/preflight.py raw/ [...] --remux --qc-frames cut/qc/ +``` + +It is slow, so run it in the background and let transcription wait on it. Record the reported `cfr_master` in `project.json` `sources`; every later step reads that path, never the VFR original, because the two have different frame timing and the desync only shows up once the creator scrubs the timeline. + +Exit 3 is source QC failing, and it is a hard stop: do not transcribe, cut, or render against it. A false `disk.ok` is also a stop. For either, and for the spatial fix, load `{skill-root}/references/source-prep.md`. + +## Transcribe and verify + +Needs the CFR master from the previous section. + +Pick the lane from `{skill-root}/references/transcription.md` (published sources take captions, not local ASR), then build the audio map and prove the transcript: + +``` +uv run {skill-root}/scripts/analyze_audio.py raw/ -o cut/audio-map.json --noise <[cut] silence-floor-db> +uv run {skill-root}/scripts/verify_transcript.py transcript/words.json --audio-map cut/audio-map.json --wpm <[owner] wpm> -o cut/transcript-check.json +``` + +The audio map is the timing source of truth for the whole stage, built once per source. + +A non-zero exit from `verify_transcript.py` is a HARD STOP: it finds audio above the silence floor that produced no words and names the regions. Nothing may be built on a transcript that has not passed, because downstream a hole in the transcript looks exactly like dead air and the cut deletes real content. `{skill-root}/references/transcription.md` carries the override for a region the creator has listened to and confirmed. + +Every lane windows in 20s isolated windows with 3s overlap. This is not a tuning knob, and never raise `--window` to go faster: parakeet drops whole paragraphs inside long windows with no error at all. Measured: 120s chunks lost three paragraphs, 90s still lost content, 20s was complete. + +## Propose the cut + +Needs a passing transcript and the audio map. This section ends at gate 2. + +``` +uv run {skill-root}/scripts/cutplan.py transcript/words.json --audio-map cut/audio-map.json --voice-bible {brand-path}/voice-bible.md -o cut/candidates.json +``` + +Plus `[cut] cutplan-flags` from the studio config. It finds the mechanical candidates and snaps each edge into an audio-verified silence. Two things it does that are easy to undo by accident: the voice bible's `cadence` block marks the connective words that are the creator's rhythm, so those are keeps and not filler; and on an `interview` source each spoken interviewer question becomes a `marker` candidate, where the marker and question go and the answer stays. Anything reported `unsnapped` never reached a silence and needs an ear. + +Judge the candidates against `script.md` and the Production Bible, then write `cut/edl.json` as `{source, source_duration, fade_ms: 30, pad_ms: 60, segments: [...]}` with ordered segments of `{source, start, end, beat, quote, reason}`. Prove it: + +``` +uv run {skill-root}/scripts/verify_edl.py cut/edl.json --audio-map cut/audio-map.json --words transcript/words.json -o cut/edl-check.json +``` + +A non-zero exit is a HARD STOP: it fails any boundary not resting in an audio-verified silence, and any segment missing its quote or reason. Re-run it after every EDL change. + +Then reconstruct what the viewer will actually hear, and read it as an argument: + +``` +uv run {skill-root}/scripts/edited_transcript.py transcript/words.json --edl cut/edl.json -o cut/edited-transcript.md -j cut/edited-words.json +``` + +Both clean and source timecodes come from here; never convert between them by hand. Run the editorial pass on that transcript per `{skill-root}/references/editorial-pass.md`, writing `cut/editorial-review.md` from `{skill-root}/assets/editorial-review-template.md`. Nothing it recommends is auto-applied. RE-RECORD items are the one exception to "the cut applies the calls": there is no pickup re-entry path, so they hand over as a shoot list and the cut proceeds without them. + +Write `cut/cutplan.md` carrying both tiers, each call with its timestamp and the quoted words. Routine silence trims group into one line. Always itemize section re-reads, bloopers and every content-tier recommendation, whatever their size. Set `approvals.cutplan = "pending"`, present it, and STOP for gate 2. + +## Apply the approved calls + +Write the creator's decisions to `cut/approved-spans.json` as `{start, end, quote, reason}`, with times re-detected against the audio, because transcript-read timecodes drift and a blind apply cuts the wrong spans. Then snap them mechanically. Never snap an edge by hand; an eyeballed snap is how a cut lands inside a word. + +``` +uv run {skill-root}/scripts/snap_spans.py cut/approved-spans.json --audio-map cut/audio-map.json -o cut/snapped-spans.json +``` + +Back up the prior EDL to `cut/edl.pre-editorial.json`, rewrite `cut/edl.json` from the snapped spans, re-run `verify_edl.py`, and append the APPLIED section to `cut/editorial-review.md`. + +## Deliver + +After approval, and again after every later re-approval that changes the cut: render the preview, export the timeline, and regenerate every other derived artifact together. `{skill-root}/references/rendering.md` carries the commands, the config wiring and the staleness check. Inspect the boundary frames for what they can see, black frames and straddles, up to 3 retries per cut. They see less than they appear to: on the corrupted project every frame looked clean while the cut underneath was built on the hole. + +Chapters or log notes written against source timecode remap onto the edited timeline with `uv run {skill-root}/scripts/remap_timecode.py cut/edl.json --direction orig-to-clean --chapters -o `, and `--direction clean-to-orig` maps back. + +Record the ISO date in `approvals.cutplan`, append `cut` to `stages_done`, and set `stage` to the next entry in project.json's `stages` array. + +## Routed re-entries + +Two entry points run after the `cut` stage has closed. Both touch no gates, approvals or stage fields. + +Composited preview: mc-pipeline routes here once mc-graphics writes `graphics/HANDOFF.md`, and again whenever an overlay is re-rendered. Re-render the preview composited per `{skill-root}/references/rendering.md`, report any `overlays_missing`, present it, and stop. + +Final render: when the project reaches the final stage, offer the final-quality render per `{skill-root}/references/rendering.md`. Finishing in the creator's own editor from the exported timeline is an equally supported path; either closes gate 4. ## Cutting rules (non-negotiable) -- Never cut inside a word. Pad cut edges 30 to 200 ms. -- 30 ms audio fades on every cut boundary. -- Every EDL segment records: source, start, end, the quoted words, and the reason for the cut. -- Self-verify: extract frames at each cut boundary and inspect; up to 3 retries per cut. -- Raw recordings must be constant frame rate (mitigates FCPXML desync). The step 2 preflight catches and remuxes VFR before transcription; never cut against a VFR file. -- Never shrink or letterbox the source video to make room for graphics; overlays composite over the full frame in safe zones. - -## Checklist - -- No cut lands inside a word (check edl times against word timestamps). -- Every edl segment has quote + reason. -- Preflight ran on every source; any VFR file was remuxed and its CFR master recorded in project.json `sources`. -- QC frames inspected for edge defects; disk preflight passed before rendering. -- renders/preview.mp4 watched (spot-check at minimum three boundaries) before declaring done; after the graphics render, the composited preview re-checked and `overlays_missing` is empty or explained. -- FCPXML sync verified in the editor on the first project this converter touches. +- Never cut inside a word. Edges land inside an audio-verified silence, which makes this structural rather than aspirational. Pad 30 to 200 ms. +- 30 ms audio fades on every cut boundary (`fade_ms` in the EDL). +- Never shrink or letterbox the source video to make room for graphics; overlays composite over the full frame in safe zones. Nothing enforces this one, and the beats and graphics stages inherit whatever canvas this stage leaves them. + +## Gates + +| Gate | Script | Where | +|---|---|---| +| Source QC | `preflight.py` (exit 3) | Prepare the sources | +| Transcript completeness | `verify_transcript.py` | Transcribe and verify | +| Cut integrity | `verify_edl.py` | Propose the cut, and again after applying | +| Output integrity | `render_preview.py` / `render_final.py` | Deliver | + +Three things no script can check, so they are on you: + +- Listen to the joins in the preview. A clipped word onset is inaudible in a still, and boundary frames cannot hear it. +- Check any span reported `unsnapped` by ear before it goes into the EDL. +- Verify FCPXML sync in the editor on the first project this converter touches. diff --git a/skills/mc-cut/assets/editorial-review-template.md b/skills/mc-cut/assets/editorial-review-template.md new file mode 100644 index 0000000..3c8b0db --- /dev/null +++ b/skills/mc-cut/assets/editorial-review-template.md @@ -0,0 +1,100 @@ +# Editorial Review: the content pass + +Template for `cut/editorial-review.md`. The spec is `{skill-root}/references/editorial-pass.md`; this is the shape of its output. Replace every angle-bracket placeholder and delete any section that has no findings rather than leaving it empty. + +Written for the creator to read at gate 2 alongside `cut/cutplan.md`. Nothing in it has been applied. + +--- + +# Editorial Review + +Project: `` +Run on: the EDITED transcript (post mechanical cut), `` words, `` runtime, reconstructed from `words.json` intersected with the kept segments of `cut/edl.json`. +Judged against: `brief.md` (the goal), `script.md` (the intent), and `{brand-path}/voice-bible.md`. +Date: `` + +## How to read this + +This is the subtractive pass. It can only recommend four things, and nothing is applied until you say so: + +- CUT, remove a spoken span, with a seam check that it still flows. +- RE-RECORD, a pickup; the only way to fix or add load-bearing content. +- HAND TO BEATS, do not cut, solve it with a visual. Seeds the beat table. +- GENERATE, synthetic-voice fill. Consent-gated, tiny bridges only. + +Read this caveat first. This pass read the transcript TEXT and cannot hear the audio. Idea-level findings (structure, redundancy, logic, ordering) are high confidence and survive transcription noise. Word-level findings are marked VERIFY AUDIO: the transcript shows a stumble, but it may be a transcription artifact rather than something you actually said. Check those against the audio before cutting. Anything high-stakes (a name, a URL, a number) is listed first. + +Times are given as `clean / src`. Clean time is what the preview shows. Source time is what the EDL edits. Both come from `cut/edited-words.json`; neither was converted by hand. + +## Verdict + +`` + +## Priority items + +`` + +P1, ``. [CUT | RE-RECORD | HAND TO BEATS | VERIFY AUDIO] +`` / src ``, quoting: "``" +`` +Seam if cut: "`<...end of the preceding span>` → ``" +Recommendation: `` + +## Findings by category + +### Redundancy + +`` + +### Off-goal and pacing + +`` + +### Contradiction and confusion + +`` + +### Errors and misstatements + +`` + +### Leftover stumbles the mechanical pass missed + +`` + +## Hand to beats + +`` + +## Re-record pickup list + +`` + +## Your decision checklist + +Verify against audio first: +- [ ] `` + +Content calls, your taste: +- [ ] `` + +Route to beats, no cut: +- [ ] Approve the hand-to-beats list to seed the beat table. + +Nothing here has been applied. Tell me which items you want and I will apply the cuts (re-detecting each span against the audio and snapping its edges into silences before touching the EDL), assemble any pickup list, and carry the hand-to-beats items into the beats stage. + +--- + +## APPLIED, `` (``'s calls: ``) + +`` + +Backup of the pre-editorial cut: `cut/edl.pre-editorial.json`. Runtime `` → `` (``s removed, `` → `` segments). All cuts whole-word and silence-anchored. + +Verification note: `` + +Applied: +- `` + +Deliberately NOT applied: +- `` diff --git a/skills/mc-cut/customize.toml b/skills/mc-cut/customize.toml deleted file mode 100644 index f8d8f5e..0000000 --- a/skills/mc-cut/customize.toml +++ /dev/null @@ -1,35 +0,0 @@ -# DO NOT EDIT -- overwritten on every update. -# -# Customization surface for mc-cut. -# Override files (not edited here): -# {project-root}/_bmad/custom/mc-cut.toml (team) -# {project-root}/_bmad/custom/mc-cut.user.toml (personal) -# -# Studio-wide config (owner, paths, editor, transcription, assets, tools) -# lives in [modules.manticore] in {project-root}/_bmad/custom/config.toml, -# maintained by mc-setup. This file holds only this skill's own surface. - -[workflow] - -# Steps to run before / after the standard activation. -activation_steps_prepend = [] -activation_steps_append = [] - -# Persistent facts held for the whole run: literal sentences, or -# "file:{project-root}/..." paths whose contents are loaded as facts. -persistent_facts = [] - -# Extra flags appended to the cutplan.py invocation, e.g. -# "--min-silence 0.5" or a --marker-cues override for interview sources -# (default marker cue "question from the interviewer"; pass -# --marker-cues "question from claude" for projects recorded against the -# older convention). mc-setup's marker-cue interview question records a -# non-default cue here, in the team override file. Empty means the script -# defaults (min-silence 0.7, retake window 16, run 3). -cutplan_flags = "" - -# Extra flags appended to the render_preview.py / render_final.py -# invocations, e.g. "--height 540" (preview) or "--parallel 4 --crf 16" -# (final). Empty means the script defaults. -preview_flags = "" -final_flags = "" diff --git a/skills/mc-cut/references/editorial-pass.md b/skills/mc-cut/references/editorial-pass.md new file mode 100644 index 0000000..49dd9c6 --- /dev/null +++ b/skills/mc-cut/references/editorial-pass.md @@ -0,0 +1,88 @@ +# The editorial pass + +The cut stage has two tiers. The mechanical tier (`cutplan.py`) makes the take clean: dead air, stutters, fillers, redos, bloopers. This pass makes the video better. It reads the delivered piece as an argument and asks the question no mechanical detector can: should this section exist at all? + +Both tiers feed one gate. The creator approves the mechanical trims and the content calls together at gate 2. + +## The constraint that defines it + +You can only subtract. + +This is not editing a document. The words are recorded audio in the creator's voice, so there is no rewriting a clumsy sentence and no adding a clarifying clause. Every recommendation resolves to one of five operations: + +| Operation | What it means | When | +|---|---|---| +| CUT | Remove a spoken span | Redundant, off-goal, or clearly weak, and the seam still reads | +| RE-RECORD | Flag a creator pickup | Load-bearing but wrong; the only true way to fix or add | +| HAND TO BEATS | Do not cut; solve it visually | Thin or confusing but fixable with a diagram, overlay, or punch-in | +| GENERATE | Synthetic-voice fill, consent-gated | Tiny bridges only, and never without explicit consent; it fabricates the creator's voice | +| REORDER | Move a self-contained span | Propose only, never auto-apply; risk scales with format (see below) | + +Because it is subtractive, every proposed cut carries a seam check: the confirmation that the audio still reads grammatically and logically once the span is gone. You cannot rewrite the connective tissue, only remove whole spoken spans, so a cut that leaves a broken join is not a cut you can make. That seam constraint is exactly what separates this from prose editing. + +## When it runs, and why the placement is load-bearing + +Inside the cut stage, after the mechanical EDL is assembled, before gate 2. + +It reads the EDITED transcript, `cut/edited-transcript.md` from `edited_transcript.py`, which is `words.json` intersected with the kept EDL segments. Not the script, and not the raw transcript: + +- The script is what was planned. Delivery diverges from it: ad-libs, dropped lines, live rewrites, and the strongest findings are often about ad-libbed content that was never in the script at all. +- The raw transcript still contains everything the mechanical pass removed, so a pass reading it reviews words the viewer never hears. + +Content is cheapest to change before you decorate. If this pass recommends cutting a section and beats has already built an overlay and a punch-in for it, that visual work is dead and beats must be re-planned. Settle what stays before spending effort on how it looks. Re-record flags loop back through the cut, and beats plans against the final cut, so pickups must land first. + +This pass is never a separate stage and never adds a fifth gate: four gates is a settled invariant, and folding the pass in front of gate 2 removes rework rather than adding a stop. Approving gate 2 before this pass runs forces a rebuild of the approved EDL. + +## What it hunts for + +1. Redundancy. A point already made, restated with no new value. Conceptual, not verbatim: the mechanical retake detector cannot see it because the words differ. Respect deliberate repetition. Callbacks, rule of three, anadiplosis, and the intentional emphasis the script or voice bible calls for are craft, not accident. Cut accidental repetition, keep rhetorical repetition. +2. Off-goal or pacing drag. Stretches that do not serve the promised payoff: tangents, over-explanation, throat-clearing paragraphs. The test is "does this earn its runtime for this video's goal?", judged against `brief.md`. +3. Contradiction or confusion. Statements that conflict, a claim later walked back, an ambiguous antecedent, a muddied argument. Open loops belong here: a promise the video sets up and never pays off is a structural defect, and it has two clean fixes, cut the setup or re-record the payoff. +4. Errors and misstatements. Factually wrong or misspoken content. Two outcomes: cuttable (remove it and the piece still stands) or load-bearing (the point matters but is wrong), which is a re-record. + +## The rule that outranks all of the above + +Never apply a finding from a transcript-read timecode. Re-detect the span against the audio first. + +The pass produces good findings and bad coordinates: source-timecode estimates read off a transcript drift to the wrong windows, so a blind apply cuts the wrong spans. Re-detecting each span by pattern against the kept audio before touching the EDL is what catches it. + +So: + +- Quote findings by CLEAN time for the human (that is what the preview shows), and apply them by SOURCE time against the EDL. `edited_transcript.py` emits both for every word so neither is ever derived by hand. +- Before applying any span, re-locate it in `cut/edited-words.json` by its words, then snap the resulting cut edges into audio silences from `cut/audio-map.json`, exactly as the mechanical tier does. A cut edge that cannot reach a silence needs an ear before it is applied. +- Word-level findings are lower confidence than idea-level ones and must be marked. You are reading a transcript, not hearing audio: a doubled word in the text may be a transcription artifact rather than a real stumble, and acting on one triggers a pointless re-record. Idea-level findings survive transcription noise; word-level findings do not. + +## Reuse the reviewers that already ship + +BMad already has the engines this needs. Adapt them, constrain them to subtract-only, and map their findings onto the five operations. Invoke them as skills; never read into another skill's folder. + +- `bmad-editorial-review-structure` proposes cuts, reorganization, and simplification while preserving comprehension. The core engine for redundancy, drag, and ordering. +- `bmad-review-adversarial-general` and `bmad-review-edge-case-hunter` for contradiction, confusion, and error hunting. +- `bmad-editorial-review-prose` for sentence-level clarity, which here means "does the seam read", not "is this the best phrasing". + +## Output + +`cut/editorial-review.md`, from the template in `{skill-root}/assets/editorial-review-template.md`. A recommendation list for the human at the gate. Nothing is auto-applied, ever. + +Every item carries: the span (clean and source timecodes plus the quote), the type, a severity, the reasoning, the proposed operation, a seam check for anything cuttable, and an explicit confidence marker on word-level findings. + +Fold the itemized calls into `cut/cutplan.md` so gate 2 presents one list: the mechanical judgment calls and the content calls together, with routine trims grouped into a line and everything the creator might disagree with itemized. + +## Two extensions, on request only + +Both still honor "select spoken spans, never fabricate words". Neither is part of the default pass; offer them, do not run them. + +- Resequencing. Because the pass understands the argument, it can propose moving a self-contained segment for better flow. Risk scales with format: low on loose long-form and livestream VODs where segments are modular, high on a tight talking-head where continuity, jump cuts, and eyeline all break. Offer freely on `livestream-vod`, propose cautiously on `talking-head`, never auto-apply. +- Highlight and sizzle mining. The same content ranking that finds redundancy finds the best moments: hot takes, quotable lines, energy peaks. It can assemble a cold-open sizzle or a highlights reel, which is a large win on a 1.5 hour VOD, and it is still subtractive. Natural consumers are the video's own cold open, mc-package, and short-form derivatives. + +## Checklist + +- The pass read `cut/edited-transcript.md`, not `script.md` and not the raw transcript. +- Every finding carries both clean and source timecodes, taken from the emitted transcript rather than converted by hand. +- Every word-level finding is marked as needing audio confirmation; idea-level findings are not. +- Every proposed cut has a seam check quoting the join it would create. +- Deliberate repetition (callbacks, rule of three, the script's own emphasis) was checked before anything was called redundant. +- Re-record flags are collected into a pickup list the creator can shoot from. +- Hand-to-beats items are listed separately, ready to seed the beat table. +- Nothing was applied. The document is a recommendation list, and the creator's calls at gate 2 decide what happens. +- When calls are applied, each span was re-detected against the audio and its edges snapped into silences before the EDL was touched. diff --git a/skills/mc-cut/references/rendering.md b/skills/mc-cut/references/rendering.md new file mode 100644 index 0000000..89dc361 --- /dev/null +++ b/skills/mc-cut/references/rendering.md @@ -0,0 +1,88 @@ +# Rendering and export + +The preview render, the gate-4 final render, and the editor timeline export. +Load this when rendering or exporting. Full flag lists live in each script's +docstring; this file carries the wiring and the decisions. + +## Where encoder settings live + +The studio config is their single home. Emit the `[render]` and `[video]` keys +first, then `[cut] preview-flags` or `[cut] final-flags` last. + +Because the last occurrence wins, restating a key the config owns inside a flags +string silently overrides it, and which value applies then depends on emission +order rather than on any stated rule. So the flags strings are an escape hatch +for what the config does not model, such as `--segment-target-seconds`. Never +put `--height`, `--crf`, or the loudness flags in them. + +## Preview + +``` +uv run {skill-root}/scripts/render_preview.py cut/edl.json \ + -o renders/preview.mp4 --boundary-frames cut/boundaries/ \ + plus [cut] preview-flags +``` + +Defaults to 720p CRF 28 when `[render]` leaves them unset. Never +loudness-normalized. Check `"validated": true` in the summary. + +Composited, once the graphics stage has rendered overlays into `graphics/`, add: + +``` +--beats beats/beats.md --graphics-dir graphics/ +``` + +Report any `overlays_missing` from the summary; each one is a beat whose overlay +has not landed in `graphics/`. + +## Final render, offered at gate 4 + +``` +uv run {skill-root}/scripts/render_final.py cut/edl.json \ + -o renders/final.mp4 --beats beats/beats.md --graphics-dir graphics/ \ + plus [cut] final-flags +``` + +Wire the config in: `--codec` and `--crf` from `[render]`, `--height` from the +height of `[video]` delivery-resolution, `--loudness-target` from `[render] +loudness-target`, and append `--no-loudnorm` when `[render] loudnorm` is false. + +It bakes the same EDL the creator approved, with graphics composited from the +approved beat table. Finishing in the creator's own editor from the always +exported timeline is an equally supported path; either closes gate 4. + +What the script handles without instruction: hardware encode selection with a +one-frame test encode per candidate and an libx264 fallback, a disk preflight, +progress reporting, boundary-frame checks, and two-pass loudnorm when enabled. + +Persistent incremental segments are worth knowing about, because they change +what a re-render costs. The timeline is partitioned into content-addressed +segments under `renders/segments/`, and a re-render re-encodes only the segments +whose inputs actually changed. A single tweaked graphic on a long video is a +seconds-long re-render, so re-rendering after a small fix is cheap and there is +no reason to batch changes to avoid it. `--segment-target-seconds` (default 600) +tunes segment size. + +## Editor timeline export + +Always exported, per `[editor] timeline-format`. + +| Value | Command | Notes | +|---|---|---| +| `fcpxml` | `uv run {skill-root}/scripts/edl_to_fcpxml.py cut/edl.json -o cut/rough.fcpxml` | Resolve and Final Cut import it natively. Refuses VFR sources loudly | +| `xmeml`, `edl` | Not yet implemented | Premiere users work from cutplan.md, edl.json and the rendered preview. See TODO.md | +| `none` | Skipped | Descript and manual workflows. The deliverables are cutplan.md, edl.json and renders/preview.mp4 as the cut map | + +`{skill-root}/scripts/resolve_import.py` is a stub. Do not offer it. It is named here only +because the file exists and reads as usable; when its STATUS line says +implemented, offer it only where `[mcp] davinci-resolve` is true. + +## Derived artifacts drift silently + +The EDL is the single source of truth. The preview, the boundary frames, the +timeline export and cutplan.md are all derived from it, and nothing stops an +older one sitting next to a newer EDL looking current. + +The preview writes an `.key` sidecar naming the render identity it was +built from. When that key does not match the current EDL's, the derived set is +stale and all of it regenerates together. diff --git a/skills/mc-cut/references/source-prep.md b/skills/mc-cut/references/source-prep.md new file mode 100644 index 0000000..fda83c6 --- /dev/null +++ b/skills/mc-cut/references/source-prep.md @@ -0,0 +1,63 @@ +# Source preparation + +Everything between a file landing in `raw/` and it being safe to transcribe or +cut. Load this when preflight reports a problem, or when a source needs a +spatial fix. + +`uv run {skill-root}/scripts/preflight.py raw/ [...] --remux --qc-frames cut/qc/` + +Run it on every source before anything else. It is slow on large files, so start +it in the background and let transcription wait on it. + +## What it checks + +| Check | Failure | What to do | +|---|---|---| +| Frame rate | VFR source | `--remux` re-encodes to constant frame rate and reports a `cfr_master` path | +| Disk | `disk.ok` false | Stop and tell the creator. The script refuses the remux itself rather than filling the disk | +| Source QC | exit 3 | Hard stop. See below | +| Duration and streams | probe failure | The file is unusable; get another export | + +The disk estimate is rough (about 3x source size plus the estimated CFR +masters). It is a floor, not a guarantee, so a render can still run tight on a +nearly full volume. + +## The CFR master replaces the original + +Record the reported `cfr_master` in `project.json` `sources`. Every later step +(transcription, EDL times, renders, timeline export) reads that path, never the +VFR original. The two files have different frame timing, so mixing them +desynchronizes the FCPXML export in a way that looks fine until the creator +scrubs the timeline in their editor. + +## Source QC exit 3 + +The script samples frames across the take and asserts on them, exiting 3 on a +flat decorative border ring, or an active area whose aspect does not match the +container. Report the inferred active-content rectangle and get the creator's +call. Do not transcribe, cut, or render against a source that failed QC. + +Two ways forward, and they are different decisions: + +The framing is a defect. Crop it out: + +``` +uv run {skill-root}/scripts/normalize_source.py raw/ \ + -o raw/-normalized.mp4 --auto \ + [--offset-x N] [--target-aspect 16:9] [--output-size WxH] +``` + +Register the corrected file in `project.json` `sources` as the new +`cfr_master`. This must happen before the beats and graphics stages, because +overlays are positioned against the canvas. + +A spatial crop moves nothing in time. The script asserts the duration is +unchanged and refuses to publish otherwise, so an existing transcript, EDL, +cutplan and beat table all stay valid. Do not re-transcribe and do not re-cut +after a normalize. + +The framing is intentional. Re-run preflight with `--allow-qc-defects`. This is +a recorded override, so get the creator's confirmation first. + +Keep both distinct from creative reframing (punch-ins, motion zooms), which +belongs to the beats stage working on an already-clean canvas. diff --git a/skills/mc-cut/references/transcription.md b/skills/mc-cut/references/transcription.md new file mode 100644 index 0000000..7f9e399 --- /dev/null +++ b/skills/mc-cut/references/transcription.md @@ -0,0 +1,75 @@ +# Transcription lanes + +Which transcriber to run and how. Load this when choosing a lane, when the +source is already published, or when the transcript gate reports dropped +speech. + +The windowing invariant that governs every lane lives in SKILL.md, because it +looks like a tuning knob and is not. + +## Choosing the lane + +The source is already published (a livestream VOD, a footage-first project, +anything with captions on YouTube). Pull the captions: + +``` +yt-dlp --write-auto-subs +``` + +Free, effectively perfect, any length, and it avoids local ASR entirely. Local +ASR is only for raw unpublished recordings, which is exactly where the +transcription bugs lived and why they went unnoticed for so long. Record the +provenance in the transcript header either way. + +The source is a raw recording. Run local ASR: + +``` +uv run {skill-root}/scripts/transcribe.py raw/ \ + -o transcript/words.json --provider <[transcription] provider> +``` + +Suffix the output `.words.json` when the project has multiple +sources. + +| Provider | Behaviour | +|---|---| +| `auto` | The default. parakeet-mlx on macOS Apple Silicon, onnx-asr everywhere else | +| `parakeet-mlx` | Forces the MLX lane | +| `onnx-asr` | Forces the ONNX lane | +| `elevenlabs-scribe` | Metered, opt-in behind `[transcription]`. Sends audio to a third party, so never a default | + +Both local lanes run the same parakeet-tdt-0.6b-v3 weights and window +identically. All local lanes are free; the model downloads once on first run. + +## CUDA machines + +``` +uv run --with "onnx-asr[gpu,hub]" python {skill-root}/scripts/transcribe.py ... +``` + +PEP 508 markers cannot detect GPUs, so the GPU extra has to be requested +explicitly. The `python` command rather than `uv run` is required: it skips the +script's cpu-extra dependency so onnxruntime-gpu never co-installs alongside +onnxruntime, which would silently fall back to CPU. The script warns on stderr +when an NVIDIA GPU is visible but CUDA is unavailable. + +## When the transcript gate fails + +`verify_transcript.py` finds audio above the silence floor that produced no +words and names the regions. The default reading is dropped speech, and the +recovery is to re-run the take and re-verify. + +Not every flagged region is lost speech. A laugh, a music bed, or an off-mic +aside reads the same way to a coverage scan. When the creator has listened to a +region and confirmed it: + +``` +uv run {skill-root}/scripts/verify_transcript.py transcript/words.json \ + --audio-map cut/audio-map.json --wpm <[owner] wpm> \ + -o cut/transcript-check.json \ + --accept-region - --reason "" +``` + +Repeatable. The acceptance must fully cover the flagged region, and both the +region and the reason land in `transcript-check.json`. That is the only way past +this gate, and it is a recorded decision rather than a silent one. diff --git a/skills/mc-cut/scripts/analyze_audio.py b/skills/mc-cut/scripts/analyze_audio.py new file mode 100644 index 0000000..7e99317 --- /dev/null +++ b/skills/mc-cut/scripts/analyze_audio.py @@ -0,0 +1,347 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.11" +# /// +"""Audio silence map for the cut stage: the TIMING source of truth. + +Usage: + uv run {skill-root}/scripts/analyze_audio.py {projects-path}//raw/ \ + -o {projects-path}//cut/audio-map.json \ + [--noise -30] [--map-granularity 0.10] + +Why this exists (read before changing any of it): + Cutting decisions used to be derived from TRANSCRIPT timestamps: the gap + between one word's end and the next word's start. That is wrong, and it + shipped a corrupted cut on the first real project (2026-07-24). parakeet + absorbs a pause into the preceding word's end, so the word "about." + was timestamped 32.16 -> 34.64: a single "word" lasting 2.5 seconds + because it swallowed the silence after it. Every gap computed from those + timestamps reads about 0.0 across real dead air, so the silence detector + was blind. On a take with over 5 minutes of dead air it found 12 silences. + ffmpeg silencedetect on the same audio found 402 intervals totalling 309 + seconds. + + So: the TRANSCRIPT is the authority on CONTENT (what words were said, in + what order). The AUDIO is the authority on TIMING (where silence is, and + therefore where it is safe to cut). This script produces the second one. + Do not reintroduce gap-derived silence anywhere downstream. + + A second property makes this the right primitive: a cut that lands inside + an audio-verified silence CANNOT clip a word. The "never cut inside a + word" rule stops being an assertion about timestamps and becomes a + structural guarantee. + +Contract: + input any media file (audio or video); ffmpeg reads the audio stream. + output audio-map.json: + {"media", "duration", "noise_db", "map_granularity", + "silent_seconds", "speech_seconds", + "counts": {"silence", "speech"}, + "silence": [{"start", "end", "dur"}, ...], + "speech": [{"start", "end", "dur"}, ...]} + The two interval lists are exact complements over [0, duration], + both sorted, non-overlapping, times to 3 decimals. + summary json.dumps on stdout: counts, silent_seconds, speech_seconds, + output path. + +Consumers: + verify_transcript.py the completeness gate (speech with no words is + DROPPED SPEECH, not dead air) + cutplan.py silence candidates and candidate edge snapping + +Both read the same file, so the two never disagree about where silence is and +the expensive decode happens once. + +Tuning: + --noise dBFS floor below which audio counts as silence (default + -30). Room tone on a decent mic sits well under this; a + noisy room may need -35 or -40. Too high and speech onsets + get eaten; too low and nothing registers as silent. + --map-granularity shortest interval to report (default 0.10s). Deliberately + FINER than any cutting threshold: consumers filter up from + this list, and a coarse map cannot be refined later. + +Exit codes: 0 ok, 1 ffmpeg/probe failure, 2 usage error. + +STATUS: implemented (pure parsing and interval logic covered by +scripts/tests/test-analyze_audio.py). +""" + +import argparse +import json +import re +import subprocess +import sys +from pathlib import Path + +DEFAULT_NOISE_DB = -30.0 +# Map granularity, NOT a cutting threshold. It must stay well below every +# consumer's threshold, because consumers filter UP from this list and a +# coarse map cannot be refined later. +# +# Two consumers need the small intervals specifically: +# - edge snapping (cutplan.py) moves a cut edge into the nearest silence, +# and the gaps between doubled words or around a clipped onset are +# 0.1 to 0.2s. A map that started at 0.3s would leave exactly those +# edges unsnappable, which is the "n-now" artifact class. +# - the transcript gate measures audible-but-untranscribed audio, so its +# silence input has to be complete, not just the big pauses. +# On the real 2026-07-24 take this yields 929 intervals; at 0.3 it yields +# 400, and the 529 it drops are precisely the ones snapping needs. +DEFAULT_MAP_GRANULARITY = 0.10 + +_START_RE = re.compile(r"silence_start:\s*(-?[\d.]+)") +_END_RE = re.compile(r"silence_end:\s*(-?[\d.]+)") + + +def r3(x): + return round(float(x), 3) + + +def probe_duration(media): + """Media duration in seconds via ffprobe. Raises on failure.""" + out = subprocess.run( + ["ffprobe", "-v", "error", "-show_entries", "format=duration", + "-of", "default=nokey=1:noprint_wrappers=1", str(media)], + capture_output=True, text=True, check=True, + ) + return float(out.stdout.strip()) + + +def parse_silencedetect(stderr_text, duration): + """Parse ffmpeg silencedetect stderr into silence intervals (pure). + + silencedetect emits paired lines: + [silencedetect @ ...] silence_start: 12.345 + [silencedetect @ ...] silence_end: 14.567 | silence_duration: 2.222 + A file that ENDS in silence gets a silence_start with no matching + silence_end, so the open interval is closed at `duration`. Intervals are + clamped into [0, duration], zero-length ones dropped, and the result is + sorted and merged so consumers can rely on non-overlapping order. + """ + intervals = [] + open_start = None + for line in stderr_text.splitlines(): + m = _START_RE.search(line) + if m: + open_start = float(m.group(1)) + continue + m = _END_RE.search(line) + if m and open_start is not None: + intervals.append((open_start, float(m.group(1)))) + open_start = None + if open_start is not None: + intervals.append((open_start, duration)) + return _clean(intervals, duration) + + +def _clean(intervals, duration): + """Clamp into [0, duration], drop empties, sort, merge overlaps (pure).""" + cleaned = [] + for start, end in intervals: + start = max(0.0, min(float(start), duration)) + end = max(0.0, min(float(end), duration)) + if end - start > 1e-6: + cleaned.append((start, end)) + cleaned.sort() + merged = [] + for start, end in cleaned: + if merged and start <= merged[-1][1] + 1e-6: + merged[-1][1] = max(merged[-1][1], end) + else: + merged.append([start, end]) + return [(s, e) for s, e in merged] + + +def complement(intervals, duration): + """The gaps between intervals over [0, duration] (pure). + + Given the silence list this yields the speech list, and vice versa. Both + directions are used: consumers ask "is this span silent" and "is this + span speech" and neither should have to invert the other by hand. + """ + out = [] + cursor = 0.0 + for start, end in intervals: + if start - cursor > 1e-6: + out.append((cursor, start)) + cursor = max(cursor, end) + if duration - cursor > 1e-6: + out.append((cursor, duration)) + return out + + +def as_records(intervals): + """Interval tuples to the JSON record shape (pure).""" + return [{"start": r3(s), "end": r3(e), "dur": r3(e - s)} + for s, e in intervals] + + +def to_pairs(records): + """JSON records back to interval tuples (pure). The read side of + as_records, for consumers loading an audio-map.json.""" + return [(float(r["start"]), float(r["end"])) for r in records] + + +def total(intervals): + """Summed length of an interval list (pure).""" + return sum(e - s for s, e in intervals) + + +def overlap_seconds(intervals, start, end): + """How much of [start, end] is covered by `intervals` (pure). + + The primitive both consumers are built on: "how silent is this span?" + for the transcript gate, and "is this candidate edge inside silence?" + for the cutter. + """ + if end <= start: + return 0.0 + covered = 0.0 + for a, b in intervals: + if b <= start: + continue + if a >= end: + break + covered += min(b, end) - max(a, start) + return covered + + +def silent_fraction(silence, start, end): + """Fraction of [start, end] that is silent, 0.0 to 1.0 (pure).""" + span = end - start + if span <= 0: + return 1.0 + return overlap_seconds(silence, start, end) / span + + +def enclosing(intervals, t): + """The interval containing t, or None (pure).""" + for a, b in intervals: + if a <= t <= b: + return (a, b) + if a > t: + break + return None + + +def nearest_silence(silence, t, max_shift, direction="nearest"): + """Nearest point inside a silence interval to t, within max_shift (pure). + + Returns t unchanged when it already sits in silence, the nearest + qualifying point inside some silence interval when one is close enough, + or None when nothing qualifies within max_shift. Used to snap cut edges + so a cut can never land mid-word (see the module docstring). + + direction constrains which way t may move: + "back" only earlier (or unchanged) + "forward" only later (or unchanged) + "nearest" either way + + Direction matters and "nearest" is the wrong default for cut edges. A cut + candidate is a span to REMOVE, so its start must only move earlier and + its end only later: the span may widen into surrounding silence, which is + always safe, and can never invert or collapse. Snapping both edges to the + unconstrained nearest silence can pull an end backwards past its own + start and annihilate the candidate. + """ + if enclosing(silence, t) is not None: + return t + best = None + for a, b in silence: + # The point inside [a, b] closest to t, biased just inside the edge. + cand = min(max(t, a), b) + if direction == "back" and cand > t: + continue + if direction == "forward" and cand < t: + continue + dist = abs(cand - t) + if dist <= max_shift and (best is None or dist < best[0]): + best = (dist, cand) + return None if best is None else best[1] + + +def run_silencedetect(media, noise_db, granularity): + """Run ffmpeg silencedetect and return its stderr text. Raises on failure.""" + proc = subprocess.run( + ["ffmpeg", "-hide_banner", "-nostats", "-i", str(media), + "-af", f"silencedetect=noise={noise_db}dB:d={granularity}", + "-f", "null", "-"], + capture_output=True, text=True, + ) + # silencedetect writes to stderr and ffmpeg exits 0; a non-zero exit is a + # real failure (no audio stream, unreadable file). + if proc.returncode != 0: + raise RuntimeError(proc.stderr.strip()[-2000:]) + return proc.stderr + + +def build(media, duration, silence_pairs, noise_db, granularity): + """Assemble the audio-map payload (pure).""" + speech_pairs = complement(silence_pairs, duration) + return { + "media": str(media), + "duration": r3(duration), + "noise_db": noise_db, + "map_granularity": granularity, + "silent_seconds": r3(total(silence_pairs)), + "speech_seconds": r3(total(speech_pairs)), + "counts": {"silence": len(silence_pairs), "speech": len(speech_pairs)}, + "silence": as_records(silence_pairs), + "speech": as_records(speech_pairs), + } + + +def main(argv=None): + p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + p.add_argument("media", help="path to the media file (audio or video)") + p.add_argument("-o", "--output", required=True, help="path to audio-map.json") + p.add_argument("--noise", type=float, default=DEFAULT_NOISE_DB, + help=f"silence floor in dBFS (default {DEFAULT_NOISE_DB})") + p.add_argument("--map-granularity", type=float, + default=DEFAULT_MAP_GRANULARITY, + help=f"shortest silence to report (default " + f"{DEFAULT_MAP_GRANULARITY}s). This is MAP RESOLUTION, " + "not a cutting threshold: keep it finer than every " + "consumer, since a coarse map cannot be refined later") + args = p.parse_args(argv) + + media = Path(args.media) + if not media.is_file(): + print(f"analyze_audio: media not found: {media}", file=sys.stderr) + return 2 + if args.map_granularity <= 0: + print("analyze_audio: --map-granularity must be positive", + file=sys.stderr) + return 2 + + try: + duration = probe_duration(media) + stderr_text = run_silencedetect(media, args.noise, + args.map_granularity) + except (subprocess.CalledProcessError, RuntimeError, ValueError) as e: + print(f"analyze_audio: cannot analyze {media}: {e}", file=sys.stderr) + return 1 + + silence_pairs = parse_silencedetect(stderr_text, duration) + payload = build(media, duration, silence_pairs, args.noise, + args.map_granularity) + + output = Path(args.output) + output.parent.mkdir(parents=True, exist_ok=True) + output.write_text(json.dumps(payload, indent=2, ensure_ascii=False) + "\n", + encoding="utf-8") + + print(json.dumps({ + "ok": True, + "output": str(output), + "duration": payload["duration"], + "counts": payload["counts"], + "silent_seconds": payload["silent_seconds"], + "speech_seconds": payload["speech_seconds"], + })) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/mc-cut/scripts/composite_core.py b/skills/mc-cut/scripts/composite_core.py index e40de7b..e421e3a 100644 --- a/skills/mc-cut/scripts/composite_core.py +++ b/skills/mc-cut/scripts/composite_core.py @@ -29,6 +29,7 @@ import hashlib import json +import os import platform import shutil import subprocess @@ -203,6 +204,223 @@ def resolve_overlays(beats, graphics_dir): return found, missing +# --- fast preview compositing: proxies + overlay lanes ----------------------- +# +# The first real project's preview render took 25+ minutes and stalled twice. +# Two costs multiplied: +# (a) the cut sliced a 4K master into 234 segments, so every render did 234 +# seek+decodes into 4K video to build a 720p preview; +# (b) all 56 overlays went into ONE filtergraph as a 56-deep overlay stack, +# so every output frame walked 56 compositing steps whether or not any +# overlay was actually on screen. +# +# Both have a cheap fix and neither needs a different tool: +# (a) PROXY MASTERS. Transcode each source once, linearly, to the preview +# height, and cut the preview from the proxy. A linear read plus 234 +# cheap seeks beats 234 expensive ones, the proxy is reused by every +# later re-render, and the final render still cuts from the true master. +# (b) OVERLAY LANES. Overlays are intervals in time and mostly do not +# overlap. Pack them into the fewest non-overlapping lanes (greedy +# interval scheduling), build each lane as a cheap time-sequential +# CONCAT of transparent gaps and overlay clips, then stack only the +# lanes. Stack depth becomes max-concurrent-overlays instead of +# total-overlays: 56 became 2 on the real project. +# +# Measured together on that project: about 3 minutes instead of about an hour. + +LANE_CODEC_ARGS = { + # Lossless with alpha, and it run-length encodes flat transparency, so a + # mostly-empty overlay lane costs almost nothing on disk. The default. + "qtrle": ["-c:v", "qtrle", "-pix_fmt", "argb"], + # The validated original. Much larger files (a 16 min 720p lane runs to + # gigabytes), kept for anyone who needs ProRes intermediates. + "prores": ["-c:v", "prores_ks", "-profile:v", "4444", + "-pix_fmt", "yuva444p10le"], +} +DEFAULT_LANE_CODEC = "qtrle" + + +def plan_overlay_lanes(overlays): + """Pack overlays into the fewest non-overlapping time lanes (pure). + + Greedy interval scheduling over overlays sorted by start: each overlay + goes into the first lane whose last overlay has already ended, otherwise + it opens a new lane. The lane count is exactly the maximum number of + overlays on screen at once, which is the whole point: it becomes the + depth of the final overlay stack. + + Returns a list of lanes, each a list of overlay dicts sorted by start. + """ + lanes = [] + for ov in sorted(overlays, key=lambda o: (o["start"], o["dur"])): + start = ov["start"] + for lane in lanes: + last = lane[-1] + if last["start"] + last["dur"] <= start + 1e-6: + lane.append(ov) + break + else: + lanes.append([ov]) + return lanes + + +def build_lane_filter(lane, total, size, fps): + """(inputs, filter_complex) for one overlay lane (pure). + + The lane is a single video the length of the whole timeline: transparent + gap, overlay, transparent gap, overlay, ..., trailing gap, concatenated. + Gaps are lavfi color sources zeroed to full transparency (a color source + carries no usable alpha of its own, so aa=0 is not optional). Every + element is normalized to the same size, rate and rgba format because + concat refuses mismatched inputs. + """ + if not lane: + # A lane with no overlays is a fully-transparent pass, which costs a + # whole encode to composite nothing. The caller skips it. + return [], "" + w, h = size + inputs, filters, labels = [], [], [] + n = 0 + cursor = 0.0 + + def add_gap(dur): + nonlocal n + if dur <= 0.001: + return + inputs.extend(["-f", "lavfi", "-t", f"{dur:.3f}", + "-i", f"color=c=black:s={w}x{h}:r={fps}"]) + filters.append(f"[{n}:v]format=rgba,colorchannelmixer=aa=0," + f"fps={fps},setpts=PTS-STARTPTS[l{n}]") + labels.append(f"[l{n}]") + n += 1 + + def add_overlay(ov): + nonlocal n + if ov.get("image"): + inputs.extend(["-loop", "1", "-t", f"{ov['dur']:.3f}", + "-i", str(ov["path"])]) + else: + inputs.extend(["-i", str(ov["path"])]) + filters.append( + f"[{n}:v]format=rgba,scale={w}:{h}:force_original_aspect_ratio=" + f"decrease,pad={w}:{h}:(ow-iw)/2:(oh-ih)/2:color=#00000000," + f"fps={fps},trim=duration={ov['dur']:.3f},setpts=PTS-STARTPTS" + f"[l{n}]") + labels.append(f"[l{n}]") + n += 1 + + for ov in lane: + add_gap(ov["start"] - cursor) + add_overlay(ov) + cursor = ov["start"] + ov["dur"] + add_gap(total - cursor) + + if not labels: + return [], "" + graph = ";".join(filters) + ";" + "".join(labels) + \ + f"concat=n={len(labels)}:v=1:a=0[laneout]" + return inputs, graph + + +def build_lane_command(lane, total, size, fps, out_path, + codec=DEFAULT_LANE_CODEC): + """ffmpeg argv rendering one overlay lane to an alpha-bearing file (pure).""" + inputs, graph = build_lane_filter(lane, total, size, fps) + if not graph: + return [] + return (["ffmpeg", "-y", "-hide_banner", "-v", "error"] + inputs + + ["-filter_complex", graph, "-map", "[laneout]", + "-t", f"{total:.3f}"] + + LANE_CODEC_ARGS.get(codec, LANE_CODEC_ARGS[DEFAULT_LANE_CODEC]) + + [str(out_path)]) + + +def build_lane_composite_command(base, lane_files, output, crf=28, + preset="veryfast", + extra_output_flags=()): + """ffmpeg argv stacking the (few) lane files onto the base cut (pure). + + This is the pass whose depth used to be the overlay count. It is now the + lane count, and the base carries the audio through untouched. + """ + argv = ["ffmpeg", "-y", "-hide_banner", "-v", "error", "-i", str(base)] + for lf in lane_files: + argv += ["-i", str(lf)] + graph, prev = "", "0:v" + for i in range(len(lane_files)): + tag = f"c{i + 1}" + graph += f"[{prev}][{i + 1}:v]overlay=format=auto:shortest=0[{tag}];" + prev = tag + graph += f"[{prev}]format=yuv420p[vout]" + argv += ["-filter_complex", graph, "-map", "[vout]", "-map", "0:a?", + "-c:v", "libx264", "-preset", preset, "-crf", str(crf), + "-c:a", "aac", "-b:a", "160k", "-movflags", "+faststart"] + argv += list(extra_output_flags) + argv += [str(output)] + return argv + + +def proxy_path(proxy_dir, source, height): + """Where a source's preview proxy lives (pure). + + Named by the source stem plus height so a project's proxies are readable + on disk; freshness is decided by the sidecar digest, not the name. + """ + return Path(proxy_dir) / f"{Path(source).stem}-{height}p.mp4" + + +def proxy_is_fresh(proxy, source): + """True when a proxy exists and was built from this exact source (pure-ish). + + The sidecar records the source's content digest at build time, so a + re-recorded take with the same filename correctly invalidates its proxy. + """ + proxy = Path(proxy) + sidecar = proxy.with_name(proxy.name + ".src") + if not proxy.is_file() or not sidecar.is_file(): + return False + try: + return sidecar.read_text(encoding="utf-8").strip() == \ + content_digest(source, cheap=True) + except OSError: + return False + + +def build_proxy_command(source, out, height, crf=26, preset="veryfast"): + """ffmpeg argv transcoding a source to a preview proxy (pure). + + One linear pass. Audio is re-encoded rather than copied so the proxy is + seekable and self-contained; the preview's audio comes from here too, and + the final render never touches proxies. + """ + return ["ffmpeg", "-y", "-hide_banner", "-v", "error", "-i", str(source), + "-vf", f"scale=-2:{height}", "-c:v", "libx264", "-preset", preset, + "-crf", str(crf), "-pix_fmt", "yuv420p", "-c:a", "aac", + "-b:a", "160k", "-movflags", "+faststart", str(out)] + + +def write_proxy_sidecar(proxy, source): + """Record which source build this proxy, for proxy_is_fresh.""" + proxy = Path(proxy) + proxy.with_name(proxy.name + ".src").write_text( + content_digest(source, cheap=True) + "\n", encoding="utf-8") + + +def proxied_edl(edl, mapping): + """A copy of the EDL with every source swapped for its proxy (pure). + + Timecodes are untouched: a proxy is the same footage at a smaller frame + size, so every EDL time stays valid against it. Sources with no proxy in + the mapping are left pointing at the original. + """ + out = json.loads(json.dumps(edl)) + for seg in out.get("segments", []): + seg["source"] = mapping.get(seg["source"], seg["source"]) + if out.get("source") in mapping: + out["source"] = mapping[out["source"]] + return out + + # --- ffmpeg filtergraph and command ----------------------------------------- @@ -546,6 +764,152 @@ def ffmpeg_version(): return parts[2] if len(parts) >= 3 and parts[0] == "ffmpeg" else "unknown" +# --- render identity, atomic publish, output validation --------------------- +# +# On 2026-07-24 a stale background render (an EDL that had since been +# superseded) finished late and wrote renders/preview.mp4 CONCURRENTLY with +# the current render. Two processes interleaved bytes into one path and the +# creator was handed an unplayable file: "Invalid NAL unit size", "missing +# picture in access unit". Nothing detected it, because the only post-render +# check was the container duration, which a corrupt file still reports +# happily. +# +# Three defences, all needed: +# 1. Never write the deliverable path directly. Render to a unique temp +# path and os.replace it into place, which is atomic on one filesystem, +# so a reader sees either the old file or the new one and never a +# half-written one. +# 2. Key a render by its CONTENT, and re-check that key just before +# publishing. A render whose EDL changed underneath it is stale and must +# discard its output instead of clobbering a newer one. Content keying +# beats process cancellation because it also catches a crashed, detached, +# or forgotten job, which is exactly what happened. +# 3. Prove the file decodes before calling it a deliverable. + + +def render_key(edl, params=None, overlay_digests=None, ffmpeg=None): + """Content identity of a render: what is being rendered, never when. + + Two renders agree on a key exactly when they would produce the same + bytes: same EDL, same parameters, same overlay file contents, same + ffmpeg. The key is what makes supersede detection work without locks. + """ + payload = { + "edl": edl, + "params": dict(sorted((params or {}).items())), + "overlays": dict(sorted((overlay_digests or {}).items())), + "ffmpeg": ffmpeg if ffmpeg is not None else ffmpeg_version(), + } + raw = json.dumps(payload, sort_keys=True, ensure_ascii=False, + default=str).encode("utf-8") + return hashlib.sha256(raw).hexdigest()[:32] + + +def temp_render_path(output, key): + """A per-process temp path beside the output. + + The pid is load-bearing: keying the temp file on content alone would give + two concurrent renders of the SAME edl one shared temp path, recreating + the interleaved-write corruption one level down. + """ + output = Path(output) + return output.with_name(f".{output.stem}.{key[:12]}.{os.getpid()}.part" + f"{output.suffix}") + + +def validate_render(path, expected_duration=None, tolerance=0.5): + """Decode the whole file and check its duration. Returns (ok, problems). + + The decode pass is the check that was missing: `ffmpeg -v error -f null -` + walks every packet and prints on any decode error. Container duration + alone cannot see a corrupt bitstream. + """ + problems = [] + path = Path(path) + if not path.is_file(): + return False, [f"output not written: {path}"] + if path.stat().st_size == 0: + return False, [f"output is empty: {path}"] + try: + proc = subprocess.run( + ["ffmpeg", "-v", "error", "-i", str(path), "-f", "null", "-"], + capture_output=True, text=True) + except OSError as e: + return False, [f"cannot run ffmpeg to validate: {e}"] + stderr = proc.stderr.strip() + if proc.returncode != 0 or stderr: + problems.append("decode errors: " + (stderr[-800:] or + f"exit {proc.returncode}")) + actual = probe_duration(path) + if actual is None: + problems.append("no readable duration") + elif expected_duration is not None and \ + abs(actual - expected_duration) > tolerance: + problems.append(f"duration {actual:.3f}s vs expected " + f"{expected_duration:.3f}s (tolerance {tolerance}s)") + return not problems, problems + + +def current_render_key(edl_path, params=None, overlay_digests=None, + ffmpeg=None): + """Re-read the EDL from disk and compute its key now, or None. + + None means the EDL is unreadable, in which case the caller must NOT treat + the render as superseded: an unreadable EDL is a separate problem and + discarding a good render over it would lose work. + """ + try: + edl = json.loads(Path(edl_path).read_text(encoding="utf-8")) + except (OSError, ValueError): + return None + return render_key(edl, params, overlay_digests, ffmpeg) + + +def publish_render(tmp_path, output, key=None): + """Atomically move a validated temp render into place. + + Also writes a .key sidecar naming the render identity now in the + file. Derived artifacts (fcpxml, boundary frames, cutplan summary) can + record the same key, so drift after a re-cut is detectable rather than + silent. + """ + tmp_path, output = Path(tmp_path), Path(output) + output.parent.mkdir(parents=True, exist_ok=True) + os.replace(tmp_path, output) + if key: + output.with_name(output.name + ".key").write_text(key + "\n", + encoding="utf-8") + return output + + +def write_json_atomic(path, payload): + """Write JSON via a temp file and an atomic replace. + + Any file a running render might read concurrently must be written this + way. A plain write truncates first, so a reader arriving mid-write sees a + torn or empty file: cut/edl.json in particular is read by every render, + including ones already in flight. Torn reads here are not corruption of + the deliverable (current_render_key treats an unreadable EDL as "cannot + tell" and declines to supersede), but they do abort renders for no + reason. + """ + path = Path(path) + path.parent.mkdir(parents=True, exist_ok=True) + tmp = path.with_name(f".{path.name}.{os.getpid()}.tmp") + tmp.write_text(json.dumps(payload, indent=2, ensure_ascii=False) + "\n", + encoding="utf-8") + os.replace(tmp, path) + return path + + +def discard_render(tmp_path): + """Remove an abandoned temp render, ignoring an already-gone file.""" + try: + Path(tmp_path).unlink() + except OSError: + pass + + def segment_identity(edl, seg): """Position-independent content identity of a render-segment: its EDL slice as ordered (source, source-relative start, end) plus fade_ms. Never diff --git a/skills/mc-cut/scripts/cutplan.py b/skills/mc-cut/scripts/cutplan.py index a159d94..f6b5515 100644 --- a/skills/mc-cut/scripts/cutplan.py +++ b/skills/mc-cut/scripts/cutplan.py @@ -6,8 +6,24 @@ Usage: uv run {skill-root}/scripts/cutplan.py {projects-path}//transcript/words.json \ + --audio-map {projects-path}//cut/audio-map.json \ -o {projects-path}//cut/candidates.json +The two-source rule (binding; the defect that shipped a corrupt cut): + The TRANSCRIPT is the authority on CONTENT: which words were said, in + what order, so a detector can recognize a filler, a stutter, a retake. + The AUDIO is the authority on TIMING: where silence is, and therefore + where a cut is safe. Every candidate's TEXT comes from words.json; every + candidate's EDGES are snapped into an audio-verified silence. + + Never time a cut off transcript timestamps. parakeet absorbs a pause into + the preceding word's end, so word gaps read about 0.0 across real dead + air and word ends reach past the sound. Both failure modes shipped on + 2026-07-24: the silence detector found 12 silences in a take with over 5 + minutes of dead air (the audio had 402 totalling 309s), and a stutter + candidate whose end came from an absorbed word end clipped the repeat's + onset into an audible "n-now". See analyze_audio.py. + Contract: input Scribe word-level transcript JSON (from transcribe.py) with shape {"media": str, "duration": float, "text": str, @@ -17,27 +33,45 @@ find, each with timestamps, the surrounding words, a human reason, and a severity. Shape: {"media", "duration", - "thresholds": {"min_silence", "retake_window", "retake_run"}, - "counts": {"silence", "filler", "stutter", "retake", "marker"}, + "thresholds": {"min_silence", "retake_window", "retake_run", + "snap_ms"}, + "silence_source": "audio", + "counts": {"silence", "filler", "stutter", "retake", "blooper", + "marker"}, + "trimmable_silence_seconds": float, + "unsnapped": int, "candidates": [ {"type", ["cls"], "start", "end", "dur", - "text", "reason", "severity"}, ... ]} + "text", "reason", "severity", "snapped"}, ... ]} Candidates are sorted by start time; times carry 2 decimals; "cls" - (hard|soft) appears on filler candidates only. + (hard|soft) appears on filler candidates only. "snapped" is false + when an edge could not reach an audio silence within --snap-ms, + meaning that candidate's edges still rest on transcript + timestamps and need an ear before they are cut. note this script finds CANDIDATES only; taste calls (keep or cut) happen in mc-cut's plan, which the creator approves at gate 2. Cutting rules live in this skill's SKILL.md (the "Cutting rules" section). +Every candidate's span is THE PART TO REMOVE, uniformly across types. + Detectors (all pure stdlib): - silence any inter-word gap >= min-silence (default 0.7s), plus leading - silence (0 -> first word) and trailing silence (last word -> - duration). severity med if < 2.0s, high if >= 2.0s. + silence dead-air tightening from the AUDIO map. Interior silences + >= min-silence (default 0.30s) are trimmed to keep-ms (default + 200ms) of breathing room, split evenly so both edges land + inside silence; leading and trailing silence trims to keep-ms + off the first and last word. Silences BELOW min-silence are + untouched: those micro-beats are the speaker's rhythm, and + flattening them is what makes a cut sound machine-gunned. + severity med if the silence is < 2.0s, high if >= 2.0s. Carries + silence_start/silence_end/keep_ms alongside the trim span. filler hard fillers (um uh hmm er ah mm; severity high) matched word- boundary, case-insensitive, punctuation-stripped; consecutive hard fillers merge into one candidate spanning the run. soft fillers - (sentence-initial connectives: so right okay well actually - basically anyway "you know" "i mean"; severity low) flagged only at - a sentence start (prev word ends ./!/? or gap_before >= 0.5) so - mid-sentence natural use is not flagged. + (severity low, cadence-sensitive, DEFAULT IS TO KEEP) flagged + only at a sentence start (prev word ends ./!/? or gap_before >= + 0.5). The soft list is deliberately tiny and the creator's + voice bible extends or overrides it via --voice-bible; see + SOFT_SINGLE's comment for why blanket soft-filler cutting is a + bug, not a feature. stutter immediate normalized word repetition ("weird weird"); candidate covers the first occurrence. severity med. retake (a) spoken cues (case-insensitive): "take N", "try that again", @@ -46,6 +80,22 @@ normalized words that reappears within the next retake-window (default 16) words -- the EARLIER occurrence is the candidate. severity high. + (c) SECTION re-reads (cls "section"): a run of >= section-run + (default 8) words recurring within section-window-s (default + 45s). The candidate spans the first attempt's start to the + restart's start, so the abandoned take AND its reset pause both + go. The locality window is in SECONDS on purpose: that is what + distinguishes a redo from a deliberate callback minutes later, + and word-distance cannot express it. + blooper expletives and reset phrases the creator would never ship + ("Oh fuck.", "scratch that", "sorry"). severity high with cls + "reset" when the take STOPPED next to it (a silence of + reset-silence seconds or more within reset-look seconds; a + near-certain flub), med with cls "ambiguous" in continuous + speech (possibly scripted, e.g. "that damn term") so an ear + settles it. Measured separation on real footage: 0.77s of + silence beside the scripted line, 7.65s beside the blooper. + --blooper-cues overrides the vocabulary. marker interview-mode cue: any marker phrase (default "question from the interviewer") in the normalized word stream; marks a segment boundary (the creator read an interviewer question aloud on @@ -57,9 +107,16 @@ CLI: positional words.json -o/--output candidates.json (required) - --min-silence FLOAT (default 0.7) + --audio-map PATH audio-map.json (REQUIRED; no gap fallback exists) + --snap-ms INT (default 250) candidate edge snap budget + --min-silence FLOAT (default 0.30) tightening floor + --keep-ms INT (default 200) breathing room kept per silence --retake-window INT (default 16) --retake-run INT (default 3) + --section-run INT (default 8) words that make a section re-read + --section-window-s FLT (default 45.0) redo-versus-callback locality + --voice-bible PATH creator's cadence keep/cut lists + --blooper-cues STR comma-separated blooper vocabulary override --marker-cues STR comma-separated override (default "question from the interviewer"; the legacy phrase "question from claude" is a supported alternative) @@ -71,10 +128,75 @@ import json import string import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent)) # noqa: E402 +import analyze_audio as audio # noqa: E402 + +# How far a candidate edge may move to reach an audio-verified silence. +DEFAULT_SNAP_MS = 250 +# Breathing room left in a tightened interior silence. 200ms reads as a beat; +# zero reads as machine-gunned. +DEFAULT_KEEP_MS = 200 +# Section-redo matcher: a long word run inside a seconds-wide locality +# window. See _find_section_redo for why the window is time, not words. +DEFAULT_SECTION_RUN = 8 +DEFAULT_SECTION_WINDOW_S = 45.0 +# Blooper context: a genuine flub is next to a long STOP, not a breath. See +# _near_reset for the measured separation (0.77s scripted vs 7.65s blooper). +DEFAULT_RESET_SILENCE_S = 2.0 +DEFAULT_RESET_LOOK_S = 1.5 +# Interior silences shorter than this are the speaker's rhythm, not dead air. +# +# Measured on the real 20.5 minute take from 2026-07-24 (929 silences at or +# above 0.1s, 383s of silence total). How much trimmable dead air each +# candidate floor reaches: +# +# floor silences hit trimmable share of all trimmable dead air +# 0.20 452 230.6s 100.0% +# 0.30 400 228.5s 99.1% +# 0.45 248 199.9s 86.7% +# 0.70 92 148.0s 64.2% +# +# 0.30 is the knee. It catches 99% of the dead air while sitting clear of +# that speaker's natural rhythm (their MEDIAN silence is 0.19s, which is +# inter-phrase cadence, not dead air). The 0.45 this shipped with left about +# 29 seconds of slack in a 20 minute video, which is exactly the "loose" +# quality the first cut was criticized for. Going below 0.30 gains under a +# percent and starts eating the speaker's rhythm. +# +# This is a PACING knob and it is per-creator: a slower, more deliberate +# delivery wants it higher, which is why the studio config exposes it through +# [cut] cutplan-flags. +DEFAULT_MIN_SILENCE = 0.30 HARD_FILLERS = {"um", "uh", "hmm", "er", "ah", "mm"} -SOFT_SINGLE = {"so", "right", "okay", "well", "actually", "basically", "anyway"} +# Soft fillers are CADENCE-SENSITIVE and the default list is deliberately +# short. It used to include "so", "right", "okay", "well" and "anyway", which +# meant the mechanical pass flagged all 19 of one creator's sentence-initial +# "So"s on a single take. Their voice bible names "so" as their natural +# connective glue; blanket-cutting it destroys the speaker's cadence and +# produces a technically-clean, tonally-dead read. +# +# A speaker's connective words are TASTE, and taste lives in files. The +# per-creator list comes from the voice bible via --voice-bible (see +# parse_voice_cadence); this default is only the small set that is hard to +# defend as anyone's deliberate rhythm. +SOFT_SINGLE = {"basically", "actually", "literally"} SOFT_PHRASES = [["you", "know"], ["i", "mean"]] +# Expletives and reset phrases: a genuine blooper the creator would never +# ship. Detected separately from retakes because the ACTION differs (a +# blooper is always cut; a retake is a choice between takes) and because the +# vocabulary is per-creator. --blooper-cues overrides. +BLOOPER_WORDS = {"fuck", "shit", "damn", "crap", "ugh", "oops", "oof", + "bollocks", "bugger"} +BLOOPER_PHRASES = [ + ["let", "me", "redo", "that"], + ["let", "me", "try", "again"], + ["scratch", "that"], + ["start", "over"], + ["sorry"], +] NUMBER_WORDS = {"one", "two", "three", "four", "five", "six", "seven", "eight", "nine", "ten"} # spoken retake cues, checked longest-first at each position ("take N" handled @@ -117,32 +239,129 @@ def _cand(ctype, start, end, text, reason, severity, cls=None): return c -def detect_silence(words, duration, min_silence): +def _word_at(words, t, side): + """The word just before ('prev') or just after ('next') time t (pure). + Used only to LABEL a silence candidate, never to time it.""" + if side == "prev": + best = None + for w in words: + if w["end"] <= t + 0.01: + best = w + else: + break + return best + for w in words: + if w["start"] >= t - 0.01: + return w + return None + + +def detect_silence(words, duration, min_silence, silence_intervals, + keep_ms=DEFAULT_KEEP_MS): + """Dead-air tightening candidates from the AUDIO silence map. + + silence_intervals is the (start, end) list out of analyze_audio.py. The + transcript is used only to name the neighbouring words in the reason + string, so a reader of cutplan.md knows where a trim sits. + + This used to compute `cur["start"] - prev["end"]` over transcript + timestamps. That was the defect that shipped a bad cut on 2026-07-24: + parakeet absorbs a pause into the preceding word's end, so those gaps + read about 0.0 across real dead air. On a take with over 5 minutes of + dead air the gap method found 12 silences; the audio found 402 totalling + 309 seconds. Do not put word gaps back here. + + A candidate's span is THE PART TO REMOVE, not the whole silence, which + matches what every other detector here emits and makes the candidate + list uniformly "spans you might cut". + + Tightening, not flattening (the single biggest quality lever on a + talking-head read): an interior silence at or above min_silence is + trimmed down to keep_ms of breathing room rather than removed outright, + with the kept beat split evenly on both sides so both cut edges sit + inside silence by construction. Silences BELOW min_silence are left + completely alone: those micro-beats are the speaker's rhythm, and + removing them is what makes an edit sound machine-gunned. Leading and + trailing silence is different, there is nothing to breathe between, so + the head and tail trim all the way to keep_ms off the first and last + word. + """ out = [] - if not words: - return out - lead = words[0]["start"] - if lead >= min_silence: - dur = r2(lead) - sev = "high" if dur >= 2.0 else "med" - out.append(_cand("silence", 0.0, lead, "", - f'{fmt(dur)}s gap before "{words[0]["word"]}"', sev)) - for prev, cur in zip(words, words[1:]): - gap = cur["start"] - prev["end"] - if gap >= min_silence: - dur = r2(gap) - sev = "high" if dur >= 2.0 else "med" - out.append(_cand("silence", prev["end"], cur["start"], "", - f'{fmt(dur)}s gap before "{cur["word"]}"', sev)) - tail = duration - words[-1]["end"] - if tail >= min_silence: - dur = r2(tail) + keep = keep_ms / 1000.0 + for start, end in silence_intervals: + dur = end - start + if dur < min_silence: + continue sev = "high" if dur >= 2.0 else "med" - out.append(_cand("silence", words[-1]["end"], duration, "", - f'{fmt(dur)}s gap after "{words[-1]["word"]}"', sev)) + head = start <= 0.01 + tail = end >= duration - 0.01 + if head: + cut_start, cut_end = start, max(start, end - keep) + nxt = _word_at(words, end, "next") + where = f' before "{nxt["word"]}"' if nxt else " at the head" + elif tail: + cut_start, cut_end = min(end, start + keep), end + prv = _word_at(words, start, "prev") + where = f' after "{prv["word"]}"' if prv else " at the tail" + else: + if dur <= keep: + continue + pad = keep / 2.0 + cut_start, cut_end = start + pad, end - pad + nxt = _word_at(words, end, "next") + where = f' before "{nxt["word"]}"' if nxt else "" + removed = cut_end - cut_start + if removed <= 0.01: + continue + c = _cand("silence", cut_start, cut_end, "", + f"{fmt(r2(dur))}s silence{where}, tighten to " + f"{fmt(r2(dur - removed))}s " + f"(remove {fmt(r2(removed))}s)", sev) + c["silence_start"] = r2(start) + c["silence_end"] = r2(end) + c["keep_ms"] = keep_ms + out.append(c) return out +def snap_candidates(cands, silence_intervals, max_shift): + """Move every candidate edge into an audio-verified silence (pure). + + A cut landing inside real silence CANNOT clip a word, which turns the + "never cut inside a word" rule from an assertion about timestamps into a + structural guarantee. Edges that cannot reach a silence within max_shift + are left alone and marked `snapped: false`, so the gate sees which + candidates still carry timestamp risk instead of the script pretending + they are safe. + + This is also the fix for the audible "n-now" artifact in the first real + cut: a stutter candidate's end came from a pause-absorbed word end, so it + reached into the repeat's onset and clipped it. Snapping puts the edge in + the silence between the two takes. + + Snapping is DIRECTIONAL: a start may only move earlier and an end only + later, so a candidate can widen into surrounding silence (always safe for + a span being removed) but can never invert or collapse onto itself. + """ + for c in cands: + if c["type"] == "silence": + c["snapped"] = True + continue + start = audio.nearest_silence(silence_intervals, c["start"], max_shift, + direction="back") + end = audio.nearest_silence(silence_intervals, c["end"], max_shift, + direction="forward") + c["snapped"] = start is not None and end is not None + if start is not None: + c["start"] = r2(start) + if end is not None: + c["end"] = r2(end) + if c["end"] < c["start"]: + c["end"] = c["start"] + c["dur"] = r2(c["end"] - c["start"]) + return cands + + def _is_sentence_start(words, i): if i == 0: return True @@ -153,14 +372,23 @@ def _is_sentence_start(words, i): return gap >= 0.5 -def detect_fillers(words, nwords): +def detect_fillers(words, nwords, soft_single=None, soft_phrases=None, + hard_fillers=None): + """Hard fillers always, soft (cadence-sensitive) fillers conservatively. + + soft_single/soft_phrases come from the voice bible when one is supplied, + so a creator's connective glue is never flagged. See SOFT_SINGLE's + comment for why the default list is short.""" + hard_fillers = HARD_FILLERS if hard_fillers is None else hard_fillers + soft_single = SOFT_SINGLE if soft_single is None else soft_single + soft_phrases = SOFT_PHRASES if soft_phrases is None else soft_phrases out = [] n = len(words) i = 0 while i < n: - if nwords[i] in HARD_FILLERS: + if nwords[i] in hard_fillers: j = i - while j + 1 < n and nwords[j + 1] in HARD_FILLERS: + while j + 1 < n and nwords[j + 1] in hard_fillers: j += 1 text = " ".join(w["word"] for w in words[i:j + 1]) if j > i: @@ -174,19 +402,20 @@ def detect_fillers(words, nwords): # soft fillers: only at a sentence start if _is_sentence_start(words, i): matched = None - for phrase in SOFT_PHRASES: + for phrase in soft_phrases: L = len(phrase) if nwords[i:i + L] == phrase: matched = (L, " ".join(phrase)) break - if matched is None and nwords[i] in SOFT_SINGLE: + if matched is None and nwords[i] in soft_single: matched = (1, nwords[i]) if matched is not None: L, canon = matched text = " ".join(w["word"] for w in words[i:i + L]) out.append(_cand("filler", words[i]["start"], words[i + L - 1]["end"], text, - f'soft filler "{text}"', "low", cls="soft")) + f'soft filler "{text}" (cadence-sensitive, ' + "default is to KEEP)", "low", cls="soft")) i += L continue i += 1 @@ -264,6 +493,187 @@ def detect_retakes(words, nwords, run_min, window): return out +def _find_section_redo(words, nwords, run_min, window_s): + """Section-scale re-reads: (abandoned_start_i, restart_i, run) (pure). + + The mechanical short-range matcher (_find_verbatim) looks 16 WORDS ahead + for a 3-word repeat, which catches adjacent stumbles and nothing larger. + On the real take it undersized a memory-paragraph redo from about 34s to + about 11s and missed the reset entirely: the creator restarted a whole + section, and the abandoned attempt shipped. + + So this matcher is deliberately the opposite shape: a LONG run (many + words, so it cannot fire on coincidence) inside a locality window + measured in SECONDS rather than words. The seconds constraint is what + separates a redo from a callback. A creator re-reading a paragraph + restarts within a few tens of seconds; a deliberate callback to an + earlier line lands minutes later and must NOT be treated as a retake. + Word-distance cannot express that difference because the abandoned take + plus the reset pause is itself a variable number of words. + + The candidate span runs from the FIRST attempt's start to the RESTART's + start, so it swallows the abandoned take and the reset pause together, + which is the whole point: trimming only the repeated words leaves the + stall in the middle. + """ + n = len(nwords) + out = [] + i = 0 + while i < n: + best = None + j = i + 1 + while j < n and words[j]["start"] - words[i]["start"] <= window_s: + m = 0 + while (i + m < j and j + m < n and nwords[i + m] + and nwords[i + m] == nwords[j + m]): + m += 1 + if m >= run_min and (best is None or m > best[1]): + best = (j, m) + j += 1 + if best is not None: + out.append((i, best[0], best[1])) + i = best[0] + else: + i += 1 + return out + + +def detect_section_redos(words, nwords, run_min, window_s): + out = [] + for a, restart, run in _find_section_redo(words, nwords, run_min, + window_s): + text = " ".join(w["word"] for w in words[a:a + run]) + span = words[restart]["start"] - words[a]["start"] + c = _cand("retake", words[a]["start"], words[restart]["start"], text, + f'section re-read: {run} words repeated after ' + f'{fmt(r2(span))}s, keep the later take and cut the ' + f'abandoned one ("{text[:60]}...")', "high") + c["cls"] = "section" + out.append(c) + return out + + +def _near_reset(words, i, silence_intervals, look_s=DEFAULT_RESET_LOOK_S, + reset_min=DEFAULT_RESET_SILENCE_S): + """True when the take STOPPED next to word i (pure). + + The signal that separates a genuine blooper from scripted usage. Someone + who swears inside a scripted line keeps talking; someone who fluffs a + take stops dead, and that stop is long and audible. + + Measured on the real 2026-07-24 take, which contains one of each: + "that damn term" (scripted) largest adjacent silence 0.77s + "Oh fuck." (blooper) followed by 7.65s of silence + So the discriminator is not "is there a pause nearby", which is true of + almost any word in natural speech, but "is there a LONG stop". An + earlier version of this asked only for 0.5s within 3s and duly marked + the scripted line as almost certainly a flub. + + Proximity is measured span-to-span with overlap counting as adjacent, + because parakeet absorbs a pause into the preceding word's end: the + blooper's word span literally overlaps the silence that follows it, and + an edge-to-edge comparison misses exactly the case that matters. + """ + w = words[i] + for start, end in silence_intervals: + if end - start < reset_min: + continue + distance = max(start - w["end"], w["start"] - end) + if distance <= look_s: + return True + return False + + +def detect_bloopers(words, nwords, silence_intervals, blooper_words, + blooper_phrases, look_s=DEFAULT_RESET_LOOK_S): + """Expletives and reset phrases as their own candidate type. + + The real take contained an explicit "Oh fuck." at 13:59 that the + mechanical pass left in; it would have shipped. The retake detector never + saw it because CUE_PHRASES has no expletives, and the filler detector + never saw it because it is not a filler. + + Severity encodes the scripted-versus-genuine judgment rather than + guessing it away: a marker next to a long pause is a near-certain blooper + (high), one in continuous speech might be deliberate (med, flagged for an + ear). Nothing here is auto-cut; these are gate-2 candidates like + everything else. + """ + out = [] + n = len(words) + i = 0 + while i < n: + matched = None + for phrase in blooper_phrases: + L = len(phrase) + if nwords[i:i + L] == phrase: + matched = (L, " ".join(phrase)) + break + if matched is None and nwords[i] in blooper_words: + matched = (1, nwords[i]) + if matched is None: + i += 1 + continue + L, canon = matched + text = " ".join(w["word"] for w in words[i:i + L]) + reset = _near_reset(words, i, silence_intervals, look_s) + if reset: + reason = f'blooper "{text}" next to a pause; almost certainly a flub' + sev = "high" + else: + reason = (f'"{text}" in continuous speech; may be scripted usage, ' + "confirm by ear before cutting") + sev = "med" + c = _cand("blooper", words[i]["start"], words[i + L - 1]["end"], text, + reason, sev) + c["cls"] = "reset" if reset else "ambiguous" + out.append(c) + i += L + return out + + +def parse_voice_cadence(text): + """Read keep/cut word lists out of a voice bible's cadence block (pure). + + The voice bible is the creator's taste file, so their connective words + belong there and not in this script's constants. The block is a fenced + markdown code block tagged `cadence`: + + ```cadence + keep: so, here's the thing, which means, look + cut: um, uh, hmm, basically + ``` + + keep wins over cut on conflict: preserving a speaker's rhythm is the + safer failure, since a kept filler is a small blemish and a cut cadence + word changes how they sound. Returns (keep_set, cut_set) of normalized + phrases; a file with no cadence block yields two empty sets, which leaves + the conservative defaults in place. + """ + keep, cut = set(), set() + in_block = False + for line in text.splitlines(): + stripped = line.strip() + if stripped.startswith("```"): + tag = stripped[3:].strip().lower() + if in_block: + in_block = False + elif tag == "cadence": + in_block = True + continue + if not in_block or ":" not in stripped: + continue + key, _, value = stripped.partition(":") + target = {"keep": keep, "cut": cut}.get(key.strip().lower()) + if target is None: + continue + for item in value.split(","): + phrase = " ".join(norm(w) for w in item.split()) + if phrase: + target.add(phrase) + return keep, cut - keep + + def detect_markers(words, nwords, cues): out = [] n = len(words) @@ -285,24 +695,53 @@ def detect_markers(words, nwords, cues): return out -def build(data, min_silence, retake_window, retake_run, marker_cues): +CANDIDATE_TYPES = ("silence", "filler", "stutter", "retake", "blooper", + "marker") + + +def build(data, min_silence, retake_window, retake_run, marker_cues, + silence_intervals, snap_ms=DEFAULT_SNAP_MS, + keep_ms=DEFAULT_KEEP_MS, section_run=DEFAULT_SECTION_RUN, + section_window_s=DEFAULT_SECTION_WINDOW_S, + voice_keep=(), voice_cut=(), + blooper_words=None, blooper_phrases=None): words = data["words"] duration = data["duration"] nwords = [norm(w["word"]) for w in words] + # Voice bible overlay: keep wins over cut, and a kept word is removed + # from every filler list including the hard ones (a creator who says + # "hmm" deliberately gets to keep it). + keep = set(voice_keep) + hard = {w for w in HARD_FILLERS if w not in keep} + soft_single = ({w for w in SOFT_SINGLE if w not in keep} + | {c for c in voice_cut if " " not in c and c not in keep}) + soft_phrases = [p for p in SOFT_PHRASES if " ".join(p) not in keep] + soft_phrases += [c.split() for c in voice_cut + if " " in c and c not in keep] + cands = [] - cands += detect_silence(words, duration, min_silence) - cands += detect_fillers(words, nwords) + cands += detect_silence(words, duration, min_silence, silence_intervals, + keep_ms=keep_ms) + cands += detect_fillers(words, nwords, soft_single=soft_single, + soft_phrases=soft_phrases, hard_fillers=hard) cands += detect_stutter(words, nwords) cands += detect_retakes(words, nwords, retake_run, retake_window) + cands += detect_section_redos(words, nwords, section_run, section_window_s) + cands += detect_bloopers( + words, nwords, silence_intervals, + BLOOPER_WORDS if blooper_words is None else blooper_words, + BLOOPER_PHRASES if blooper_phrases is None else blooper_phrases) cands += detect_markers(words, nwords, marker_cues) + snap_candidates(cands, silence_intervals, snap_ms / 1000.0) cands.sort(key=lambda c: (c["start"], c["end"], c["type"])) - counts = {t: 0 for t in ("silence", "filler", "stutter", "retake", "marker")} + counts = {t: 0 for t in CANDIDATE_TYPES} for c in cands: counts[c["type"]] += 1 + trimmable = sum(c["dur"] for c in cands if c["type"] == "silence") return { "media": data.get("media", ""), "duration": duration, @@ -310,23 +749,91 @@ def build(data, min_silence, retake_window, retake_run, marker_cues): "min_silence": min_silence, "retake_window": retake_window, "retake_run": retake_run, + "snap_ms": snap_ms, + "keep_ms": keep_ms, + "section_run": section_run, + "section_window_s": section_window_s, }, + "silence_source": "audio", + "voice_bible_applied": bool(voice_keep or voice_cut), "counts": counts, + "trimmable_silence_seconds": r2(trimmable), + "unsnapped": sum(1 for c in cands if not c.get("snapped")), "candidates": cands, } +# Arguments the SKILL owns and a configured flags string must never reach. +# The skill appends [cut] cutplan-flags AFTER these, and argparse lets a later +# occurrence win, so without this the studio config could redirect the output +# or, far worse, point --audio-map somewhere else and quietly break the +# two-source rule the whole stage rests on. The config comment states the +# boundary; this enforces it. +PROTECTED_FLAGS = ("-o", "--output", "--audio-map", "--voice-bible") + + +def protected_conflicts(argv): + """Protected flags supplied more than once (pure).""" + counts = {} + for token in argv: + name = str(token).split("=", 1)[0] + if name in PROTECTED_FLAGS: + counts[name] = counts.get(name, 0) + 1 + conflicts = {f for f, n in counts.items() if n > 1} + # -o and --output are the same destination under two spellings. + if counts.get("-o", 0) + counts.get("--output", 0) > 1: + conflicts.update({"-o", "--output"} & set(counts)) + return sorted(conflicts) + + def main(argv=None): + argv = list(sys.argv[1:] if argv is None else argv) + conflicts = protected_conflicts(argv) + if conflicts: + print("cutplan: " + ", ".join(conflicts) + " supplied more than once. " + "These are passed by the skill and must not be overridden from " + "cutplan-flags: redirecting the output or the audio map from a " + "config file breaks the two-source rule silently.", + file=sys.stderr) + return 2 + p = argparse.ArgumentParser(description="Find mechanical cut candidates in a " "word-level transcript.") p.add_argument("words", help="path to words.json (from transcribe.py)") p.add_argument("-o", "--output", required=True, help="path to candidates.json") - p.add_argument("--min-silence", type=float, default=0.7) + p.add_argument("--min-silence", type=float, default=DEFAULT_MIN_SILENCE, + help=f"shortest interior silence to tighten (default " + f"{DEFAULT_MIN_SILENCE}; below this is cadence, not " + "dead air)") + p.add_argument("--keep-ms", type=int, default=DEFAULT_KEEP_MS, + help=f"breathing room left in a tightened silence " + f"(default {DEFAULT_KEEP_MS})") p.add_argument("--retake-window", type=int, default=16) p.add_argument("--retake-run", type=int, default=3) + p.add_argument("--section-run", type=int, default=DEFAULT_SECTION_RUN, + help=f"words that must repeat to count as a section " + f"re-read (default {DEFAULT_SECTION_RUN})") + p.add_argument("--section-window-s", type=float, + default=DEFAULT_SECTION_WINDOW_S, + help=f"seconds within which a repeat is a redo rather " + f"than a callback (default {DEFAULT_SECTION_WINDOW_S})") + p.add_argument("--voice-bible", default=None, + help="path to the creator's voice bible; its ```cadence " + "block's keep:/cut: lists override the built-in " + "soft-filler defaults") + p.add_argument("--blooper-cues", default=None, + help="comma-separated override for the blooper vocabulary " + "(expletives and reset phrases)") p.add_argument("--marker-cues", default="question from the interviewer", help='comma-separated marker phrases (legacy alternative: ' '"question from claude")') + p.add_argument("--audio-map", required=True, + help="path to cut/audio-map.json from analyze_audio.py; " + "REQUIRED, silence comes from the audio and there is " + "no transcript-gap fallback (see detect_silence)") + p.add_argument("--snap-ms", type=int, default=DEFAULT_SNAP_MS, + help=f"how far a candidate edge may move to land inside an " + f"audio silence (default {DEFAULT_SNAP_MS})") args = p.parse_args(argv) try: @@ -340,11 +847,53 @@ def main(argv=None): file=sys.stderr) return 1 + try: + with open(args.audio_map, encoding="utf-8") as f: + audio_map = json.load(f) + except (OSError, json.JSONDecodeError) as e: + print(f"cutplan: cannot read audio map {args.audio_map}: {e}. " + "Generate it first with analyze_audio.py; silence must come " + "from the audio, not from transcript gaps.", file=sys.stderr) + return 1 + if "silence" not in audio_map: + print("cutplan: audio map has no 'silence' key; regenerate it with " + "analyze_audio.py", file=sys.stderr) + return 1 + silence_intervals = audio.to_pairs(audio_map["silence"]) + marker_cues = [norm_phrase for c in args.marker_cues.split(",") if (norm_phrase := " ".join(norm(w) for w in c.split()))] + voice_keep, voice_cut = set(), set() + if args.voice_bible: + try: + voice_keep, voice_cut = parse_voice_cadence( + Path(args.voice_bible).read_text(encoding="utf-8")) + except OSError as e: + print(f"cutplan: cannot read voice bible {args.voice_bible}: {e}", + file=sys.stderr) + return 1 + if not voice_keep and not voice_cut: + print(f"cutplan: no ```cadence block found in " + f"{args.voice_bible}; using the conservative built-in " + "soft-filler defaults. See the voice bible spec.", + file=sys.stderr) + + blooper_words, blooper_phrases = None, None + if args.blooper_cues is not None: + cues = [phrase for c in args.blooper_cues.split(",") + if (phrase := " ".join(norm(w) for w in c.split()))] + blooper_words = {c for c in cues if " " not in c} + blooper_phrases = [c.split() for c in cues if " " in c] + result = build(data, args.min_silence, args.retake_window, - args.retake_run, marker_cues) + args.retake_run, marker_cues, silence_intervals, + snap_ms=args.snap_ms, keep_ms=args.keep_ms, + section_run=args.section_run, + section_window_s=args.section_window_s, + voice_keep=voice_keep, voice_cut=voice_cut, + blooper_words=blooper_words, + blooper_phrases=blooper_phrases) try: with open(args.output, "w", encoding="utf-8") as f: @@ -355,7 +904,18 @@ def main(argv=None): return 1 print(json.dumps({"ok": True, "output": args.output, - "counts": result["counts"]})) + "silence_source": "audio", + "voice_bible_applied": result["voice_bible_applied"], + "counts": result["counts"], + "trimmable_silence_seconds": + result["trimmable_silence_seconds"], + "unsnapped": result["unsnapped"]})) + if result["unsnapped"]: + print(f"note: {result['unsnapped']} candidate(s) could not reach an " + f"audio silence within {args.snap_ms}ms and carry " + '"snapped": false. Their edges still rest on transcript ' + "timestamps; check them by ear before cutting.", + file=sys.stderr) return 0 diff --git a/skills/mc-cut/scripts/edited_transcript.py b/skills/mc-cut/scripts/edited_transcript.py new file mode 100644 index 0000000..7e5745a --- /dev/null +++ b/skills/mc-cut/scripts/edited_transcript.py @@ -0,0 +1,222 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.11" +# /// +"""Reconstruct what the viewer actually hears: words.json intersected with the +kept EDL segments, carrying BOTH clean and source timecodes. + +Usage: + uv run {skill-root}/scripts/edited_transcript.py \ + {projects-path}//transcript/words.json \ + --edl {projects-path}//cut/edl.json \ + -o {projects-path}//cut/edited-transcript.md \ + [-j {projects-path}//cut/edited-words.json] + +Why this exists: + The content-editorial pass (references/editorial-pass.md) reads the + delivered piece as an argument and recommends content cuts. It must read + what SURVIVED THE CUT, not the script and not the raw transcript: + - the script is what was planned, and delivery diverges from it + (ad-libs, dropped lines, live rewrites; the first real project + diverged from its script substantially); + - the raw transcript still contains everything the mechanical cut + removed, so a pass reading it would review words the viewer never + hears. + + DUAL TIMECODES are the other half of the point. A recommendation is read + by a human against the preview, which runs on CLEAN time, but applying it + means editing the EDL, which is in SOURCE time. Every finding therefore + needs both, and deriving one from the other by hand is exactly where the + first editorial pass went wrong: its transcript-read source estimates + drifted to the wrong windows, and a blind apply would have cut the wrong + spans. Emitting both here removes that whole failure mode. + +Contract: + input words.json (from transcribe.py) and cut/edl.json. + keep rule a word is kept when its MIDPOINT falls inside a kept segment, + so a word straddling a cut boundary belongs to whichever side + holds most of it and is never emitted twice. + output -o markdown: the edited transcript in reading order, broken + into paragraphs at pauses, each paragraph headed with its clean + and source timecodes. + json -j optional: {"clean_duration", "words": [{word, clean_start, + clean_end, src_start, src_end, segment}]}, the machine-readable + form for any pass that needs to map a finding back to the EDL. + summary json.dumps on stdout: kept/dropped word counts, clean duration. + +Exit codes: 0 ok, 1 failure, 2 usage error. + +STATUS: implemented (pure logic covered by +scripts/tests/test-edited_transcript.py). +""" + +import argparse +import json +import sys +from pathlib import Path + +# A pause at or above this opens a new paragraph in the readable output. +PARAGRAPH_GAP_S = 0.9 + + +def build_map(edl): + """EDL segments as [{source, start, end, offset}] in timeline order (pure).""" + out = [] + offset = 0.0 + for seg in edl["segments"]: + out.append({"source": seg["source"], "start": seg["start"], + "end": seg["end"], "offset": offset}) + offset += seg["end"] - seg["start"] + return out + + +def clean_duration(mapping): + """Total length of the edited timeline (pure).""" + return sum(s["end"] - s["start"] for s in mapping) + + +def keep_words(words, mapping, source=None): + """The words that survive the cut, with dual timecodes (pure). + + A word is kept when its midpoint lands in a kept segment: a word + straddling a boundary belongs to whichever side holds most of it, and is + emitted exactly once. Clean times are the word's own span shifted by its + segment's offset, clamped into the segment so a straddling word never + reports a clean time outside the timeline. + """ + out = [] + for w in words: + mid = (float(w["start"]) + float(w["end"])) / 2.0 + for i, s in enumerate(mapping): + if source is not None and s["source"] != source: + continue + if not (s["start"] <= mid <= s["end"]): + continue + cs = s["offset"] + (max(float(w["start"]), s["start"]) - s["start"]) + ce = s["offset"] + (min(float(w["end"]), s["end"]) - s["start"]) + out.append({ + "word": w["word"], + "clean_start": round(cs, 3), + "clean_end": round(max(cs, ce), 3), + "src_start": round(float(w["start"]), 3), + "src_end": round(float(w["end"]), 3), + "segment": i, + }) + break + return out + + +def tc(seconds): + """Seconds to m:ss (pure).""" + seconds = max(0.0, float(seconds)) + return f"{int(seconds // 60)}:{seconds % 60:05.2f}" + + +def paragraphs(kept, gap=PARAGRAPH_GAP_S): + """Group kept words into paragraphs at pauses or segment changes (pure). + + Breaking on a segment change matters as much as breaking on a pause: a + seam is where the editorial pass most needs to see whether the sentence + still reads, and burying it mid-paragraph hides exactly that. + """ + out = [] + current = [] + for i, w in enumerate(kept): + if current: + prev = kept[i - 1] + if (w["clean_start"] - prev["clean_end"] >= gap + or w["segment"] != prev["segment"]): + out.append(current) + current = [] + current.append(w) + if current: + out.append(current) + return out + + +def render_markdown(kept, mapping, gap=PARAGRAPH_GAP_S): + """The readable edited transcript (pure).""" + total = clean_duration(mapping) + lines = [ + "# Edited transcript", + "", + "What the viewer actually hears: `transcript/words.json` intersected " + "with the kept segments of `cut/edl.json`.", + "", + f"Runtime {tc(total)} ({round(total, 2)}s), {len(kept)} words.", + "", + "Each paragraph is headed `clean-time (src source-time)`. Quote a " + "finding by CLEAN time for the human at the gate, and apply it by " + "SOURCE time against the EDL. Never convert between them by hand.", + "", + ] + for para in paragraphs(kept, gap): + head = para[0] + lines.append(f"**{tc(head['clean_start'])}** " + f"(src {tc(head['src_start'])})") + lines.append("") + lines.append(" ".join(w["word"] for w in para)) + lines.append("") + return "\n".join(lines) + + +def main(argv=None): + p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + p.add_argument("words", help="path to transcript/words.json") + p.add_argument("--edl", required=True, help="path to cut/edl.json") + p.add_argument("-o", "--output", required=True, + help="path for the readable edited transcript markdown") + p.add_argument("-j", "--json", default=None, + help="optional path for the machine-readable kept words") + p.add_argument("--source", default=None, + help="restrict to one EDL source (multi-source projects)") + p.add_argument("--paragraph-gap", type=float, default=PARAGRAPH_GAP_S) + args = p.parse_args(argv) + + try: + with open(args.words, encoding="utf-8") as f: + transcript = json.load(f) + with open(args.edl, encoding="utf-8") as f: + edl = json.load(f) + except (OSError, json.JSONDecodeError) as e: + print(f"edited_transcript: cannot read input: {e}", file=sys.stderr) + return 2 + if "words" not in transcript: + print("edited_transcript: transcript has no 'words' key", + file=sys.stderr) + return 2 + if not edl.get("segments"): + print("edited_transcript: edl has no segments", file=sys.stderr) + return 2 + + mapping = build_map(edl) + words = transcript["words"] + kept = keep_words(words, mapping, args.source) + + output = Path(args.output) + output.parent.mkdir(parents=True, exist_ok=True) + output.write_text(render_markdown(kept, mapping, args.paragraph_gap), + encoding="utf-8") + + if args.json: + jout = Path(args.json) + jout.parent.mkdir(parents=True, exist_ok=True) + jout.write_text(json.dumps({ + "clean_duration": round(clean_duration(mapping), 3), + "segments": len(mapping), + "words": kept, + }, indent=2, ensure_ascii=False) + "\n", encoding="utf-8") + + print(json.dumps({ + "ok": True, + "output": str(output), + "words_kept": len(kept), + "words_dropped": len(words) - len(kept), + "clean_duration": round(clean_duration(mapping), 3), + "segments": len(mapping), + })) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/mc-cut/scripts/normalize_source.py b/skills/mc-cut/scripts/normalize_source.py new file mode 100644 index 0000000..1babbf4 --- /dev/null +++ b/skills/mc-cut/scripts/normalize_source.py @@ -0,0 +1,355 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.11" +# /// +"""Corrective spatial normalize: crop out baked-in frame defects and reframe. + +Usage: + uv run {skill-root}/scripts/normalize_source.py {projects-path}//raw/ \ + -o {projects-path}//raw/-normalized.mp4 \ + [--auto | --crop W:H:X:Y] [--offset-x N] [--offset-y N] \ + [--target-aspect 16:9] [--output-size WxH] + +Why this exists: + A 4K take arrived with a decorative frame RECORDED INTO THE PIXELS (a + black outer border plus a rounded orange ring, an Ecamm frame effect left + on during capture) and the subject framed about 5 percent left of centre. + The pipeline had no way to fix it. cut/edl.json is purely TEMPORAL, + {source, start, end}, and the renderers bake those spans with no crop, + scale or position transform anywhere. So a baked-in border could only be + corrected by hand, or downstream in the creator's own editor. + + This is the missing spatial stage. + +THE PROPERTY THAT MATTERS: THIS TOUCHES NO TIMECODES. + A spatial crop moves nothing in time. Every frame keeps its presentation + timestamp, the duration is unchanged, and the audio is copied through + untouched. So an existing transcript, EDL, cutplan, beat table and + boundary-frame set all stay valid against the corrected master. + + DO NOT re-transcribe. DO NOT re-cut. DO NOT re-plan beats. An agent that + "helpfully" rebuilds them after a normalize is throwing away the + creator's approved work for no reason. The script asserts the duration + is preserved and refuses to publish if it is not, so this is a checked + guarantee rather than a promise. + +Where it belongs in the pipeline: + Immediately after preflight, BEFORE beats and graphics. Overlays are + positioned against the canvas, so normalizing after graphics would force + every overlay to be repositioned. The corrected master is registered as + the project source (project.json `sources`, `cfr_master`) and every later + step inherits it automatically. + + Keep this distinct from CREATIVE reframing. Corrective normalize is + global, defect-driven, and belongs here. Motion zooms and punch-ins are + per-moment emphasis and belong in the beats stage, on the already-clean + canvas. + +Contract: + input a media file, plus a crop rectangle: --crop W:H:X:Y explicitly, + or --auto to infer it from the same border detection preflight + uses. + reframe --offset-x / --offset-y pan the crop window (positive is right + and down) to recentre an off-centre subject, clamped so the + window never leaves the frame. + aspect --target-aspect (e.g. 16:9) shrinks the crop to an exact + aspect, so a corrected master is never subtly non-standard. + size --output-size WxH scales the cropped result back up, so the + delivery resolution is unchanged by the correction. Omit to + keep the cropped size. + output a corrected CFR master at -o, with the source's frame rate and + its audio stream copied. The path is reported as + `normalized_master` for the caller to record in project.json. + summary json.dumps on stdout: source, output, crop, output_size, + source_duration, output_duration, timecodes_preserved. + +Exit codes: 0 ok, 1 failure (probe, encode, or a duration change), 2 usage. + +STATUS: implemented (crop geometry covered by +scripts/tests/test-normalize_source.py; encode path covered by its +synthesized-fixture integration test). +""" + +import argparse +import json +import subprocess +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent)) # noqa: E402 +import composite_core as core # noqa: E402 +import preflight # noqa: E402 + +# A normalize that changes the duration has broken the one guarantee this +# script makes, so the tolerance is tight: a frame or two, not a second. +DURATION_TOLERANCE_S = 0.1 + + +def parse_crop(text): + """'W:H:X:Y' to an (x, y, w, h) rectangle (pure). Raises ValueError.""" + parts = text.split(":") + if len(parts) != 4: + raise ValueError("crop must be W:H:X:Y") + try: + w, h, x, y = (int(p) for p in parts) + except ValueError: + raise ValueError("crop values must be integers") from None + if w <= 0 or h <= 0 or x < 0 or y < 0: + raise ValueError("crop must have positive size and non-negative origin") + return (x, y, w, h) + + +def parse_size(text): + """'WxH' to an (w, h) pair (pure). Raises ValueError.""" + parts = text.lower().split("x") + if len(parts) != 2: + raise ValueError("size must be WxH") + try: + w, h = (int(p) for p in parts) + except ValueError: + raise ValueError("size values must be integers") from None + if w <= 0 or h <= 0: + raise ValueError("size must be positive") + return (w, h) + + +def parse_aspect(text): + """'16:9' or '1.777' to a float (pure). Raises ValueError.""" + if ":" in text: + a, _, b = text.partition(":") + num, den = float(a), float(b) + if den == 0: + raise ValueError("aspect denominator must not be zero") + return num / den + value = float(text) + if value <= 0: + raise ValueError("aspect must be positive") + return value + + +def shift_crop(rect, dx, dy, frame_w, frame_h): + """Pan a crop window, clamped inside the frame (pure). + + Clamping rather than erroring is deliberate: recentring is a taste + adjustment the creator nudges, and a nudge that would run off the edge + should stop at the edge, not fail the run. + """ + x, y, w, h = rect + x = max(0, min(int(round(x + dx)), max(0, frame_w - w))) + y = max(0, min(int(round(y + dy)), max(0, frame_h - h))) + return (x, y, w, h) + + +def fit_aspect(rect, aspect): + """Shrink a rect to an exact aspect, keeping its centre (pure). + + Only ever shrinks. Growing could pull the baked-in border back into the + frame, which is the entire thing being removed. + """ + x, y, w, h = rect + if aspect <= 0: + return rect + target_w = core.even(min(w, h * aspect)) + target_h = core.even(min(h, w / aspect)) + if target_w / max(1, target_h) > aspect: + target_w = core.even(target_h * aspect) + else: + target_h = core.even(target_w / aspect) + nx = x + (w - target_w) // 2 + ny = y + (h - target_h) // 2 + return (max(0, core.even(nx)), max(0, core.even(ny)), + max(2, target_w), max(2, target_h)) + + +def clamp_rect(rect, frame_w, frame_h): + """Keep a rect inside the frame, with even dimensions (pure).""" + x, y, w, h = rect + w = core.even(max(2, min(w, frame_w))) + h = core.even(max(2, min(h, frame_h))) + x = max(0, min(core.even(x), frame_w - w)) + y = max(0, min(core.even(y), frame_h - h)) + return (x, y, w, h) + + +def build_normalize_command(src, dst, rect, fps, output_size=None, + encoder="libx264", crf=18, height=None): + """ffmpeg argv cropping (and optionally rescaling) a source (pure). + + fps is forced so the corrected master stays constant frame rate, and the + audio is stream-copied: no re-encode, no resample, no drift. + """ + x, y, w, h = rect + vf = [f"crop={w}:{h}:{x}:{y}"] + if output_size: + ow, oh = output_size + vf.append(f"scale={core.even(ow)}:{core.even(oh)}") + vf.append(f"fps={fps}") + argv = ["ffmpeg", "-y", "-hide_banner", *core.encoder_init_flags(encoder), + "-i", str(src)] + chain = ",".join(vf) + if core.encoder_needs_hwupload(encoder): + chain += ",format=nv12,hwupload" + argv += ["-vf", chain] + if core.is_hardware_encoder(encoder): + argv += ["-c:v", encoder, + "-b:v", f"{preflight.master_bitrate_for(height or h)}k"] + if encoder.endswith("_videotoolbox"): + argv += ["-allow_sw", "1"] + else: + argv += ["-c:v", encoder, "-crf", str(crf), "-preset", "medium"] + if not core.encoder_needs_hwupload(encoder): + argv += ["-pix_fmt", "yuv420p"] + argv += ["-c:a", "copy", "-movflags", "+faststart", str(dst)] + return argv + + +def auto_rect(media, duration, width, height, samples): + """Infer the active-content rectangle, or None when the source is clean. + + Shares preflight's detector so --auto corrects exactly what QC halted on; + two implementations would inevitably disagree about where the border is. + """ + verdict = preflight.qc_source(media, duration, width, height, + samples=samples) + rect = verdict.get("active_rect") + return (tuple(rect) if rect else None), verdict + + +def main(argv=None): + p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + p.add_argument("media", help="path to the source media file") + p.add_argument("-o", "--output", required=True, + help="path for the corrected master") + p.add_argument("--crop", default=None, + help="explicit crop rectangle W:H:X:Y") + p.add_argument("--auto", action="store_true", + help="infer the crop from the detected border ring") + p.add_argument("--offset-x", type=int, default=0, + help="pan the crop window right (negative for left) to " + "recentre the subject") + p.add_argument("--offset-y", type=int, default=0, + help="pan the crop window down (negative for up)") + p.add_argument("--target-aspect", default=None, + help="shrink the crop to an exact aspect, e.g. 16:9") + p.add_argument("--output-size", default=None, + help="scale the cropped result to WxH (default: keep the " + "cropped size)") + p.add_argument("--qc-samples", type=int, + default=preflight.DEFAULT_QC_SAMPLES) + p.add_argument("--crf", type=int, default=18) + args = p.parse_args(argv) + + media = Path(args.media) + if not media.is_file(): + print(f"normalize_source: media not found: {media}", file=sys.stderr) + return 2 + if bool(args.crop) == bool(args.auto): + print("normalize_source: pass exactly one of --crop or --auto", + file=sys.stderr) + return 2 + + info = preflight.probe_media(media) + if info is None: + print(f"normalize_source: cannot probe {media}", file=sys.stderr) + return 1 + width, height = info["width"], info["height"] + duration = info["duration"] + fps = preflight.nearest_standard_rate(info["avg_frame_rate"]) + + qc = None + if args.auto: + rect, qc = auto_rect(media, duration, width, height, args.qc_samples) + if rect is None: + print("normalize_source: no border ring detected; nothing to " + "correct. Pass --crop explicitly to reframe anyway.", + file=sys.stderr) + return 1 + else: + try: + rect = parse_crop(args.crop) + except ValueError as e: + print(f"normalize_source: {e}", file=sys.stderr) + return 2 + + if args.offset_x or args.offset_y: + rect = shift_crop(rect, args.offset_x, args.offset_y, width, height) + if args.target_aspect: + try: + rect = fit_aspect(rect, parse_aspect(args.target_aspect)) + except ValueError as e: + print(f"normalize_source: {e}", file=sys.stderr) + return 2 + rect = clamp_rect(rect, width, height) + + output_size = None + if args.output_size: + try: + output_size = parse_size(args.output_size) + except ValueError as e: + print(f"normalize_source: {e}", file=sys.stderr) + return 2 + + output = Path(args.output) + output.parent.mkdir(parents=True, exist_ok=True) + encoder = core.pick_encoder("auto") + tmp = output.with_name(f".{output.stem}.normalizing{output.suffix}") + + x, y, w, h = rect + print(f"normalize_source: crop={w}:{h}:{x}:{y} from {width}x{height} " + f"at {fps} ({encoder})", file=sys.stderr) + cmd = build_normalize_command(media, tmp, rect, fps, output_size, + encoder, args.crf, height) + proc = subprocess.run(cmd, capture_output=True, text=True) + if proc.returncode != 0: + core.discard_render(tmp) + print("normalize_source: encode failed:", file=sys.stderr) + print(" ".join(cmd), file=sys.stderr) + print(proc.stderr.strip()[-2000:], file=sys.stderr) + return 1 + + # The load-bearing assertion: a spatial correction must not have moved + # anything in time, or every downstream artifact silently desyncs. + out_duration = core.probe_duration(tmp) + if out_duration is None: + core.discard_render(tmp) + print("normalize_source: corrected master has no readable duration", + file=sys.stderr) + return 1 + drift = abs(out_duration - (duration or out_duration)) + if drift > DURATION_TOLERANCE_S: + core.discard_render(tmp) + print(f"normalize_source: duration changed by {drift:.3f}s " + f"({duration:.3f}s to {out_duration:.3f}s). A spatial crop must " + "not move anything in time; refusing to publish, because the " + "existing transcript and EDL would silently desync against " + "this master.", file=sys.stderr) + return 1 + + core.publish_render(tmp, output) + + final_w, final_h = (output_size if output_size else (w, h)) + summary = { + "source": str(media), + "normalized_master": str(output.resolve()), + "source_size": [width, height], + "crop": f"{w}:{h}:{x}:{y}", + "output_size": [core.even(final_w), core.even(final_h)], + "fps": fps, + "source_duration": round(duration, 3) if duration else None, + "output_duration": round(out_duration, 3), + "timecodes_preserved": True, + "auto_detected": bool(args.auto), + } + if qc: + summary["qc_defects"] = qc["defects"] + print(json.dumps(summary, indent=2)) + print("\nRegister this as the project source (project.json `sources` /" + " `cfr_master`) so every later step inherits it.\n" + "Timecodes are unchanged: the existing transcript, EDL, cutplan " + "and beat table stay valid. Do NOT re-transcribe or re-cut.", + file=sys.stderr) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/mc-cut/scripts/preflight.py b/skills/mc-cut/scripts/preflight.py index 52bd1ce..b10cbe0 100644 --- a/skills/mc-cut/scripts/preflight.py +++ b/skills/mc-cut/scripts/preflight.py @@ -33,16 +33,24 @@ it in project.json sources as the project source of truth; every later step (transcription, EDL times, renders, timeline export) must use it, never the VFR original. - qc with --qc-frames , the first and last frame of each source - are extracted as -first.jpg / -last.jpg for the - source QC pass (edge defects, wrong aspect, letterboxed or - cropped content), inspected before any render is built. + qc edge-defect QC that ASSERTS AND HALTS (exit 3). Several frames + across each take (not just the first and last: a frame effect + can be switched on mid-recording) are analysed for a flat + decorative border ring and for an active area whose aspect does + not match the container. On a hit the inferred active-content + rectangle is reported and the stage stops, because a baked-in + border cannot be fixed downstream: the EDL is time-only, so + every later stage inherits the bad canvas. Fix with + normalize_source.py, or pass --allow-qc-defects when the + framing is intentional. --qc-frames additionally writes + the sampled stills for the creator to eyeball. summary json.dumps on stdout: per-file {path, codec, width, height, - duration, fps, vfr, cfr_master, qc_frames}, plus - {"disk": {free_bytes, needed_bytes, ok}} and "all_cfr". + duration, fps, vfr, cfr_master, qc_frames, qc}, plus + {"disk": {free_bytes, needed_bytes, ok}}, "all_cfr" and "qc_ok". Exit codes: 0 ok (VFR found still exits 0; the caller reads "vfr" and -"all_cfr"), 1 probe/remux failure or disk refusal of a planned remux, 2 usage. +"all_cfr"), 1 probe/remux failure or disk refusal of a planned remux, 2 +usage, 3 source QC defect (stop and get the creator's call). STATUS: implemented (pure logic covered by scripts/tests/test-preflight.py; probe/remux path covered by the synthesized-fixture integration test there). @@ -55,12 +63,24 @@ from fractions import Fraction from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent)) -import composite_core as core +sys.path.insert(0, str(Path(__file__).resolve().parent)) # noqa: E402 +import composite_core as core # noqa: E402 STANDARD_RATES = ("24000/1001", "24/1", "25/1", "30000/1001", "30/1", "50/1", "60000/1001", "60/1") +# Frames sampled across a take for the edge-defect QC pass. More than two, +# because a frame effect can be switched on after recording starts. +DEFAULT_QC_SAMPLES = 7 +# Analysis thumbnail size. Small on purpose: downscaling averages out sensor +# noise so a flat decorative border reads as genuinely flat. +QC_ANALYSIS_SIZE = (96, 54) +# Per-channel variance under which a ring of pixels counts as a flat colour. +DEFAULT_RING_VARIANCE = 12.0 +# How far a border's colour must sit from the picture inside it before it is +# a decorative frame rather than just a dark scene. +DEFAULT_RING_DISTANCE = 24.0 + def parse_rate(text): """'30000/1001' or '30' -> Fraction, None for unknown/zero sentinels.""" @@ -156,19 +176,36 @@ def probe_media(path): } -def extract_qc_frames(media, qc_dir, duration): - """First and last frame stills for the edge-defect QC pass.""" +def qc_sample_times(duration, samples=DEFAULT_QC_SAMPLES): + """Timestamps to inspect across a take (pure). + + Two frames is not a QC pass. The original check looked at frame 0 and a + frame half a second from the end, which cannot see a defect that starts + mid-take (a frame effect toggled on after recording began). These sample + evenly across the interior, avoiding the very first and last frames where + fades and encoder warm-up live. + """ + if not duration or duration <= 0: + return [0.0] + samples = max(2, samples) + lo, hi = duration * 0.02, duration * 0.98 + if hi <= lo: + return [max(0.0, duration / 2)] + step = (hi - lo) / (samples - 1) + return [round(lo + i * step, 3) for i in range(samples)] + + +def extract_qc_frames(media, qc_dir, duration, times=None): + """QC stills across the take, for the creator to eyeball after a halt.""" qc_dir.mkdir(parents=True, exist_ok=True) stem = Path(media).stem written = [] - jobs = [("first", ["-i", str(media)])] - if duration and duration > 0.5: - jobs.append(("last", ["-sseof", "-0.5", "-i", str(media)])) - for name, input_args in jobs: - dest = qc_dir / f"{stem}-{name}.jpg" + for i, t in enumerate(times if times is not None + else qc_sample_times(duration)): + dest = qc_dir / f"{stem}-qc{i:02d}.jpg" proc = subprocess.run( - ["ffmpeg", "-y", *input_args, "-frames:v", "1", "-q:v", "3", - "-update", "1", str(dest)], + ["ffmpeg", "-y", "-ss", f"{t:.3f}", "-i", str(media), + "-frames:v", "1", "-q:v", "3", "-update", "1", str(dest)], capture_output=True, text=True, ) if proc.returncode == 0 and dest.is_file(): @@ -176,6 +213,209 @@ def extract_qc_frames(media, qc_dir, duration): return written +# --- source QC: edge defects ------------------------------------------------ +# +# On the first real project a 4K take carried a decorative frame RECORDED +# INTO THE PIXELS: a black outer border plus a rounded orange ring, an Ecamm +# frame effect left on during capture, with the subject framed off-centre. +# It passed preflight, transcription, candidate detection, the EDL, gate 2 +# and the preview render completely untouched. The creator caught it by eye +# after the cut was already locked. +# +# The step-2 QC pass named this exact defect class ("black edges, wrong +# aspect, letterboxed or cropped content") and still let it through, because +# it only EXTRACTED frames and left the looking to whoever happened to +# remember. QC that cannot halt is not QC, so this asserts and exits 3. +# +# It is deliberately a HALT-AND-ASK, not an auto-fix: a uniform edge can be +# legitimate (a dark set, a vignette, an intentional letterbox). The script +# reports the inferred active-content rectangle and stops; the creator +# confirms, and normalize_source.py does the correction. + + +def _px(pixels, w, i, x, y): + o = ((y * w) + x) * 3 + return pixels[o], pixels[o + 1], pixels[o + 2] + + +def ring_pixels(pixels, w, h, depth): + """The pixels exactly `depth` in from the frame edge (pure).""" + out = [] + if depth < 0 or depth * 2 >= min(w, h): + return out + for x in range(depth, w - depth): + out.append(_px(pixels, w, 0, x, depth)) + out.append(_px(pixels, w, 0, x, h - 1 - depth)) + for y in range(depth + 1, h - depth - 1): + out.append(_px(pixels, w, 0, depth, y)) + out.append(_px(pixels, w, 0, w - 1 - depth, y)) + return out + + +def mean_color(px): + """Per-channel mean of a pixel list (pure).""" + if not px: + return (0.0, 0.0, 0.0) + n = len(px) + return tuple(sum(p[c] for p in px) / n for c in range(3)) + + +def max_channel_variance(px): + """Largest per-channel variance across a pixel list (pure). + + A decorative border is a FLAT colour, so its variance is near zero, while + real picture content at the frame edge varies. This is the whole + discriminator.""" + if len(px) < 2: + return 0.0 + means = mean_color(px) + return max( + sum((p[c] - means[c]) ** 2 for p in px) / len(px) + for c in range(3) + ) + + +def color_distance(a, b): + """Euclidean distance between two mean colours (pure).""" + return sum((a[c] - b[c]) ** 2 for c in range(3)) ** 0.5 + + +def detect_border_depth(pixels, w, h, var_limit=DEFAULT_RING_VARIANCE, + max_frac=0.25): + """How many pixels of flat border ring the frame carries (pure). + + Walks inward from the edge while each successive ring is FLAT (low + variance). Stops at the first ring carrying real picture detail. A + multi-colour decorative frame (black outer border then an orange ring) is + still one contiguous run of flat rings, so both layers are counted, which + is what the real defect needed. + + Returns 0 when the outermost ring already carries picture content, which + is the normal, healthy case. + """ + limit = int(min(w, h) * max_frac) + depth = 0 + while depth < limit: + px = ring_pixels(pixels, w, h, depth) + if not px or max_channel_variance(px) > var_limit: + break + depth += 1 + return depth + + +def border_is_distinct(pixels, w, h, depth, min_distance=DEFAULT_RING_DISTANCE): + """True when the flat border differs from the picture inside it (pure). + + Guards the false positive that matters: a genuinely dark or flat SCENE + whose edges happen to be uniform. If the border and the interior are the + same colour there is no decorative frame, just a flat shot. + """ + if depth <= 0: + return False + border = mean_color(ring_pixels(pixels, w, h, 0)) + inner = ring_pixels(pixels, w, h, depth + 1) + if not inner: + return False + return color_distance(border, mean_color(inner)) >= min_distance + + +def active_rect(w, h, depth): + """The content rectangle inside a border of `depth` (pure).""" + return (depth, depth, w - 2 * depth, h - 2 * depth) + + +def scale_rect(rect, from_size, to_size): + """Map a rectangle measured on a thumbnail onto the full frame (pure). + + Values round to even numbers because encoders reject odd dimensions. + """ + fx, fy = from_size + tx, ty = to_size + sx, sy = tx / fx, ty / fy + x, y, rw, rh = rect + return (core.even(x * sx), core.even(y * sy), + core.even(rw * sx), core.even(rh * sy)) + + +def aspect_of(rect): + """Width over height of a rectangle (pure).""" + _, _, w, h = rect + return (w / h) if h else 0.0 + + +def sample_frame_rgb(media, t, w, h): + """One frame as raw rgb24 bytes at a small analysis size, or None. + + Downscaling before analysis is deliberate: it averages away sensor noise + and compression artefacts, so a flat border reads as genuinely flat, and + it keeps the whole check in stdlib with no imaging dependency. + """ + proc = subprocess.run( + ["ffmpeg", "-v", "error", "-ss", f"{t:.3f}", "-i", str(media), + "-frames:v", "1", "-vf", f"scale={w}:{h}", + "-f", "rawvideo", "-pix_fmt", "rgb24", "-"], + capture_output=True, + ) + if proc.returncode != 0 or len(proc.stdout) < w * h * 3: + return None + return proc.stdout[:w * h * 3] + + +def qc_source(media, duration, width, height, samples=DEFAULT_QC_SAMPLES, + aspect_tolerance=0.02): + """Inspect a source for edge defects. Returns a QC verdict dict. + + Reports the WORST (deepest) border found across the sampled frames, so a + frame effect that starts mid-take is caught even though the first frame + is clean. `ok` false means stop and ask the creator. + """ + verdict = { + "ok": True, + "samples": 0, + "border_depth_frac": 0.0, + "active_rect": None, + "active_aspect": None, + "declared_aspect": (width / height) if width and height else None, + "defects": [], + } + aw, ah = QC_ANALYSIS_SIZE + worst_depth, worst_pixels = 0, None + times = qc_sample_times(duration, samples) + for t in times: + pixels = sample_frame_rgb(media, t, aw, ah) + if pixels is None: + continue + verdict["samples"] += 1 + depth = detect_border_depth(pixels, aw, ah) + if depth > worst_depth and border_is_distinct(pixels, aw, ah, depth): + worst_depth, worst_pixels = depth, pixels + if not verdict["samples"]: + verdict["defects"].append("could not sample any frame for QC") + verdict["ok"] = False + return verdict + + if worst_depth > 0 and worst_pixels is not None: + thumb_rect = active_rect(aw, ah, worst_depth) + full_rect = scale_rect(thumb_rect, (aw, ah), (width, height)) + verdict["border_depth_frac"] = round(worst_depth / min(aw, ah), 4) + verdict["active_rect"] = list(full_rect) + verdict["active_aspect"] = round(aspect_of(full_rect), 4) + verdict["defects"].append( + f"flat border ring {verdict['border_depth_frac'] * 100:.1f}% deep " + f"on every edge; active content is " + f"{full_rect[2]}x{full_rect[3]} at +{full_rect[0]}+{full_rect[1]} " + f"of {width}x{height}") + declared = verdict["declared_aspect"] + if declared and abs(verdict["active_aspect"] - declared) > \ + aspect_tolerance: + verdict["defects"].append( + f"active area aspect {verdict['active_aspect']:.3f} does not " + f"match the container's {declared:.3f} (letterboxed, " + "pillarboxed, or a non-16:9 recording region)") + verdict["ok"] = False + return verdict + + def build_summary(files, disk): return { "files": files, @@ -194,6 +434,14 @@ def main(argv=None): help="dir for first/last frame QC stills") parser.add_argument("--disk-path", default=None, help="volume to disk-check (default: first file's dir)") + parser.add_argument("--qc-samples", type=int, default=DEFAULT_QC_SAMPLES, + help=f"frames sampled across each take for the " + f"edge-defect QC (default {DEFAULT_QC_SAMPLES})") + parser.add_argument("--no-qc", action="store_true", + help="skip the edge-defect QC pass entirely") + parser.add_argument("--allow-qc-defects", action="store_true", + help="report QC defects but do not halt (use only " + "when the framing is intentional)") args = parser.parse_args(argv) files = [] @@ -263,13 +511,48 @@ def main(argv=None): return 1 entry["cfr_master"] = str(dst) - if args.qc_frames: - for entry in files: + # Source QC: assert, then HALT. Extracting frames and hoping someone + # looks at them is what let a baked-in border reach a locked cut. + qc_failed = [] + for entry in files: + media = entry["cfr_master"] or entry["path"] + if args.qc_frames: entry["qc_frames"] = extract_qc_frames( - entry["cfr_master"] or entry["path"], Path(args.qc_frames), - entry["duration"]) - - print(json.dumps(build_summary(files, disk), indent=2)) + media, Path(args.qc_frames), entry["duration"]) + if args.no_qc: + continue + entry["qc"] = qc_source(media, entry["duration"], entry["width"], + entry["height"], samples=args.qc_samples) + if not entry["qc"]["ok"]: + qc_failed.append(entry) + + summary = build_summary(files, disk) + summary["qc_ok"] = not qc_failed + print(json.dumps(summary, indent=2)) + + if qc_failed and not args.allow_qc_defects: + print("\nSOURCE QC FAILED. Stopping before transcription or any " + "render.", file=sys.stderr) + for entry in qc_failed: + print(f"\n {entry['path']}", file=sys.stderr) + for d in entry["qc"]["defects"]: + print(f" - {d}", file=sys.stderr) + rect = entry["qc"].get("active_rect") + if rect: + print(f" inferred active content: crop=" + f"{rect[2]}:{rect[3]}:{rect[0]}:{rect[1]}", + file=sys.stderr) + print("\nA baked-in border or a non-matching active area cannot be " + "fixed later: the EDL is time-only, so every downstream stage " + "inherits the bad canvas.\nEither correct it now with " + "normalize_source.py (a spatial crop moves nothing in time, so " + "an existing transcript, EDL and cutplan stay valid), or " + "re-run with --allow-qc-defects if the framing is " + "intentional.", file=sys.stderr) + if args.qc_frames: + print(f"QC stills for eyeballing: {args.qc_frames}", + file=sys.stderr) + return 3 return 0 diff --git a/skills/mc-cut/scripts/remap_timecode.py b/skills/mc-cut/scripts/remap_timecode.py index 8e46e19..6034746 100644 --- a/skills/mc-cut/scripts/remap_timecode.py +++ b/skills/mc-cut/scripts/remap_timecode.py @@ -61,8 +61,8 @@ import sys from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent)) -from composite_core import format_timecode, parse_timecode +sys.path.insert(0, str(Path(__file__).resolve().parent)) # noqa: E402 +from composite_core import format_timecode, parse_timecode # noqa: E402 TC_RE = re.compile(r"(?/proxy/, and the preview is cut + from that instead of seeking into a 4K master once per segment. + Timecodes are identical, so the EDL is untouched. The proxy is + reused by every later preview and rebuilt only when the source + content changes. --no-proxy opts out; the final render never + uses proxies. + OVERLAY LANES: overlays are packed into the fewest + non-overlapping time lanes, each built as a cheap concat of + transparent gaps and overlay clips, and only the lanes are + stacked onto the base cut. Stack depth becomes max concurrent + overlays rather than total overlays (56 became 2 on the project + that exposed this). --lane-codec picks the intermediate codec. boundary-frames optional dir. After the render, one frame just before and one just after each internal cut boundary of the OUTPUT is extracted to /boundary--a.jpg (before) and boundary--b.jpg (after), n starting at 1, so the skill can inspect each cut. + safety the deliverable path is NEVER written directly. The encode goes + to a per-process temp file, which is then decode-validated + (`ffmpeg -v error -f null -`, zero errors) and duration-checked, + checked for supersede (has the EDL changed since this render + started?), and only then atomically moved into place with a + .key sidecar recording the render identity. See the + "render identity, atomic publish, output validation" block in + composite_core.py for why each of the three exists. summary json.dumps on stdout: segments, expected_duration (sum of raw segment durations), actual_duration (ffprobe of the output), - boundary_frames, overlays / overlays_missing (composited mode), - output path. + validated, render_key, boundary_frames, overlays / + overlays_missing (composited mode), output path. -Exit codes: 0 ok (and expected vs actual duration within 0.5s), 1 failure, -2 usage. +Exit codes: 0 ok (published and validated), 1 failure (encode failed, output +failed validation, or the render was superseded; in every case the existing +output is left untouched), 2 usage. STATUS: implemented (plain mode validated on real footage; composited mode covered by the scripts/tests suite). @@ -63,10 +88,11 @@ import json import subprocess import sys +import tempfile from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent)) -import composite_core as core +sys.path.insert(0, str(Path(__file__).resolve().parent)) # noqa: E402 +import composite_core as core # noqa: E402 # Re-exported so callers and tests can use the script as the single surface. segment_durations = core.segment_durations @@ -77,6 +103,75 @@ extract_boundary_frames = core.extract_boundary_frames +def ensure_proxies(sources, project_dir, proxy_dir, height): + """Build (or reuse) a preview proxy per source. Returns {source: rel_path}. + + A proxy is one linear transcode of the whole source to the preview + height. Cutting 234 segments out of a 720p proxy is dramatically cheaper + than seeking into a 4K master 234 times, and the proxy survives across + every later re-render. Timecodes are identical, so the EDL needs no + adjustment (see core.proxied_edl). + + Sources that fail to transcode are simply left unproxied: the render then + cuts them from the master, slower but correct. + """ + proxy_dir = Path(proxy_dir) + proxy_dir.mkdir(parents=True, exist_ok=True) + mapping = {} + for src in sources: + abs_src = project_dir / src + if not abs_src.is_file(): + continue + proxy = core.proxy_path(proxy_dir, src, height) + if not core.proxy_is_fresh(proxy, abs_src): + print(f"render_preview: building {height}p proxy for {src} " + "(once; reused by every later preview)", file=sys.stderr) + # Per-process temp then atomic replace. Proxies are shared state + # keyed on source and height, so two renders started close + # together would otherwise interleave writes into one path and + # hand each other a corrupt proxy. Same failure the deliverable + # path had; it deserves the same defence. + staged = core.temp_render_path(proxy, "proxy") + cmd = core.build_proxy_command(abs_src, staged, height) + proc = subprocess.run(cmd, capture_output=True, text=True) + if proc.returncode != 0: + core.discard_render(staged) + print(f"render_preview: proxy build failed for {src}; " + "falling back to the master for this source", + file=sys.stderr) + print(proc.stderr.strip()[-800:], file=sys.stderr) + continue + core.publish_render(staged, proxy) + core.write_proxy_sidecar(proxy, abs_src) + try: + mapping[src] = str(proxy.relative_to(project_dir)) + except ValueError: + mapping[src] = str(proxy) + return mapping + + +def render_lanes(overlays, total, size, fps, work_dir, codec): + """Render one file per overlay lane. Returns (lane_files, lane_count). + + Raises RuntimeError with the ffmpeg tail when a lane fails to build. + """ + lanes = core.plan_overlay_lanes(overlays) + files = [] + for i, lane in enumerate(lanes): + out = Path(work_dir) / f"lane{i + 1}.mov" + cmd = core.build_lane_command(lane, total, size, fps, out, codec) + if not cmd: + continue + print(f"render_preview: lane {i + 1}/{len(lanes)} " + f"({len(lane)} overlays)", file=sys.stderr) + proc = subprocess.run(cmd, capture_output=True, text=True) + if proc.returncode != 0: + raise RuntimeError(f"lane {i + 1} failed: " + f"{proc.stderr.strip()[-800:]}") + files.append(out) + return files, len(lanes) + + def gather_overlays(beats_path, graphics_dir): """Parse the beat table and resolve overlay files. @@ -101,6 +196,19 @@ def main(argv=None): help="beats/beats.md to composite graphics from") parser.add_argument("--graphics-dir", default=None, help="dir holding one rendered overlay per beat id") + parser.add_argument("--no-proxy", action="store_true", + help="cut from the masters instead of building " + "preview proxies (slower on 4K sources)") + parser.add_argument("--proxy-dir", default=None, + help="where preview proxies live " + "(default: /proxy)") + parser.add_argument("--lane-codec", default=core.DEFAULT_LANE_CODEC, + choices=sorted(core.LANE_CODEC_ARGS), + help="intermediate codec for overlay lanes " + f"(default {core.DEFAULT_LANE_CODEC})") + parser.add_argument("--fps", type=float, default=30.0, + help="preview frame rate for overlay lanes " + "(default 30)") args = parser.parse_args(argv) edl_path = Path(args.edl).resolve() @@ -136,12 +244,40 @@ def main(argv=None): for reason in skipped: print(f"beat row skipped: {reason}", file=sys.stderr) + output = Path(args.output) + output.parent.mkdir(parents=True, exist_ok=True) + + # Render identity, fixed before the encode starts, and computed against + # the ORIGINAL edl so a proxy swap never changes a render's identity. + # See composite_core's "render identity, atomic publish, output + # validation" block. + params = {"height": args.height, "mode": "preview", + "beats": bool(args.beats)} + overlay_digests = {ov["id"]: core.content_digest(ov["path"]) + for ov in overlays} + ffmpeg_v = core.ffmpeg_version() + key = core.render_key(edl, params, overlay_digests, ffmpeg_v) + tmp = core.temp_render_path(output, key) + + # Preview proxies: cut the preview from small linear transcodes instead + # of seeking into 4K masters once per segment. Timecodes are unchanged, + # so only the source paths differ. + proxy_map = {} + if not args.no_proxy: + proxy_dir = (Path(args.proxy_dir) if args.proxy_dir + else output.parent / "proxy") + proxy_map = ensure_proxies(distinct, project_dir, proxy_dir, + args.height) + render_edl = core.proxied_edl(edl, proxy_map) if proxy_map else edl + # One target frame is needed to scale overlays and to normalize mixed-size - # sources so the concat inputs match (cam + screencast). + # sources so the concat inputs match (cam + screencast). Probed from what + # is actually being cut. overlay_size = None target = None if overlays or multi: - dims = core.probe_dims(project_dir / edl["segments"][0]["source"]) + dims = core.probe_dims( + project_dir / render_edl["segments"][0]["source"]) if dims is None: print("cannot probe source dimensions", file=sys.stderr) return 1 @@ -153,24 +289,92 @@ def main(argv=None): # Audio-less sources (a screen recording with no audio) get synthesized # silence so the filtergraph never references a missing :a stream. - audio_map = {src: core.probe_has_audio(project_dir / src) - for src in distinct} + audio_map = {seg["source"]: core.probe_has_audio(project_dir / + seg["source"]) + for seg in render_edl["segments"]} - output = Path(args.output) - output.parent.mkdir(parents=True, exist_ok=True) - cmd, _ = build_command(edl, project_dir, output, args.height, - overlays=overlays, overlay_size=overlay_size, - target=target, audio_map=audio_map) - proc = subprocess.run(cmd, capture_output=True, text=True) - if proc.returncode != 0: - print("ffmpeg render failed:", file=sys.stderr) - print(" ".join(cmd), file=sys.stderr) - print(proc.stderr.strip()[-2000:], file=sys.stderr) + expected = sum(segment_durations(edl)) + lane_count = 0 + work = tempfile.TemporaryDirectory(prefix="mc-preview-", + dir=str(output.parent)) + try: + if not overlays: + # No graphics yet: one pass, exactly as before. + cmd, _ = build_command(render_edl, project_dir, tmp, args.height, + target=target, audio_map=audio_map) + proc = subprocess.run(cmd, capture_output=True, text=True) + if proc.returncode != 0: + core.discard_render(tmp) + print("ffmpeg render failed:", file=sys.stderr) + print(" ".join(cmd), file=sys.stderr) + print(proc.stderr.strip()[-2000:], file=sys.stderr) + return 1 + else: + # Overlay lanes: build the base cut, pack the overlays into the + # fewest non-overlapping lanes, then stack only the lanes. Depth + # is max-concurrent-overlays, not overlay count. + base = Path(work.name) / "base.mp4" + cmd, _ = build_command(render_edl, project_dir, base, args.height, + target=target, audio_map=audio_map) + proc = subprocess.run(cmd, capture_output=True, text=True) + if proc.returncode != 0: + core.discard_render(tmp) + print("ffmpeg base render failed:", file=sys.stderr) + print(" ".join(cmd), file=sys.stderr) + print(proc.stderr.strip()[-2000:], file=sys.stderr) + return 1 + base_dur = probe_duration(base) or expected + try: + lane_files, lane_count = render_lanes( + overlays, base_dur, overlay_size, args.fps, work.name, + args.lane_codec) + except RuntimeError as e: + core.discard_render(tmp) + print(f"overlay lane render failed: {e}", file=sys.stderr) + return 1 + print(f"render_preview: compositing {len(overlays)} overlays as " + f"{lane_count} lane(s)", file=sys.stderr) + cmd = core.build_lane_composite_command(base, lane_files, tmp) + proc = subprocess.run(cmd, capture_output=True, text=True) + if proc.returncode != 0: + core.discard_render(tmp) + print("ffmpeg composite failed:", file=sys.stderr) + print(" ".join(cmd), file=sys.stderr) + print(proc.stderr.strip()[-2000:], file=sys.stderr) + return 1 + finally: + work.cleanup() + + # Prove the file decodes BEFORE it is called a deliverable. A corrupt + # mp4 still reports a plausible container duration, so the duration check + # alone (all this used to do) cannot see it. + valid, problems = core.validate_render(tmp, expected) + if not valid: + core.discard_render(tmp) + print("render failed validation, output NOT published:", + file=sys.stderr) + for p in problems: + print(f" {p}", file=sys.stderr) return 1 - expected = sum(segment_durations(edl)) - actual = probe_duration(output) + # Supersede: if the EDL changed while this render ran, this render is + # stale. Discard it rather than clobber whatever newer render is now in + # flight or already published. + live_key = core.current_render_key(edl_path, params, overlay_digests, + ffmpeg_v) + if live_key is not None and live_key != key: + core.discard_render(tmp) + print(json.dumps({"superseded": True, "rendered_key": key, + "current_key": live_key, + "output": str(output.resolve())}, indent=2)) + print(f"SUPERSEDED: {edl_path.name} changed while this render ran; " + "discarded it instead of overwriting a newer render. Re-run " + "against the current EDL.", file=sys.stderr) + return 1 + + core.publish_render(tmp, output, key) + actual = probe_duration(output) boundary_count = 0 if args.boundary_frames: boundary_count = extract_boundary_frames( @@ -180,20 +384,17 @@ def main(argv=None): "segments": len(edl["segments"]), "expected_duration_seconds": round(expected, 3), "actual_duration_seconds": round(actual, 3) if actual is not None else None, + "validated": True, + "render_key": key, + "proxied_sources": len(proxy_map), "boundary_frames": boundary_count, "output": str(output.resolve()), } if args.beats: summary["overlays"] = len(overlays) + summary["overlay_lanes"] = lane_count summary["overlays_missing"] = missing print(json.dumps(summary, indent=2)) - - if actual is None or abs(actual - expected) > 0.5: - print( - f"duration mismatch: expected {expected:.3f}s, got {actual}s", - file=sys.stderr, - ) - return 1 return 0 diff --git a/skills/mc-cut/scripts/snap_spans.py b/skills/mc-cut/scripts/snap_spans.py new file mode 100644 index 0000000..141f8c9 --- /dev/null +++ b/skills/mc-cut/scripts/snap_spans.py @@ -0,0 +1,189 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.11" +# /// +"""Snap arbitrary cut spans into audio-verified silences. + +Usage: + uv run {skill-root}/scripts/snap_spans.py {projects-path}//cut/approved-spans.json \ + --audio-map {projects-path}//cut/audio-map.json \ + [--snap-ms 250] [-o {projects-path}//cut/snapped-spans.json] + +Why this exists: + Step 7a applies the creator's content-tier calls after gate 2, and it used + to say "re-detect every content-tier span against the audio, snap the + edges into silences from the audio map". Snapping a time into the nearest + silence is a pure function that cutplan.py already owns + (snap_candidates), but cutplan's CLI only turns a transcript into + candidates, so there was no way to snap a span the creator approved. The + step was asking a model to redo by hand a computation the skill already + had in Python, at the exact point where getting it wrong cuts the wrong + words. An eyeballed snap is how a cut lands inside a word. + + So this is the same mechanic with a CLI in front of it. Judgment stays + with the creator (which spans to cut); the arithmetic comes here. + +Snapping is DIRECTIONAL, and that is load-bearing: + A span here is material to REMOVE, so its start may only move EARLIER and + its end only LATER. The span may widen into surrounding silence, which is + always safe, and can never invert or collapse onto itself. Snapping both + edges to the unconstrained nearest silence can pull an end backwards past + its own start and annihilate the span. + +Contract: + input a JSON list of spans, or an object with a "spans" key. Each span + is {"start", "end", ...}; every other key is carried through + untouched, so ids, quotes and reasons survive the round trip. + output the same spans with snapped start/end, plus "snapped" (bool) and + "shift" (seconds each edge moved) on each. -o writes the full + payload; stdout carries the summary. + summary json.dumps: {"ok", "spans", "snapped", "unsnapped": [...]} + +`ok` is true only when every span reached a silence on BOTH edges. Spans that +could not are left at their original times, marked "snapped": false, and +named in `unsnapped`: they still carry timestamp risk and need an ear before +they go into the EDL. This script never pretends an edge is safe. + +Exit codes: 0 every span snapped, 1 one or more spans unsnapped (check them +by ear), 2 usage error. + +STATUS: implemented (pure logic covered by scripts/tests/test-snap_spans.py). +""" + +import argparse +import json +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent)) # noqa: E402 +import analyze_audio as audio # noqa: E402 + +# Matches cutplan.py's DEFAULT_SNAP_MS: the two paths must agree about how far +# an edge may travel, or a span snapped here would not survive verify_edl.py. +DEFAULT_SNAP_MS = 250 + + +def r3(x): + return round(float(x), 3) + + +def snap_span(span, silence, max_shift): + """Snap one span's edges into silence, directionally (pure).""" + out = dict(span) + start = float(span["start"]) + end = float(span["end"]) + new_start = audio.nearest_silence(silence, start, max_shift, + direction="back") + new_end = audio.nearest_silence(silence, end, max_shift, + direction="forward") + out["snapped"] = new_start is not None and new_end is not None + resolved_start = start if new_start is None else new_start + resolved_end = end if new_end is None else new_end + if resolved_end < resolved_start: + resolved_end = resolved_start + out["start"] = r3(resolved_start) + out["end"] = r3(resolved_end) + out["dur"] = r3(resolved_end - resolved_start) + out["shift"] = {"start": r3(resolved_start - start), + "end": r3(resolved_end - end)} + if not out["snapped"]: + out["unsnapped_edges"] = [ + edge for edge, value in (("start", new_start), ("end", new_end)) + if value is None] + return out + + +def snap_all(spans, silence, max_shift): + """Snap every span (pure).""" + return [snap_span(s, silence, max_shift) for s in spans] + + +def build_report(snapped): + """Assemble the verdict (pure).""" + unsnapped = [s for s in snapped if not s["snapped"]] + return { + "ok": not unsnapped, + "spans": len(snapped), + "snapped": len(snapped) - len(unsnapped), + "unsnapped": [{k: s[k] for k in ("start", "end", "unsnapped_edges") + if k in s} for s in unsnapped], + } + + +def parse_spans(payload): + """Accept a bare list or an object with a spans key (pure).""" + if isinstance(payload, list): + return payload + if isinstance(payload, dict): + for key in ("spans", "candidates", "cuts"): + if isinstance(payload.get(key), list): + return payload[key] + return None + + +def main(argv=None): + p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + p.add_argument("spans", help="path to the approved spans JSON") + p.add_argument("--audio-map", required=True, + help="path to cut/audio-map.json from analyze_audio.py") + p.add_argument("-o", "--output", default=None, + help="optional path for the snapped spans JSON") + p.add_argument("--snap-ms", type=int, default=DEFAULT_SNAP_MS, + help=f"how far an edge may move to reach a silence " + f"(default {DEFAULT_SNAP_MS})") + args = p.parse_args(argv) + + try: + with open(args.spans, encoding="utf-8") as f: + payload = json.load(f) + except (OSError, json.JSONDecodeError) as e: + print(f"snap_spans: cannot read spans {args.spans}: {e}", + file=sys.stderr) + return 2 + spans = parse_spans(payload) + if spans is None: + print("snap_spans: expected a JSON list of spans, or an object with a " + "'spans' key", file=sys.stderr) + return 2 + for i, s in enumerate(spans): + if not isinstance(s, dict) or "start" not in s or "end" not in s: + print(f"snap_spans: span {i} needs both 'start' and 'end'", + file=sys.stderr) + return 2 + + try: + with open(args.audio_map, encoding="utf-8") as f: + audio_map = json.load(f) + except (OSError, json.JSONDecodeError) as e: + print(f"snap_spans: cannot read audio map {args.audio_map}: {e}", + file=sys.stderr) + return 2 + if "silence" not in audio_map: + print("snap_spans: audio map has no 'silence' key; regenerate it with " + "analyze_audio.py", file=sys.stderr) + return 2 + + silence = audio.to_pairs(audio_map["silence"]) + snapped = snap_all(spans, silence, args.snap_ms / 1000.0) + report = build_report(snapped) + + if args.output: + out = Path(args.output) + out.parent.mkdir(parents=True, exist_ok=True) + out.write_text(json.dumps(snapped, indent=2, ensure_ascii=False) + "\n", + encoding="utf-8") + else: + print(json.dumps(snapped, indent=2, ensure_ascii=False)) + + print(json.dumps(report, indent=2)) + if report["ok"]: + return 0 + print(f"\n{len(report['unsnapped'])} span(s) could not reach a silence " + "within the snap budget. They keep their original times and still " + "carry timestamp risk: check them by ear before they go into the " + "EDL.", file=sys.stderr) + return 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/mc-cut/scripts/tests/test-analyze_audio.py b/skills/mc-cut/scripts/tests/test-analyze_audio.py new file mode 100644 index 0000000..f53b944 --- /dev/null +++ b/skills/mc-cut/scripts/tests/test-analyze_audio.py @@ -0,0 +1,284 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.11" +# /// +"""Tests for analyze_audio.py: the silencedetect parser and the interval +algebra every cut-timing decision now rests on. + +The ffmpeg invocation itself is not unit-tested (it needs a media file); the +parsing of its output and all the interval math are pure and covered here.""" +import importlib.util +import json +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +SCRIPT = Path(__file__).resolve().parent.parent / "analyze_audio.py" +spec = importlib.util.spec_from_file_location("analyze_audio", SCRIPT) +aa = importlib.util.module_from_spec(spec) +spec.loader.exec_module(aa) + + +def line(kind, t, dur=None): + if kind == "start": + return f"[silencedetect @ 0x7f8] silence_start: {t}" + return (f"[silencedetect @ 0x7f8] silence_end: {t} | " + f"silence_duration: {dur}") + + +class TestParseSilencedetect(unittest.TestCase): + def test_paired_lines_become_intervals(self): + text = "\n".join([line("start", 1.5), line("end", 3.0, 1.5), + line("start", 10.0), line("end", 12.25, 2.25)]) + self.assertEqual(aa.parse_silencedetect(text, 20.0), + [(1.5, 3.0), (10.0, 12.25)]) + + def test_unclosed_final_silence_is_closed_at_duration(self): + # A file that ends in silence gets a start with no matching end. + text = "\n".join([line("start", 1.0), line("end", 2.0, 1.0), + line("start", 18.5)]) + self.assertEqual(aa.parse_silencedetect(text, 20.0), + [(1.0, 2.0), (18.5, 20.0)]) + + def test_noise_lines_are_ignored(self): + text = "\n".join(["ffmpeg version 7.1", " Stream #0:0 Video: h264", + line("start", 1.0), line("end", 2.0, 1.0), + "frame= 100 fps=50"]) + self.assertEqual(aa.parse_silencedetect(text, 5.0), [(1.0, 2.0)]) + + def test_empty_output_is_no_silence(self): + self.assertEqual(aa.parse_silencedetect("", 10.0), []) + + def test_intervals_are_clamped_into_the_media(self): + text = "\n".join([line("start", -0.5), line("end", 2.0, 2.5), + line("start", 9.0), line("end", 25.0, 16.0)]) + self.assertEqual(aa.parse_silencedetect(text, 10.0), + [(0.0, 2.0), (9.0, 10.0)]) + + def test_zero_length_intervals_are_dropped(self): + text = "\n".join([line("start", 3.0), line("end", 3.0, 0.0)]) + self.assertEqual(aa.parse_silencedetect(text, 10.0), []) + + def test_overlapping_intervals_merge(self): + text = "\n".join([line("start", 1.0), line("end", 3.0, 2.0), + line("start", 2.5), line("end", 4.0, 1.5)]) + self.assertEqual(aa.parse_silencedetect(text, 10.0), [(1.0, 4.0)]) + + def test_result_is_sorted(self): + text = "\n".join([line("start", 8.0), line("end", 9.0, 1.0), + line("start", 1.0), line("end", 2.0, 1.0)]) + self.assertEqual(aa.parse_silencedetect(text, 10.0), + [(1.0, 2.0), (8.0, 9.0)]) + + +class TestComplement(unittest.TestCase): + def test_gaps_between_intervals(self): + self.assertEqual(aa.complement([(1.0, 2.0), (5.0, 6.0)], 10.0), + [(0.0, 1.0), (2.0, 5.0), (6.0, 10.0)]) + + def test_no_intervals_is_the_whole_span(self): + self.assertEqual(aa.complement([], 10.0), [(0.0, 10.0)]) + + def test_full_coverage_is_empty(self): + self.assertEqual(aa.complement([(0.0, 10.0)], 10.0), []) + + def test_leading_and_trailing_coverage_leaves_only_the_middle(self): + self.assertEqual(aa.complement([(0.0, 1.0), (9.0, 10.0)], 10.0), + [(1.0, 9.0)]) + + def test_complement_is_an_involution_on_clean_input(self): + silence = [(1.0, 2.0), (5.0, 6.0)] + speech = aa.complement(silence, 10.0) + self.assertEqual(aa.complement(speech, 10.0), silence) + + def test_silence_and_speech_partition_the_duration(self): + silence = [(1.0, 2.0), (5.0, 6.5)] + speech = aa.complement(silence, 10.0) + self.assertAlmostEqual(aa.total(silence) + aa.total(speech), 10.0) + + +class TestOverlapSeconds(unittest.TestCase): + def test_fully_inside(self): + self.assertAlmostEqual( + aa.overlap_seconds([(0.0, 10.0)], 2.0, 5.0), 3.0) + + def test_partial_at_each_end(self): + self.assertAlmostEqual( + aa.overlap_seconds([(1.0, 3.0), (8.0, 12.0)], 2.0, 9.0), 2.0) + + def test_disjoint_is_zero(self): + self.assertAlmostEqual( + aa.overlap_seconds([(0.0, 1.0)], 5.0, 6.0), 0.0) + + def test_empty_span_is_zero(self): + self.assertAlmostEqual( + aa.overlap_seconds([(0.0, 10.0)], 5.0, 5.0), 0.0) + + def test_inverted_span_is_zero(self): + self.assertAlmostEqual( + aa.overlap_seconds([(0.0, 10.0)], 6.0, 5.0), 0.0) + + +class TestSilentFraction(unittest.TestCase): + def test_all_silent(self): + self.assertAlmostEqual( + aa.silent_fraction([(0.0, 10.0)], 1.0, 5.0), 1.0) + + def test_none_silent(self): + self.assertAlmostEqual( + aa.silent_fraction([(20.0, 30.0)], 1.0, 5.0), 0.0) + + def test_half_silent(self): + self.assertAlmostEqual( + aa.silent_fraction([(0.0, 3.0)], 1.0, 5.0), 0.5) + + def test_zero_span_reads_as_silent(self): + self.assertAlmostEqual(aa.silent_fraction([], 5.0, 5.0), 1.0) + + +class TestEnclosing(unittest.TestCase): + def test_inside(self): + self.assertEqual(aa.enclosing([(1.0, 3.0)], 2.0), (1.0, 3.0)) + + def test_boundaries_are_inclusive(self): + self.assertEqual(aa.enclosing([(1.0, 3.0)], 1.0), (1.0, 3.0)) + self.assertEqual(aa.enclosing([(1.0, 3.0)], 3.0), (1.0, 3.0)) + + def test_outside_is_none(self): + self.assertIsNone(aa.enclosing([(1.0, 3.0)], 5.0)) + + def test_between_intervals_is_none(self): + self.assertIsNone(aa.enclosing([(1.0, 2.0), (5.0, 6.0)], 3.0)) + + +class TestNearestSilence(unittest.TestCase): + SIL = [(0.0, 1.0), (5.0, 6.0), (9.0, 10.0)] + + def test_point_already_in_silence_is_unchanged(self): + self.assertEqual(aa.nearest_silence(self.SIL, 5.5, 0.5), 5.5) + + def test_snaps_to_the_nearest_edge(self): + self.assertAlmostEqual(aa.nearest_silence(self.SIL, 4.8, 0.5), 5.0) + self.assertAlmostEqual(aa.nearest_silence(self.SIL, 6.2, 0.5), 6.0) + + def test_out_of_budget_returns_none(self): + self.assertIsNone(aa.nearest_silence(self.SIL, 3.0, 0.5)) + + def test_budget_boundary_is_inclusive(self): + self.assertAlmostEqual(aa.nearest_silence(self.SIL, 4.5, 0.5), 5.0) + + def test_direction_back_never_moves_forward(self): + # 6.2 is nearer to 6.0 (back) than to 9.0 (forward). + self.assertAlmostEqual( + aa.nearest_silence(self.SIL, 6.2, 0.5, direction="back"), 6.0) + self.assertIsNone( + aa.nearest_silence(self.SIL, 6.2, 0.5, direction="forward")) + + def test_direction_forward_never_moves_back(self): + self.assertAlmostEqual( + aa.nearest_silence(self.SIL, 4.8, 0.5, direction="forward"), 5.0) + self.assertIsNone( + aa.nearest_silence(self.SIL, 4.8, 0.5, direction="back")) + + def test_directional_snapping_cannot_invert_a_span(self): + # The regression behind the collapsed-candidate bug: an end at 0.9 + # with silence both behind (0.0 to 1.0... ) and ahead. Unconstrained + # "nearest" could pull an end backwards past its own start; forward + # snapping cannot. + sil = [(0.0, 0.5), (1.5, 2.0)] + start = aa.nearest_silence(sil, 0.5, 0.8, direction="back") + end = aa.nearest_silence(sil, 0.9, 0.8, direction="forward") + self.assertEqual(start, 0.5) + self.assertAlmostEqual(end, 1.5) + self.assertGreater(end, start) + + +class TestRecordRoundTrip(unittest.TestCase): + def test_as_records_then_to_pairs_is_identity(self): + pairs = [(1.0, 2.5), (5.25, 6.0)] + self.assertEqual(aa.to_pairs(aa.as_records(pairs)), pairs) + + def test_records_carry_dur(self): + rec = aa.as_records([(1.0, 2.5)])[0] + self.assertEqual(rec["dur"], 1.5) + + def test_total_sums_lengths(self): + self.assertAlmostEqual(aa.total([(0.0, 1.0), (5.0, 7.5)]), 3.5) + + +class TestBuild(unittest.TestCase): + def test_payload_shape_and_partition(self): + payload = aa.build("m.mp4", 10.0, [(1.0, 2.0), (5.0, 6.0)], -30.0, 0.3) + self.assertEqual(payload["counts"]["silence"], 2) + self.assertEqual(payload["silent_seconds"], 2.0) + self.assertEqual(payload["speech_seconds"], 8.0) + self.assertAlmostEqual( + payload["silent_seconds"] + payload["speech_seconds"], 10.0) + self.assertEqual(payload["noise_db"], -30.0) + self.assertEqual(payload["map_granularity"], 0.3) + + def test_no_silence_means_all_speech(self): + payload = aa.build("m.mp4", 10.0, [], -30.0, 0.3) + self.assertEqual(payload["silent_seconds"], 0.0) + self.assertEqual(payload["speech_seconds"], 10.0) + self.assertEqual(payload["counts"]["speech"], 1) + + +class TestRealTakeShape(unittest.TestCase): + """The measured shape of the take that broke the pipeline: about 402 + silence intervals totalling about 309s in a 1230s take. The gap-based + detector it replaced found 12.""" + + def _synthetic_take(self): + # 402 silences averaging 0.77s spread through a 20.5 minute take. + text = [] + t = 0.5 + for _ in range(402): + end = t + 0.769 + text.append(line("start", round(t, 3))) + text.append(line("end", round(end, 3), 0.769)) + t = end + 2.29 + return "\n".join(text) + + def test_parses_hundreds_of_intervals(self): + pairs = aa.parse_silencedetect(self._synthetic_take(), 1230.0) + self.assertEqual(len(pairs), 402) + + def test_totals_are_in_the_measured_range(self): + pairs = aa.parse_silencedetect(self._synthetic_take(), 1230.0) + payload = aa.build("m.mp4", 1230.0, pairs, -30.0, 0.3) + self.assertGreater(payload["silent_seconds"], 240.0) + self.assertGreater(payload["counts"]["silence"], 300) + + +class TestCli(unittest.TestCase): + def test_missing_media_is_usage_error(self): + with tempfile.TemporaryDirectory() as tmp: + r = subprocess.run( + [sys.executable, str(SCRIPT), str(Path(tmp) / "nope.mp4"), + "-o", str(Path(tmp) / "map.json")], + capture_output=True, text=True) + self.assertEqual(r.returncode, 2) + self.assertIn("media not found", r.stderr) + + def test_non_positive_map_granularity_is_usage_error(self): + with tempfile.TemporaryDirectory() as tmp: + media = Path(tmp) / "m.mp4" + media.write_bytes(b"not really media") + r = subprocess.run( + [sys.executable, str(SCRIPT), str(media), "-o", + str(Path(tmp) / "map.json"), "--map-granularity", "0"], + capture_output=True, text=True) + self.assertEqual(r.returncode, 2) + + def test_help_exits_zero(self): + r = subprocess.run([sys.executable, str(SCRIPT), "--help"], + capture_output=True, text=True) + self.assertEqual(r.returncode, 0) + self.assertIn("--noise", r.stdout) + + +if __name__ == "__main__": + unittest.main() diff --git a/skills/mc-cut/scripts/tests/test-cutplan.py b/skills/mc-cut/scripts/tests/test-cutplan.py index 6b2261b..808443b 100644 --- a/skills/mc-cut/scripts/tests/test-cutplan.py +++ b/skills/mc-cut/scripts/tests/test-cutplan.py @@ -5,6 +5,7 @@ """Tests for cutplan.py: the cut-candidate finder must catch every planted defect class, gate soft fillers to sentence starts, and emit the pinned schema shape (cls on fillers only, candidates sorted by start).""" +import importlib.util import json import subprocess import sys @@ -13,6 +14,10 @@ from pathlib import Path SCRIPT = Path(__file__).resolve().parent.parent / "cutplan.py" +_spec = importlib.util.spec_from_file_location("cutplan", SCRIPT) +cutplan = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(cutplan) +cutplan_parse = cutplan.parse_voice_cadence def word(w, start, end, i, gap_before=0.0, gap_after=0.0, conf=1.0): @@ -38,17 +43,55 @@ def seq(pairs, wgap=0.0): return words -def run(words, duration=None, extra=None, expect=0): +def silence_from_words(words, duration): + """The audio map a perfect detector would produce for a synthetic take. + + Silence is everything not covered by a word span. In these fixtures the + words are laid out deliberately, so their gaps ARE the real silence and + this stands in for analyze_audio.py without needing ffmpeg or a media + file. Real transcripts must never derive silence this way, which is the + whole point of the audio map (see cutplan.py's two-source rule). + """ + intervals = [] + cursor = 0.0 + for w in words: + if w["start"] - cursor > 1e-6: + intervals.append({"start": round(cursor, 3), + "end": round(w["start"], 3), + "dur": round(w["start"] - cursor, 3)}) + cursor = max(cursor, w["end"]) + if duration - cursor > 1e-6: + intervals.append({"start": round(cursor, 3), + "end": round(duration, 3), + "dur": round(duration - cursor, 3)}) + return intervals + + +def run(words, duration=None, extra=None, expect=0, silence=None, + with_audio_map=True): if duration is None: duration = words[-1]["end"] if words else 0.0 data = {"media": "m.mp4", "duration": duration, "text": "", "words": words} + if silence is None: + silence = silence_from_words(words, duration) with tempfile.TemporaryDirectory() as tmp: src = Path(tmp) / "words.json" out = Path(tmp) / "candidates.json" + amap = Path(tmp) / "audio-map.json" src.write_text(json.dumps(data)) - r = subprocess.run([sys.executable, str(SCRIPT), str(src), "-o", - str(out), *(extra or [])], - capture_output=True, text=True) + amap.write_text(json.dumps({ + "media": "m.mp4", "duration": duration, + "noise_db": -30.0, "min_silence": 0.3, + "silent_seconds": sum(s["dur"] for s in silence), + "speech_seconds": duration - sum(s["dur"] for s in silence), + "counts": {"silence": len(silence), "speech": 0}, + "silence": silence, "speech": [], + })) + argv = [sys.executable, str(SCRIPT), str(src), "-o", str(out)] + if with_audio_map: + argv += ["--audio-map", str(amap)] + argv += list(extra or []) + r = subprocess.run(argv, capture_output=True, text=True) assert r.returncode == expect, f"rc={r.returncode} stderr={r.stderr}" result = json.loads(out.read_text()) if r.returncode == 0 else None return result, r @@ -59,26 +102,93 @@ def by_type(result, t): class TestSilence(unittest.TestCase): + """A silence candidate's span is THE PART TO REMOVE, not the whole + silence: interior gaps are tightened down to keep_ms of breathing room, + head and tail are trimmed to keep_ms off the first and last word.""" + def test_leading_mid_trailing(self): words = seq([("Hello,", 0.4, 1.0), ("world.", 0.4, 1.5)]) result, _ = run(words, duration=round(words[-1]["end"] + 0.9, 2)) sil = by_type(result, "silence") - starts = [c["start"] for c in sil] self.assertEqual(len(sil), 3) # leading, mid, trailing - self.assertEqual(starts[0], 0.0) # leading from 0 - self.assertEqual(sil[-1]["end"], result["duration"]) # trailing to dur + self.assertEqual(sil[0]["start"], 0.0) # head trims from 0 + self.assertEqual(sil[-1]["end"], result["duration"]) # tail to dur + + def test_head_trim_leaves_keep_ms_before_the_first_word(self): + words = seq([("Hello,", 0.4, 1.0)]) + result, _ = run(words, duration=2.0) + head = by_type(result, "silence")[0] + self.assertEqual(head["start"], 0.0) + self.assertAlmostEqual(head["end"], 0.8, places=2) # 1.0 - 0.2 + + def test_tail_trim_leaves_keep_ms_after_the_last_word(self): + words = seq([("Hello.", 0.4, 0.1)]) + result, _ = run(words, duration=3.0) + tail = by_type(result, "silence")[-1] + self.assertAlmostEqual(tail["start"], 0.7, places=2) # 0.5 + 0.2 + self.assertEqual(tail["end"], 3.0) + + def test_interior_gap_is_tightened_not_removed(self): + # A 2.0s interior silence keeps 200ms, split evenly, so 1.8s goes. + words = seq([("a", 0.3), ("b", 0.3, 2.0)]) + result, _ = run(words, duration=3.0) + mid = [c for c in by_type(result, "silence") + if c["silence_start"] == 0.3][0] + self.assertAlmostEqual(mid["dur"], 1.8, places=2) + self.assertAlmostEqual(mid["start"], 0.4, places=2) + self.assertAlmostEqual(mid["end"], 2.2, places=2) + self.assertEqual(mid["keep_ms"], 200) + + def test_tightened_edges_sit_inside_the_silence(self): + words = seq([("a", 0.3), ("b", 0.3, 2.0)]) + result, _ = run(words, duration=3.0) + mid = [c for c in by_type(result, "silence") + if c["silence_start"] == 0.3][0] + self.assertGreater(mid["start"], mid["silence_start"]) + self.assertLess(mid["end"], mid["silence_end"]) + + def test_keep_ms_is_configurable(self): + words = seq([("a", 0.3), ("b", 0.3, 2.0)]) + tight, _ = run(words, duration=3.0, extra=["--keep-ms", "0"]) + loose, _ = run(words, duration=3.0, extra=["--keep-ms", "600"]) + t = [c for c in by_type(tight, "silence") + if c["silence_start"] == 0.3][0] + l = [c for c in by_type(loose, "silence") + if c["silence_start"] == 0.3][0] + self.assertAlmostEqual(t["dur"], 2.0, places=2) + self.assertAlmostEqual(l["dur"], 1.4, places=2) def test_severity_threshold(self): words = seq([("a", 0.3), ("b", 0.3, 1.0), ("c", 0.3, 2.5)]) - result, _ = run(words) - sev = {c["dur"]: c["severity"] for c in by_type(result, "silence")} + result, _ = run(words, duration=4.5) + sev = {round(c["silence_end"] - c["silence_start"], 2): c["severity"] + for c in by_type(result, "silence")} self.assertEqual(sev[1.0], "med") self.assertEqual(sev[2.5], "high") - def test_below_threshold_ignored(self): - words = seq([("a", 0.3), ("b", 0.3, 0.5)]) + def test_micro_beats_below_threshold_are_left_alone(self): + # 0.2s of silence is the speaker's rhythm, not dead air (the median + # silence on the real reference take was 0.19s). Cutting these is + # what makes an edit sound machine-gunned. + words = seq([("a", 0.3), ("b", 0.3, 0.2)]) result, _ = run(words, duration=round(words[-1]["end"] + 0.2, 2)) - self.assertEqual(by_type(result, "silence"), []) + interior = [c for c in by_type(result, "silence") + if c["silence_start"] > 0.01] + self.assertEqual(interior, []) + + def test_min_silence_default_is_the_tightening_floor(self): + words = seq([("a", 0.3), ("b", 0.3, 0.5)]) + result, _ = run(words, duration=round(words[-1]["end"] + 0.1, 2)) + interior = [c for c in by_type(result, "silence") + if 0.01 < c["silence_start"]] + self.assertEqual(len(interior), 1) # 0.5s is above the 0.45 floor + + def test_trimmable_seconds_are_totalled(self): + words = seq([("a", 0.3), ("b", 0.3, 2.0)]) + result, _ = run(words, duration=3.0) + self.assertAlmostEqual( + result["trimmable_silence_seconds"], + sum(c["dur"] for c in by_type(result, "silence")), places=2) class TestFiller(unittest.TestCase): @@ -100,25 +210,138 @@ def test_consecutive_hard_merge(self): self.assertIn("run", hard[0]["reason"]) def test_soft_at_sentence_start(self): - # "So" opens the clip -> flagged; the mid-sentence "right" is not - words = seq([("So", 0.3), ("that", 0.3), ("is", 0.3), ("right", 0.3)]) + # "Basically" opens the clip -> flagged; mid-sentence use is not. + words = seq([("Basically", 0.3), ("that", 0.3), ("is", 0.3), + ("it.", 0.3)]) result, _ = run(words) soft = [c for c in by_type(result, "filler") if c["cls"] == "soft"] - self.assertEqual([c["text"] for c in soft], ["So"]) + self.assertEqual([c["text"] for c in soft], ["Basically"]) def test_soft_gated_after_period(self): - # "so" mid-sentence (no period, no gap) is NOT flagged - words = seq([("I", 0.3), ("think", 0.3), ("so", 0.3)]) + # mid-sentence (no period, no gap) is NOT flagged + words = seq([("I", 0.3), ("think", 0.3), ("basically", 0.3)]) result, _ = run(words) self.assertEqual([c for c in by_type(result, "filler") if c["cls"] == "soft"], []) def test_soft_gated_by_gap(self): - # "so" after a >=0.5 gap counts as a sentence start - words = seq([("wait", 0.3), ("so", 0.3, 0.6), ("yes", 0.3)]) + # a soft filler after a >=0.5 gap counts as a sentence start + words = seq([("wait", 0.3), ("basically", 0.3, 0.6), ("yes", 0.3)]) result, _ = run(words) soft = [c for c in by_type(result, "filler") if c["cls"] == "soft"] - self.assertEqual([c["text"] for c in soft], ["so"]) + self.assertEqual([c["text"] for c in soft], ["basically"]) + + +class TestCadenceWordsAreNotFillers(unittest.TestCase): + """The regression that made a creator's edit sound dead. + + The mechanical pass flagged all 19 sentence-initial "So"s on a real take. + The voice bible names "so" as that speaker's natural connective glue. + Cutting them is technically clean and tonally wrong, so the default list + no longer contains cadence words at all.""" + + def test_sentence_initial_so_is_not_flagged(self): + words = seq([("So", 0.3), ("here", 0.3), ("we", 0.3), ("are.", 0.3)]) + result, _ = run(words) + self.assertEqual([c for c in by_type(result, "filler") + if c["cls"] == "soft"], []) + + def test_other_cadence_words_are_not_flagged(self): + for w in ("Right", "Okay", "Well", "Anyway", "Now", "Look"): + words = seq([(w, 0.3), ("here", 0.3), ("we", 0.3), ("go.", 0.3)]) + result, _ = run(words) + soft = [c for c in by_type(result, "filler") + if c["cls"] == "soft"] + self.assertEqual(soft, [], f"{w} should not be a soft filler") + + def test_hard_fillers_are_still_cut(self): + words = seq([("So", 0.3), ("uh", 0.3), ("here.", 0.3)]) + result, _ = run(words) + hard = [c for c in by_type(result, "filler") if c["cls"] == "hard"] + self.assertEqual([c["text"] for c in hard], ["uh"]) + + def test_soft_candidates_say_the_default_is_to_keep(self): + words = seq([("Basically", 0.3), ("yes.", 0.3)]) + result, _ = run(words) + soft = [c for c in by_type(result, "filler") if c["cls"] == "soft"][0] + self.assertIn("KEEP", soft["reason"]) + + +class TestVoiceBible(unittest.TestCase): + """Per-creator cadence lives in the creator's file, not in this script.""" + + def _bible(self, tmp, body): + p = Path(tmp) / "voice-bible.md" + p.write_text(body) + return p + + def test_cut_list_adds_a_soft_filler(self): + with tempfile.TemporaryDirectory() as tmp: + bible = self._bible(tmp, "# Voice\n\n```cadence\ncut: honestly\n```\n") + words = seq([("Honestly", 0.3), ("yes.", 0.3)]) + result, _ = run(words, extra=["--voice-bible", str(bible)]) + soft = [c for c in by_type(result, "filler") + if c["cls"] == "soft"] + self.assertEqual([c["text"] for c in soft], ["Honestly"]) + self.assertTrue(result["voice_bible_applied"]) + + def test_keep_list_protects_a_default_soft_filler(self): + with tempfile.TemporaryDirectory() as tmp: + bible = self._bible(tmp, "```cadence\nkeep: basically\n```\n") + words = seq([("Basically", 0.3), ("yes.", 0.3)]) + result, _ = run(words, extra=["--voice-bible", str(bible)]) + self.assertEqual([c for c in by_type(result, "filler") + if c["cls"] == "soft"], []) + + def test_keep_list_protects_even_a_hard_filler(self): + with tempfile.TemporaryDirectory() as tmp: + bible = self._bible(tmp, "```cadence\nkeep: hmm\n```\n") + words = seq([("Hmm", 0.3), ("yes.", 0.3)]) + result, _ = run(words, extra=["--voice-bible", str(bible)]) + self.assertEqual(by_type(result, "filler"), []) + + def test_keep_wins_over_cut_on_conflict(self): + with tempfile.TemporaryDirectory() as tmp: + bible = self._bible( + tmp, "```cadence\nkeep: honestly\ncut: honestly\n```\n") + words = seq([("Honestly", 0.3), ("yes.", 0.3)]) + result, _ = run(words, extra=["--voice-bible", str(bible)]) + self.assertEqual([c for c in by_type(result, "filler") + if c["cls"] == "soft"], []) + + def test_multi_word_cut_phrase(self): + with tempfile.TemporaryDirectory() as tmp: + bible = self._bible(tmp, "```cadence\ncut: kind of\n```\n") + words = seq([("Kind", 0.3), ("of", 0.3), ("yes.", 0.3)]) + result, _ = run(words, extra=["--voice-bible", str(bible)]) + soft = [c for c in by_type(result, "filler") + if c["cls"] == "soft"] + self.assertEqual([c["text"] for c in soft], ["Kind of"]) + + def test_bible_without_a_cadence_block_warns_and_uses_defaults(self): + with tempfile.TemporaryDirectory() as tmp: + bible = self._bible(tmp, "# Voice\n\nNo machine-readable block.\n") + words = seq([("Basically", 0.3), ("yes.", 0.3)]) + result, r = run(words, extra=["--voice-bible", str(bible)]) + self.assertIn("no ```cadence block", r.stderr) + self.assertFalse(result["voice_bible_applied"]) + self.assertEqual(len([c for c in by_type(result, "filler") + if c["cls"] == "soft"]), 1) + + def test_missing_bible_fails_loudly(self): + _, r = run(seq([("hi.", 0.3)]), expect=1, + extra=["--voice-bible", "/nope/voice-bible.md"]) + self.assertIn("cannot read voice bible", r.stderr) + + def test_parser_ignores_non_cadence_fences(self): + keep, cut = cutplan_parse("```python\nkeep: so\n```\n") + self.assertEqual((keep, cut), (set(), set())) + + def test_parser_reads_a_cadence_fence(self): + keep, cut = cutplan_parse( + "```cadence\nkeep: so, here's the thing\ncut: um\n```\n") + self.assertEqual(keep, {"so", "here's the thing"}) + self.assertEqual(cut, {"um"}) class TestStutter(unittest.TestCase): @@ -221,7 +444,8 @@ def test_sorted_by_start(self): def test_counts_has_all_types(self): result, _ = run(seq([("hi.", 0.3)])) self.assertEqual(set(result["counts"]), - {"silence", "filler", "stutter", "retake", "marker"}) + {"silence", "filler", "stutter", "retake", "blooper", + "marker"}) def test_thresholds_reflect_flags(self): result, _ = run(seq([("hi.", 0.3)]), @@ -229,15 +453,293 @@ def test_thresholds_reflect_flags(self): "--retake-window", "20"]) self.assertEqual(result["thresholds"], {"min_silence": 1.5, "retake_window": 20, - "retake_run": 4}) + "retake_run": 4, "snap_ms": 250, "keep_ms": 200, + "section_run": 8, "section_window_s": 45.0}) + + +class TestSectionRedo(unittest.TestCase): + """A whole re-read section, not just the repeated words. + + The short-range matcher looks 16 WORDS ahead for a 3-word repeat, which + undersized a real section redo from about 34s to about 11s and left the + abandoned take in the video.""" + + def _para(self, prefix, start, word_dur=0.3, n=10): + return [(f"{prefix}{i}" if i else "memory", word_dur) + for i in range(n)] + + def _redo_words(self, restart_gap): + # Ten words, a pause, then the same ten words again. + text = ["memory", "systems", "are", "still", "the", "hard", "part", + "of", "this", "work"] + pairs = [(w, 0.3) for w in text] + pairs += [(text[0], 0.3, restart_gap)] + pairs += [(w, 0.3) for w in text[1:]] + return seq(pairs) + + def test_full_abandoned_span_is_the_candidate(self): + words = self._redo_words(2.0) + result, _ = run(words, duration=round(words[-1]["end"] + 0.5, 2)) + sec = [c for c in by_type(result, "retake") + if c.get("cls") == "section"] + self.assertEqual(len(sec), 1) + # From the first attempt's start to the restart's start, so the + # abandoned take AND the reset pause both go. + self.assertAlmostEqual(sec[0]["start"], words[0]["start"], places=1) + self.assertAlmostEqual(sec[0]["end"], words[10]["start"], places=1) + + def test_span_includes_the_reset_pause(self): + words = self._redo_words(2.0) + result, _ = run(words, duration=round(words[-1]["end"] + 0.5, 2)) + sec = [c for c in by_type(result, "retake") + if c.get("cls") == "section"][0] + # 10 words x 0.3s of abandoned take plus the 2.0s reset. + self.assertGreater(sec["dur"], 4.5) + + def test_distant_repeat_is_a_callback_not_a_redo(self): + # The same ten words 90s later is a deliberate callback. The locality + # window is what tells them apart, and it must be in SECONDS. + words = self._redo_words(90.0) + result, _ = run(words, duration=round(words[-1]["end"] + 0.5, 2)) + self.assertEqual([c for c in by_type(result, "retake") + if c.get("cls") == "section"], []) + + def test_section_window_is_configurable(self): + words = self._redo_words(90.0) + dur = round(words[-1]["end"] + 0.5, 2) + wide, _ = run(words, duration=dur, + extra=["--section-window-s", "200"]) + self.assertEqual(len([c for c in by_type(wide, "retake") + if c.get("cls") == "section"]), 1) + + def test_short_repeat_does_not_trip_the_section_matcher(self): + words = seq([("the", 0.3), ("cat", 0.3), ("sat", 0.3), + ("the", 0.3, 1.0), ("cat", 0.3), ("sat", 0.3)]) + result, _ = run(words, duration=round(words[-1]["end"] + 0.5, 2)) + self.assertEqual([c for c in by_type(result, "retake") + if c.get("cls") == "section"], []) + + def test_section_run_is_configurable(self): + words = seq([("the", 0.3), ("cat", 0.3), ("sat", 0.3), + ("the", 0.3, 1.0), ("cat", 0.3), ("sat", 0.3)]) + dur = round(words[-1]["end"] + 0.5, 2) + result, _ = run(words, duration=dur, extra=["--section-run", "3"]) + self.assertEqual(len([c for c in by_type(result, "retake") + if c.get("cls") == "section"]), 1) + + +class TestBlooper(unittest.TestCase): + """The "Oh fuck." at 13:59 that the mechanical pass left in, and would + have shipped.""" + + def test_expletive_next_to_a_pause_is_a_high_severity_blooper(self): + words = seq([("Oh", 0.3), ("fuck.", 0.3), ("Let", 0.3, 2.0), + ("me", 0.3), ("go.", 0.3)]) + silence = [{"start": 0.6, "end": 2.6, "dur": 2.0}] + result, _ = run(words, duration=round(words[-1]["end"] + 0.5, 2), + silence=silence) + bl = by_type(result, "blooper") + self.assertEqual([c["text"] for c in bl], ["fuck."]) + self.assertEqual(bl[0]["severity"], "high") + self.assertEqual(bl[0]["cls"], "reset") + + def test_expletive_in_continuous_speech_is_flagged_for_an_ear(self): + # "that damn term" is scripted usage, not a flub. + words = seq([("that", 0.3), ("damn", 0.3), ("term", 0.3), + ("again.", 0.3)]) + result, _ = run(words, duration=round(words[-1]["end"] + 0.1, 2), + silence=[]) + bl = by_type(result, "blooper") + self.assertEqual(len(bl), 1) + self.assertEqual(bl[0]["severity"], "med") + self.assertEqual(bl[0]["cls"], "ambiguous") + self.assertIn("confirm by ear", bl[0]["reason"]) + + def test_reset_phrase_is_a_blooper(self): + words = seq([("Scratch", 0.3), ("that.", 0.3), ("Again.", 0.3)]) + result, _ = run(words, duration=round(words[-1]["end"] + 0.1, 2), + silence=[]) + self.assertEqual([c["text"] for c in by_type(result, "blooper")], + ["Scratch that."]) + + def test_clean_speech_has_no_bloopers(self): + words = seq([("This", 0.3), ("is", 0.3), ("fine.", 0.3)]) + result, _ = run(words) + self.assertEqual(by_type(result, "blooper"), []) + + def test_blooper_vocabulary_is_overridable(self): + words = seq([("That", 0.3), ("blast.", 0.3)]) + dur = round(words[-1]["end"] + 0.1, 2) + default, _ = run(words, duration=dur, silence=[]) + custom, _ = run(words, duration=dur, silence=[], + extra=["--blooper-cues", "blast"]) + self.assertEqual(by_type(default, "blooper"), []) + self.assertEqual([c["text"] for c in by_type(custom, "blooper")], + ["blast."]) + + def test_silence_source_is_recorded_as_audio(self): + result, _ = run(seq([("hi.", 0.3)])) + self.assertEqual(result["silence_source"], "audio") + + +class TestAudioMapIsMandatory(unittest.TestCase): + """The no-silent-fallback guarantee. Silence must come from the audio; + a missing or malformed map has to stop the stage, never quietly degrade + back to transcript gaps (the 2026-07-24 defect).""" + + def _amap(self, tmp, payload): + p = Path(tmp) / "audio-map.json" + p.write_text(json.dumps(payload)) + return p + + def test_omitting_the_audio_map_is_a_usage_error(self): + _, r = run(seq([("hi.", 0.3)]), expect=2, with_audio_map=False) + self.assertIn("--audio-map", r.stderr) + + def test_unreadable_audio_map_fails_loudly(self): + with tempfile.TemporaryDirectory() as tmp: + src = Path(tmp) / "words.json" + src.write_text(json.dumps({"media": "m.mp4", "duration": 1.0, + "text": "", "words": []})) + r = subprocess.run( + [sys.executable, str(SCRIPT), str(src), "-o", + str(Path(tmp) / "out.json"), "--audio-map", + str(Path(tmp) / "nope.json")], + capture_output=True, text=True) + self.assertEqual(r.returncode, 1) + self.assertIn("analyze_audio.py", r.stderr) + + def test_audio_map_without_silence_key_fails(self): + with tempfile.TemporaryDirectory() as tmp: + src = Path(tmp) / "words.json" + src.write_text(json.dumps({"media": "m.mp4", "duration": 1.0, + "text": "", "words": []})) + amap = self._amap(tmp, {"duration": 1.0}) + r = subprocess.run( + [sys.executable, str(SCRIPT), str(src), "-o", + str(Path(tmp) / "out.json"), "--audio-map", str(amap)], + capture_output=True, text=True) + self.assertEqual(r.returncode, 1) + self.assertIn("silence", r.stderr) + + +class TestSilenceComesFromAudioNotGaps(unittest.TestCase): + """The core regression. Transcript timestamps absorb pauses, so the same + words must yield silence candidates driven by the AUDIO map, whatever the + word timestamps happen to say.""" + + def test_pause_absorbed_word_end_still_yields_a_silence_candidate(self): + # The real shape of the bug: "about." is timestamped 32.16 -> 34.64, + # a 2.5s "word" that swallowed the pause after it. The gap to the + # next word reads 0.0, but the audio is silent from 32.4 to 34.6. + words = [word("about.", 32.16, 34.64, 0), + word("One", 34.64, 34.9, 1)] + silence = [{"start": 32.4, "end": 34.6, "dur": 2.2}] + result, _ = run(words, duration=35.0, silence=silence) + sil = by_type(result, "silence") + self.assertEqual(len(sil), 1) + # The full silence is 32.4 to 34.6; the candidate is the part to + # REMOVE, tightened to 200ms of breathing room split evenly. + self.assertEqual(sil[0]["silence_start"], 32.4) + self.assertEqual(sil[0]["silence_end"], 34.6) + self.assertAlmostEqual(sil[0]["start"], 32.5, places=2) + self.assertAlmostEqual(sil[0]["end"], 34.5, places=2) + self.assertEqual(sil[0]["severity"], "high") + # The transcript said the gap here was 0.0 (34.64 - 34.64). The + # audio said 2.2s. The audio wins, which is the whole fix. + self.assertEqual(words[1]["start"] - words[0]["end"], 0.0) + + def test_zero_gap_words_over_silent_audio_are_still_found(self): + # Words laid end to end (every transcript gap 0.0) over audio that is + # actually silent in the middle. The old gap detector found nothing. + words = seq([("a", 0.3), ("b", 0.3), ("c", 0.3)]) + silence = [{"start": 0.3, "end": 0.6, "dur": 0.3}, + {"start": 0.6, "end": 2.4, "dur": 1.8}] + result, _ = run(words, duration=3.0, silence=silence) + self.assertGreaterEqual(len(by_type(result, "silence")), 1) + + def test_no_audio_silence_means_no_silence_candidates(self): + words = seq([("a", 0.3), ("b", 0.3, 5.0)]) + result, _ = run(words, duration=10.0, silence=[]) + self.assertEqual(by_type(result, "silence"), []) + + +class TestEdgeSnapping(unittest.TestCase): + """Candidate edges move into audio-verified silence, which is what makes + 'never cut inside a word' structural rather than aspirational.""" + + def test_edges_snap_into_neighbouring_silence(self): + # A hard filler whose transcript end overruns into the next word by + # 60ms; real silence sits at 1.0 to 1.5. + words = [word("uh", 0.5, 1.06, 0), word("okay.", 1.5, 1.9, 1)] + silence = [{"start": 0.0, "end": 0.5, "dur": 0.5}, + {"start": 1.0, "end": 1.5, "dur": 0.5}] + result, _ = run(words, duration=2.5, silence=silence) + fil = by_type(result, "filler")[0] + self.assertTrue(fil["snapped"]) + self.assertGreaterEqual(fil["end"], 1.0) + self.assertLessEqual(fil["end"], 1.5) + + def test_unsnappable_edges_are_flagged_not_faked(self): + # No silence anywhere near the candidate: the edge stays put and the + # candidate is marked so the gate knows it carries timestamp risk. + words = [word("uh", 5.0, 5.4, 0), word("okay.", 5.4, 5.8, 1)] + result, _ = run(words, duration=6.0, silence=[]) + fil = by_type(result, "filler")[0] + self.assertFalse(fil["snapped"]) + self.assertEqual(fil["start"], 5.0) + self.assertEqual(result["unsnapped"], len( + [c for c in result["candidates"] if not c["snapped"]])) + + def test_unsnapped_count_is_reported_on_stderr(self): + words = [word("uh", 5.0, 5.4, 0), word("okay.", 5.4, 5.8, 1)] + _, r = run(words, duration=6.0, silence=[]) + self.assertIn("could not reach an audio silence", r.stderr) + + def test_stutter_edge_lands_in_the_silence_between_takes(self): + # The "n-now" artifact: the first "now" is timestamped through the + # pause (0.5 -> 1.15) so cutting to its end clipped the repeat's + # onset at 1.2. Snapping puts the edge inside the 1.0 to 1.2 silence. + words = [word("now,", 0.5, 1.15, 0), word("now", 1.2, 1.5, 1)] + silence = [{"start": 0.0, "end": 0.5, "dur": 0.5}, + {"start": 1.0, "end": 1.2, "dur": 0.2}] + result, _ = run(words, duration=2.0, silence=silence) + stut = by_type(result, "stutter")[0] + self.assertTrue(stut["snapped"]) + self.assertLessEqual(stut["end"], 1.2) + self.assertGreaterEqual(stut["end"], 1.0) + + def test_silence_candidates_are_never_moved(self): + words = seq([("a", 0.3), ("b", 0.3, 1.0)]) + result, _ = run(words) + for c in by_type(result, "silence"): + self.assertTrue(c["snapped"]) + + def test_snap_budget_is_configurable(self): + # Start sits on the edge of the leading silence (always snappable); + # the end is 0.6s from the next silence, so only the loose budget + # reaches it. Both edges must snap for the candidate to count. + words = [word("uh", 0.5, 0.9, 0), word("okay.", 2.0, 2.4, 1)] + silence = [{"start": 0.0, "end": 0.5, "dur": 0.5}, + {"start": 1.5, "end": 2.0, "dur": 0.5}] + tight, _ = run(words, duration=3.0, silence=silence, + extra=["--snap-ms", "50"]) + loose, _ = run(words, duration=3.0, silence=silence, + extra=["--snap-ms", "800"]) + self.assertFalse(by_type(tight, "filler")[0]["snapped"]) + self.assertTrue(by_type(loose, "filler")[0]["snapped"]) + self.assertEqual(by_type(loose, "filler")[0]["end"], 1.5) class TestExitCodes(unittest.TestCase): def test_missing_input_fails(self): with tempfile.TemporaryDirectory() as tmp: + amap = Path(tmp) / "audio-map.json" + amap.write_text(json.dumps({"duration": 1.0, "silence": []})) r = subprocess.run([sys.executable, str(SCRIPT), str(Path(tmp) / "nope.json"), "-o", - str(Path(tmp) / "out.json")], + str(Path(tmp) / "out.json"), + "--audio-map", str(amap)], capture_output=True, text=True) self.assertEqual(r.returncode, 1) @@ -250,11 +752,67 @@ def test_bad_json_fails(self): with tempfile.TemporaryDirectory() as tmp: src = Path(tmp) / "bad.json" src.write_text("{not json") + amap = Path(tmp) / "audio-map.json" + amap.write_text(json.dumps({"duration": 1.0, "silence": []})) r = subprocess.run([sys.executable, str(SCRIPT), str(src), "-o", - str(Path(tmp) / "out.json")], + str(Path(tmp) / "out.json"), + "--audio-map", str(amap)], capture_output=True, text=True) self.assertEqual(r.returncode, 1) +class TestProtectedFlags(unittest.TestCase): + """cutplan-flags is appended AFTER the skill's own arguments, and argparse + lets the later one win. Without this, an override file could point + --audio-map elsewhere and break the two-source rule with no visible sign.""" + + def test_a_single_occurrence_is_fine(self): + self.assertEqual(cutplan.protected_conflicts( + ["w.json", "-o", "out.json", "--audio-map", "a.json"]), []) + + def test_a_duplicated_audio_map_is_a_conflict(self): + self.assertEqual(cutplan.protected_conflicts( + ["--audio-map", "a.json", "--audio-map", "evil.json"]), + ["--audio-map"]) + + def test_a_duplicated_output_is_a_conflict(self): + self.assertEqual(cutplan.protected_conflicts( + ["-o", "out.json", "-o", "elsewhere.json"]), ["-o"]) + + def test_the_two_spellings_of_output_collide(self): + self.assertEqual(cutplan.protected_conflicts( + ["-o", "out.json", "--output", "elsewhere.json"]), + ["--output", "-o"]) + + def test_equals_form_is_caught(self): + self.assertEqual(cutplan.protected_conflicts( + ["--audio-map", "a.json", "--audio-map=evil.json"]), + ["--audio-map"]) + + def test_a_duplicated_voice_bible_is_a_conflict(self): + self.assertEqual(cutplan.protected_conflicts( + ["--voice-bible", "v.md", "--voice-bible", "other.md"]), + ["--voice-bible"]) + + def test_unprotected_flags_may_repeat(self): + self.assertEqual(cutplan.protected_conflicts( + ["--min-silence", "0.3", "--min-silence", "0.5"]), []) + + def test_the_cli_refuses_a_redirected_audio_map(self): + with tempfile.TemporaryDirectory() as tmp: + words = Path(tmp) / "w.json" + words.write_text(json.dumps({"words": []})) + amap = Path(tmp) / "audio-map.json" + amap.write_text(json.dumps({"duration": 1.0, "silence": []})) + r = subprocess.run( + [sys.executable, str(SCRIPT), str(words), + "-o", str(Path(tmp) / "out.json"), + "--audio-map", str(amap), + "--audio-map", str(Path(tmp) / "evil.json")], + capture_output=True, text=True) + self.assertEqual(r.returncode, 2) + self.assertIn("two-source rule", r.stderr) + + if __name__ == "__main__": unittest.main() diff --git a/skills/mc-cut/scripts/tests/test-edited_transcript.py b/skills/mc-cut/scripts/tests/test-edited_transcript.py new file mode 100644 index 0000000..c1e48bb --- /dev/null +++ b/skills/mc-cut/scripts/tests/test-edited_transcript.py @@ -0,0 +1,201 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.11" +# /// +"""Tests for edited_transcript.py: the reconstruction the content-editorial +pass reads, and its dual timecodes.""" +import importlib.util +import json +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +SCRIPT = Path(__file__).resolve().parent.parent / "edited_transcript.py" +spec = importlib.util.spec_from_file_location("edited_transcript", SCRIPT) +et = importlib.util.module_from_spec(spec) +spec.loader.exec_module(et) + + +def words(spans): + return [{"word": t, "start": s, "end": e, "confidence": 1.0, "i": i, + "gap_before": 0.0, "gap_after": 0.0} + for i, (t, s, e) in enumerate(spans)] + + +# Keeps source 0-10 and 20-30; 10-20 is cut away. +EDL = {"source": "raw/cam.mp4", "segments": [ + {"source": "raw/cam.mp4", "start": 0.0, "end": 10.0}, + {"source": "raw/cam.mp4", "start": 20.0, "end": 30.0}, +]} +MAPPING = et.build_map(EDL) + + +class TestBuildMap(unittest.TestCase): + def test_offsets_accumulate(self): + self.assertEqual([s["offset"] for s in MAPPING], [0.0, 10.0]) + + def test_clean_duration_is_the_kept_total(self): + self.assertEqual(et.clean_duration(MAPPING), 20.0) + + +class TestKeepWords(unittest.TestCase): + def test_word_in_a_kept_span_survives(self): + kept = et.keep_words(words([("hello", 1.0, 1.5)]), MAPPING) + self.assertEqual(len(kept), 1) + self.assertEqual(kept[0]["clean_start"], 1.0) + self.assertEqual(kept[0]["src_start"], 1.0) + + def test_word_in_a_removed_span_is_dropped(self): + self.assertEqual(et.keep_words(words([("gone", 14.0, 14.5)]), + MAPPING), []) + + def test_clean_time_is_shifted_for_later_segments(self): + # Source 25.0 sits 5s into the second kept span, which starts at 10. + kept = et.keep_words(words([("later", 25.0, 25.5)]), MAPPING) + self.assertEqual(kept[0]["clean_start"], 15.0) + self.assertEqual(kept[0]["src_start"], 25.0) + + def test_both_timecodes_are_always_present(self): + kept = et.keep_words(words([("a", 1.0, 1.5), ("b", 25.0, 25.5)]), + MAPPING) + for w in kept: + for field in ("clean_start", "clean_end", "src_start", "src_end"): + self.assertIn(field, w) + + def test_word_straddling_a_boundary_is_kept_once(self): + # Midpoint 9.75 is inside the first kept span. + kept = et.keep_words(words([("edge", 9.5, 10.0)]), MAPPING) + self.assertEqual(len(kept), 1) + self.assertEqual(kept[0]["segment"], 0) + + def test_word_mostly_outside_is_dropped(self): + # Midpoint 10.5 falls in the removed span. + self.assertEqual(et.keep_words(words([("edge", 9.8, 11.2)]), + MAPPING), []) + + def test_no_word_is_emitted_twice(self): + kept = et.keep_words(words([("a", 9.9, 10.1)]), MAPPING) + self.assertLessEqual(len(kept), 1) + + def test_clean_times_stay_inside_the_timeline(self): + kept = et.keep_words(words([("edge", 9.5, 10.4)]), MAPPING) + total = et.clean_duration(MAPPING) + for w in kept: + self.assertGreaterEqual(w["clean_start"], 0.0) + self.assertLessEqual(w["clean_end"], total + 1e-6) + + def test_reading_order_is_preserved(self): + kept = et.keep_words(words([ + ("one", 1.0, 1.4), ("cut", 15.0, 15.4), ("two", 25.0, 25.4)]), + MAPPING) + self.assertEqual([w["word"] for w in kept], ["one", "two"]) + + def test_source_filter_selects_one_source(self): + edl = {"segments": [ + {"source": "a.mp4", "start": 0.0, "end": 10.0}, + {"source": "b.mp4", "start": 0.0, "end": 10.0}]} + m = et.build_map(edl) + kept = et.keep_words(words([("x", 1.0, 1.4)]), m, source="b.mp4") + self.assertEqual(kept[0]["segment"], 1) + + +class TestParagraphs(unittest.TestCase): + def test_pause_opens_a_new_paragraph(self): + kept = et.keep_words(words([ + ("a", 1.0, 1.4), ("b", 1.4, 1.8), ("c", 4.0, 4.4)]), MAPPING) + self.assertEqual(len(et.paragraphs(kept)), 2) + + def test_continuous_speech_is_one_paragraph(self): + kept = et.keep_words(words([ + ("a", 1.0, 1.4), ("b", 1.4, 1.8), ("c", 1.8, 2.2)]), MAPPING) + self.assertEqual(len(et.paragraphs(kept)), 1) + + def test_segment_change_opens_a_new_paragraph(self): + # A seam is exactly where the pass needs to check the join reads, + # so it must not be buried mid-paragraph. + kept = et.keep_words(words([("a", 9.0, 9.4), ("b", 20.0, 20.4)]), + MAPPING) + self.assertEqual(len(et.paragraphs(kept)), 2) + + def test_empty_input_is_no_paragraphs(self): + self.assertEqual(et.paragraphs([]), []) + + +class TestRenderMarkdown(unittest.TestCase): + def test_output_carries_both_timecodes(self): + kept = et.keep_words(words([("later", 25.0, 25.5)]), MAPPING) + md = et.render_markdown(kept, MAPPING) + self.assertIn("0:15.00", md) # clean + self.assertIn("src 0:25.00", md) # source + + def test_output_states_the_runtime(self): + kept = et.keep_words(words([("a", 1.0, 1.4)]), MAPPING) + self.assertIn("0:20.00", et.render_markdown(kept, MAPPING)) + + def test_output_warns_against_hand_conversion(self): + kept = et.keep_words(words([("a", 1.0, 1.4)]), MAPPING) + self.assertIn("Never convert between them by hand", + et.render_markdown(kept, MAPPING)) + + def test_words_appear_in_order(self): + kept = et.keep_words(words([ + ("one", 1.0, 1.4), ("two", 1.4, 1.8)]), MAPPING) + md = et.render_markdown(kept, MAPPING) + self.assertLess(md.index("one"), md.index("two")) + + +class TestCli(unittest.TestCase): + def _run(self, word_spans, extra=None): + tmp = tempfile.TemporaryDirectory() + d = Path(tmp.name) + (d / "words.json").write_text(json.dumps( + {"duration": 30.0, "words": words(word_spans)})) + (d / "edl.json").write_text(json.dumps(EDL)) + r = subprocess.run( + [sys.executable, str(SCRIPT), str(d / "words.json"), + "--edl", str(d / "edl.json"), "-o", str(d / "edited.md"), + "-j", str(d / "edited.json"), *(extra or [])], + capture_output=True, text=True) + return r, d, tmp + + def test_writes_both_outputs(self): + r, d, tmp = self._run([("a", 1.0, 1.4), ("b", 25.0, 25.4)]) + self.assertEqual(r.returncode, 0, r.stderr) + self.assertTrue((d / "edited.md").is_file()) + self.assertTrue((d / "edited.json").is_file()) + tmp.cleanup() + + def test_summary_counts_kept_and_dropped(self): + r, _, tmp = self._run([("a", 1.0, 1.4), ("gone", 15.0, 15.4)]) + summary = json.loads(r.stdout) + self.assertEqual(summary["words_kept"], 1) + self.assertEqual(summary["words_dropped"], 1) + tmp.cleanup() + + def test_json_output_carries_dual_timecodes(self): + r, d, tmp = self._run([("b", 25.0, 25.4)]) + data = json.loads((d / "edited.json").read_text()) + w = data["words"][0] + self.assertEqual(w["clean_start"], 15.0) + self.assertEqual(w["src_start"], 25.0) + self.assertEqual(data["clean_duration"], 20.0) + tmp.cleanup() + + def test_missing_input_is_a_usage_error(self): + r = subprocess.run( + [sys.executable, str(SCRIPT), "/nope/words.json", + "--edl", "/nope/edl.json", "-o", "/tmp/out.md"], + capture_output=True, text=True) + self.assertEqual(r.returncode, 2) + + def test_help_exits_zero(self): + r = subprocess.run([sys.executable, str(SCRIPT), "--help"], + capture_output=True, text=True) + self.assertEqual(r.returncode, 0) + self.assertIn("--edl", r.stdout) + + +if __name__ == "__main__": + unittest.main() diff --git a/skills/mc-cut/scripts/tests/test-normalize_source.py b/skills/mc-cut/scripts/tests/test-normalize_source.py new file mode 100644 index 0000000..8ed3ff5 --- /dev/null +++ b/skills/mc-cut/scripts/tests/test-normalize_source.py @@ -0,0 +1,322 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.11" +# /// +"""Tests for normalize_source.py: the crop geometry, and the one guarantee +the whole script rests on, that a spatial correction moves nothing in time.""" +import importlib.util +import json +import shutil +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +SCRIPTS = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(SCRIPTS)) +SCRIPT = SCRIPTS / "normalize_source.py" +spec = importlib.util.spec_from_file_location("normalize_source", SCRIPT) +ns = importlib.util.module_from_spec(spec) +spec.loader.exec_module(ns) + +import composite_core as core # noqa: E402 + +FFMPEG = shutil.which("ffmpeg") and shutil.which("ffprobe") + + +class TestParsers(unittest.TestCase): + def test_parse_crop(self): + self.assertEqual(ns.parse_crop("3360:1880:240:140"), + (240, 140, 3360, 1880)) + + def test_parse_crop_rejects_wrong_arity(self): + with self.assertRaises(ValueError): + ns.parse_crop("100:200:300") + + def test_parse_crop_rejects_non_integers(self): + with self.assertRaises(ValueError): + ns.parse_crop("a:b:c:d") + + def test_parse_crop_rejects_non_positive_size(self): + with self.assertRaises(ValueError): + ns.parse_crop("0:100:0:0") + + def test_parse_size(self): + self.assertEqual(ns.parse_size("3840x2160"), (3840, 2160)) + + def test_parse_size_is_case_insensitive(self): + self.assertEqual(ns.parse_size("1920X1080"), (1920, 1080)) + + def test_parse_size_rejects_garbage(self): + with self.assertRaises(ValueError): + ns.parse_size("1920") + + def test_parse_aspect_ratio_form(self): + self.assertAlmostEqual(ns.parse_aspect("16:9"), 16 / 9) + + def test_parse_aspect_decimal_form(self): + self.assertAlmostEqual(ns.parse_aspect("1.5"), 1.5) + + def test_parse_aspect_rejects_zero_denominator(self): + with self.assertRaises(ValueError): + ns.parse_aspect("16:0") + + +class TestShiftCrop(unittest.TestCase): + def test_shifts_right_and_down(self): + self.assertEqual(ns.shift_crop((100, 100, 200, 100), 50, 20, 1000, 500), + (150, 120, 200, 100)) + + def test_shifts_left_with_a_negative_offset(self): + # The real defect: the subject sat about 5 percent left of centre. + self.assertEqual(ns.shift_crop((100, 0, 200, 100), -60, 0, 1000, 500), + (40, 0, 200, 100)) + + def test_clamps_at_the_left_edge_rather_than_failing(self): + self.assertEqual(ns.shift_crop((10, 10, 200, 100), -500, -500, + 1000, 500), (0, 0, 200, 100)) + + def test_clamps_at_the_right_edge(self): + x, _, w, _ = ns.shift_crop((700, 0, 200, 100), 500, 0, 1000, 500) + self.assertEqual(x + w, 1000) + + def test_zero_offset_is_a_no_op(self): + rect = (100, 50, 200, 100) + self.assertEqual(ns.shift_crop(rect, 0, 0, 1000, 500), rect) + + +class TestFitAspect(unittest.TestCase): + def test_already_correct_aspect_is_unchanged_in_size(self): + _, _, w, h = ns.fit_aspect((0, 0, 1920, 1080), 16 / 9) + self.assertAlmostEqual(w / h, 16 / 9, places=2) + + def test_too_wide_rect_is_narrowed(self): + _, _, w, h = ns.fit_aspect((0, 0, 2000, 1000), 16 / 9) + self.assertAlmostEqual(w / h, 16 / 9, places=2) + self.assertLessEqual(w, 2000) + self.assertLessEqual(h, 1000) + + def test_too_tall_rect_is_shortened(self): + _, _, w, h = ns.fit_aspect((0, 0, 1600, 1200), 16 / 9) + self.assertAlmostEqual(w / h, 16 / 9, places=2) + self.assertLessEqual(h, 1200) + + def test_fit_only_ever_shrinks(self): + # Growing could pull the baked-in border back into frame, which is + # the whole thing being removed. + for rect in [(0, 0, 2000, 1000), (0, 0, 1600, 1200), + (10, 10, 999, 501)]: + _, _, w, h = ns.fit_aspect(rect, 16 / 9) + self.assertLessEqual(w, rect[2]) + self.assertLessEqual(h, rect[3]) + + def test_fit_keeps_the_centre(self): + x, y, w, h = ns.fit_aspect((100, 100, 2000, 1000), 16 / 9) + self.assertAlmostEqual(x + w / 2, 100 + 1000, delta=2) + self.assertAlmostEqual(y + h / 2, 100 + 500, delta=2) + + def test_dimensions_are_even(self): + _, _, w, h = ns.fit_aspect((0, 0, 1999, 1001), 16 / 9) + self.assertEqual(w % 2, 0) + self.assertEqual(h % 2, 0) + + +class TestClampRect(unittest.TestCase): + def test_oversized_rect_is_clamped_to_the_frame(self): + self.assertEqual(ns.clamp_rect((0, 0, 5000, 5000), 1920, 1080), + (0, 0, 1920, 1080)) + + def test_origin_is_pulled_back_inside(self): + x, y, w, h = ns.clamp_rect((1900, 1000, 200, 200), 1920, 1080) + self.assertLessEqual(x + w, 1920) + self.assertLessEqual(y + h, 1080) + + def test_all_values_are_even(self): + for v in ns.clamp_rect((3, 5, 101, 203), 1920, 1080): + self.assertEqual(v % 2, 0) + + +class TestBuildNormalizeCommand(unittest.TestCase): + def test_crop_filter_is_in_wh_xy_order(self): + cmd = ns.build_normalize_command("in.mp4", "out.mp4", + (240, 140, 3360, 1880), "30/1", + encoder="libx264") + vf = cmd[cmd.index("-vf") + 1] + self.assertIn("crop=3360:1880:240:140", vf) + + def test_audio_is_copied_never_re_encoded(self): + cmd = ns.build_normalize_command("in.mp4", "out.mp4", (0, 0, 100, 100), + "30/1", encoder="libx264") + self.assertIn("-c:a", cmd) + self.assertEqual(cmd[cmd.index("-c:a") + 1], "copy") + + def test_frame_rate_is_forced_so_the_master_stays_cfr(self): + cmd = ns.build_normalize_command("in.mp4", "out.mp4", (0, 0, 100, 100), + "30000/1001", encoder="libx264") + self.assertIn("fps=30000/1001", cmd[cmd.index("-vf") + 1]) + + def test_output_size_adds_a_scale_after_the_crop(self): + cmd = ns.build_normalize_command("in.mp4", "out.mp4", + (0, 0, 3360, 1880), "30/1", + output_size=(3840, 2160), + encoder="libx264") + vf = cmd[cmd.index("-vf") + 1] + self.assertLess(vf.index("crop="), vf.index("scale=")) + + def test_no_output_size_means_no_scale(self): + cmd = ns.build_normalize_command("in.mp4", "out.mp4", (0, 0, 100, 100), + "30/1", encoder="libx264") + self.assertNotIn("scale=", cmd[cmd.index("-vf") + 1]) + + def test_no_trim_or_seek_flags_anywhere(self): + # The guarantee, at the command level: nothing here can move time. + cmd = ns.build_normalize_command("in.mp4", "out.mp4", (0, 0, 100, 100), + "30/1", encoder="libx264") + for flag in ("-ss", "-t", "-to", "-itsoffset"): + self.assertNotIn(flag, cmd) + + +class TestCliUsage(unittest.TestCase): + def _run(self, argv): + return subprocess.run([sys.executable, str(SCRIPT), *argv], + capture_output=True, text=True) + + def test_missing_media_is_usage_error(self): + r = self._run(["/nope/take.mp4", "-o", "/tmp/out.mp4", "--auto"]) + self.assertEqual(r.returncode, 2) + + def test_help_exits_zero(self): + r = self._run(["--help"]) + self.assertEqual(r.returncode, 0) + self.assertIn("--auto", r.stdout) + + +@unittest.skipUnless(FFMPEG, "ffmpeg/ffprobe not installed") +class TestNormalizeEndToEnd(unittest.TestCase): + """Issue G's second half: correcting a baked-in border, and proving the + correction left every timecode alone.""" + + @classmethod + def setUpClass(cls): + cls.tmp = tempfile.TemporaryDirectory() + cls.dir = Path(cls.tmp.name) + cls.src = cls.dir / "bordered.mp4" + # 320x180 with content shrunk to 240x134 inside a black frame, plus + # audio, so the copy-through path is exercised. + r = subprocess.run( + ["ffmpeg", "-y", "-f", "lavfi", "-t", "3", "-i", + "testsrc2=size=320x180:rate=30", "-f", "lavfi", "-t", "3", + "-i", "sine=frequency=440:sample_rate=48000", "-shortest", + "-vf", "scale=240:134,pad=320:180:40:23:black", + "-c:v", "libx264", "-preset", "ultrafast", "-crf", "28", + "-pix_fmt", "yuv420p", "-c:a", "aac", str(cls.src)], + capture_output=True, text=True) + assert r.returncode == 0, r.stderr + + @classmethod + def tearDownClass(cls): + cls.tmp.cleanup() + + def _run(self, argv): + return subprocess.run([sys.executable, str(SCRIPT), *argv], + capture_output=True, text=True) + + def test_explicit_crop_produces_a_corrected_master(self): + out = self.dir / "fixed-explicit.mp4" + r = self._run([str(self.src), "-o", str(out), + "--crop", "240:134:40:23"]) + self.assertEqual(r.returncode, 0, r.stderr) + self.assertTrue(out.is_file()) + self.assertEqual(core.probe_dims(out), (240, 134)) + + def test_auto_detects_and_removes_the_border(self): + out = self.dir / "fixed-auto.mp4" + r = self._run([str(self.src), "-o", str(out), "--auto"]) + self.assertEqual(r.returncode, 0, r.stderr) + w, h = core.probe_dims(out) + self.assertLess(w, 320) + self.assertLess(h, 180) + + def test_duration_is_preserved(self): + """The guarantee that lets the existing EDL survive.""" + out = self.dir / "fixed-dur.mp4" + r = self._run([str(self.src), "-o", str(out), + "--crop", "240:134:40:23"]) + self.assertEqual(r.returncode, 0, r.stderr) + summary = json.loads(r.stdout) + self.assertTrue(summary["timecodes_preserved"]) + self.assertAlmostEqual(summary["output_duration"], + summary["source_duration"], delta=0.1) + + def test_audio_survives(self): + out = self.dir / "fixed-audio.mp4" + self._run([str(self.src), "-o", str(out), "--crop", "240:134:40:23"]) + self.assertTrue(core.probe_has_audio(out)) + + def test_output_size_restores_the_delivery_resolution(self): + out = self.dir / "fixed-scaled.mp4" + r = self._run([str(self.src), "-o", str(out), + "--crop", "240:134:40:23", "--output-size", "320x180"]) + self.assertEqual(r.returncode, 0, r.stderr) + self.assertEqual(core.probe_dims(out), (320, 180)) + + def test_recentre_offset_moves_the_window(self): + left = self.dir / "fixed-left.mp4" + right = self.dir / "fixed-right.mp4" + self._run([str(self.src), "-o", str(left), "--crop", "200:112:60:34", + "--offset-x", "-20"]) + self._run([str(self.src), "-o", str(right), "--crop", "200:112:60:34", + "--offset-x", "20"]) + a = json.loads(self._run([str(self.src), "-o", str(left), + "--crop", "200:112:60:34", + "--offset-x", "-20"]).stdout) + b = json.loads(self._run([str(self.src), "-o", str(right), + "--crop", "200:112:60:34", + "--offset-x", "20"]).stdout) + self.assertNotEqual(a["crop"], b["crop"]) + + def test_target_aspect_is_honoured(self): + out = self.dir / "fixed-aspect.mp4" + r = self._run([str(self.src), "-o", str(out), + "--crop", "240:134:40:23", "--target-aspect", "16:9"]) + self.assertEqual(r.returncode, 0, r.stderr) + w, h = core.probe_dims(out) + self.assertAlmostEqual(w / h, 16 / 9, places=1) + + def test_clean_source_with_auto_reports_nothing_to_do(self): + clean = self.dir / "clean.mp4" + r = subprocess.run( + ["ffmpeg", "-y", "-f", "lavfi", "-t", "2", "-i", + "testsrc2=size=320x180:rate=30", "-c:v", "libx264", + "-preset", "ultrafast", "-crf", "28", "-pix_fmt", "yuv420p", + str(clean)], capture_output=True, text=True) + assert r.returncode == 0, r.stderr + r = self._run([str(clean), "-o", str(self.dir / "noop.mp4"), "--auto"]) + self.assertEqual(r.returncode, 1) + self.assertIn("no border ring detected", r.stderr) + + def test_summary_tells_the_caller_not_to_re_cut(self): + out = self.dir / "fixed-msg.mp4" + r = self._run([str(self.src), "-o", str(out), + "--crop", "240:134:40:23"]) + self.assertIn("Do NOT re-transcribe or re-cut", r.stderr) + + def test_crop_and_auto_are_mutually_exclusive(self): + r = self._run([str(self.src), "-o", str(self.dir / "x.mp4"), + "--crop", "240:134:40:23", "--auto"]) + self.assertEqual(r.returncode, 2) + + def test_one_of_crop_or_auto_is_required(self): + r = self._run([str(self.src), "-o", str(self.dir / "x.mp4")]) + self.assertEqual(r.returncode, 2) + + def test_no_temp_file_survives(self): + out = self.dir / "fixed-clean-temp.mp4" + self._run([str(self.src), "-o", str(out), "--crop", "240:134:40:23"]) + leftovers = [p for p in self.dir.iterdir() if ".normalizing" in p.name] + self.assertEqual(leftovers, []) + + +if __name__ == "__main__": + unittest.main() diff --git a/skills/mc-cut/scripts/tests/test-preflight.py b/skills/mc-cut/scripts/tests/test-preflight.py index 7bc0bee..e5d55ca 100644 --- a/skills/mc-cut/scripts/tests/test-preflight.py +++ b/skills/mc-cut/scripts/tests/test-preflight.py @@ -179,10 +179,13 @@ def test_probe_qc_and_disk(self): self.assertIsNone(f["cfr_master"]) self.assertAlmostEqual(f["duration"], 2.0, delta=0.2) self.assertAlmostEqual(f["fps"], 30.0, delta=0.1) - self.assertEqual(len(f["qc_frames"]), 2) + # Several samples across the take, not just first and last: a frame + # effect can be switched on after recording starts. + self.assertEqual(len(f["qc_frames"]), mod.DEFAULT_QC_SAMPLES) for frame in f["qc_frames"]: self.assertTrue(Path(frame).is_file()) self.assertTrue(summary["all_cfr"]) + self.assertTrue(summary["qc_ok"]) self.assertIn("free_bytes", summary["disk"]) def test_disk_gate_refuses_remux_before_any_write(self): @@ -230,5 +233,230 @@ def test_low_disk_without_remux_reports_and_exits_0(self): self.assertTrue(summary["all_cfr"]) +def frame(w, h, fill=(30, 30, 30), border=None, depth=0): + """A synthetic rgb24 frame: optional flat border ring around noisy-ish + interior content.""" + px = bytearray() + for y in range(h): + for x in range(w): + d = min(x, y, w - 1 - x, h - 1 - y) + if border is not None and d < depth: + px += bytes(border) + else: + # Deterministic but varied interior, so it never reads flat. + px += bytes(((x * 37 + y * 91) % 256, + (x * 17 + y * 53) % 256, + (x * 71 + y * 29) % 256)) + return bytes(px) + + +def layered_frame(w, h, layers): + """A frame whose border is several flat colours, outermost first: + layers is [(colour, thickness), ...]. The real defect was a black outer + border plus a rounded orange ring.""" + px = bytearray() + for y in range(h): + for x in range(w): + d = min(x, y, w - 1 - x, h - 1 - y) + acc, colour = 0, None + for c, t in layers: + if d < acc + t: + colour = c + break + acc += t + if colour is None: + px += bytes(((x * 37 + y * 91) % 256, + (x * 17 + y * 53) % 256, + (x * 71 + y * 29) % 256)) + else: + px += bytes(colour) + return bytes(px) + + +class TestRingGeometry(unittest.TestCase): + def test_ring_zero_is_the_outer_edge(self): + px = frame(10, 8) + ring = mod.ring_pixels(px, 10, 8, 0) + # perimeter of a 10x8 rectangle + self.assertEqual(len(ring), 2 * 10 + 2 * (8 - 2)) + + def test_ring_one_is_the_next_rectangle_in(self): + px = frame(10, 8) + ring = mod.ring_pixels(px, 10, 8, 1) + self.assertEqual(len(ring), 2 * 8 + 2 * (6 - 2)) + + def test_ring_beyond_the_centre_is_empty(self): + self.assertEqual(mod.ring_pixels(frame(10, 8), 10, 8, 4), []) + + def test_negative_depth_is_empty(self): + self.assertEqual(mod.ring_pixels(frame(10, 8), 10, 8, -1), []) + + +class TestVarianceAndColour(unittest.TestCase): + def test_flat_pixels_have_zero_variance(self): + self.assertEqual(mod.max_channel_variance([(10, 20, 30)] * 50), 0.0) + + def test_varied_pixels_have_positive_variance(self): + px = [(i, 255 - i, i // 2) for i in range(0, 250, 10)] + self.assertGreater(mod.max_channel_variance(px), 100.0) + + def test_single_pixel_has_no_variance(self): + self.assertEqual(mod.max_channel_variance([(1, 2, 3)]), 0.0) + + def test_mean_colour(self): + self.assertEqual(mod.mean_color([(0, 0, 0), (10, 20, 30)]), + (5.0, 10.0, 15.0)) + + def test_colour_distance(self): + self.assertAlmostEqual( + mod.color_distance((0, 0, 0), (0, 3, 4)), 5.0) + + +class TestDetectBorderDepth(unittest.TestCase): + W, H = 96, 54 + + def test_clean_frame_has_no_border(self): + self.assertEqual( + mod.detect_border_depth(frame(self.W, self.H), self.W, self.H), 0) + + def test_flat_black_border_is_measured(self): + px = frame(self.W, self.H, border=(0, 0, 0), depth=4) + self.assertEqual( + mod.detect_border_depth(px, self.W, self.H), 4) + + def test_multi_colour_decorative_frame_counts_every_layer(self): + # The real defect: black outer border then an orange ring. + px = layered_frame(self.W, self.H, + [((0, 0, 0), 3), ((255, 140, 0), 2)]) + self.assertEqual(mod.detect_border_depth(px, self.W, self.H), 5) + + def test_border_detection_is_bounded(self): + # An entirely flat frame must not report a border the size of itself. + px = frame(self.W, self.H, border=(5, 5, 5), depth=self.H) + depth = mod.detect_border_depth(px, self.W, self.H) + self.assertLessEqual(depth, int(min(self.W, self.H) * 0.25)) + + +class TestBorderIsDistinct(unittest.TestCase): + W, H = 96, 54 + + def test_border_unlike_the_picture_is_distinct(self): + px = frame(self.W, self.H, border=(0, 0, 0), depth=4) + self.assertTrue(mod.border_is_distinct(px, self.W, self.H, 4)) + + def test_zero_depth_is_never_distinct(self): + px = frame(self.W, self.H) + self.assertFalse(mod.border_is_distinct(px, self.W, self.H, 0)) + + def test_flat_scene_matching_its_edges_is_not_a_border(self): + # The false positive that matters: a genuinely dark, flat shot. + px = bytes([20, 20, 20] * (self.W * self.H)) + depth = mod.detect_border_depth(px, self.W, self.H) + self.assertFalse(mod.border_is_distinct(px, self.W, self.H, depth)) + + +class TestRectMath(unittest.TestCase): + def test_active_rect_insets_on_every_edge(self): + self.assertEqual(mod.active_rect(100, 60, 5), (5, 5, 90, 50)) + + def test_scale_rect_maps_thumbnail_to_full_frame(self): + rect = mod.scale_rect((6, 3, 84, 48), (96, 54), (3840, 2160)) + self.assertEqual(rect, (240, 120, 3360, 1920)) + + def test_scaled_dimensions_are_even(self): + for v in mod.scale_rect((5, 3, 85, 47), (96, 54), (1920, 1080)): + self.assertEqual(v % 2, 0) + + def test_aspect_of(self): + self.assertAlmostEqual(mod.aspect_of((0, 0, 1920, 1080)), 16 / 9) + + def test_aspect_of_zero_height_is_zero(self): + self.assertEqual(mod.aspect_of((0, 0, 100, 0)), 0.0) + + +class TestQcSampleTimes(unittest.TestCase): + def test_samples_span_the_interior(self): + times = mod.qc_sample_times(100.0, 5) + self.assertEqual(len(times), 5) + self.assertGreater(times[0], 0.0) + self.assertLess(times[-1], 100.0) + + def test_samples_are_ordered(self): + times = mod.qc_sample_times(600.0, 7) + self.assertEqual(times, sorted(times)) + + def test_more_than_two_samples_by_default(self): + # The original check looked at exactly two frames and missed a defect + # that started mid-take. + self.assertGreater(len(mod.qc_sample_times(600.0)), 2) + + def test_zero_duration_is_a_single_sample(self): + self.assertEqual(mod.qc_sample_times(0.0), [0.0]) + + +@unittest.skipUnless(FFMPEG, "ffmpeg/ffprobe not installed") +class TestQcHaltsEndToEnd(unittest.TestCase): + """Issue G, reproduced: a take with a baked-in border must STOP the + stage, before transcription or any render is built on the bad canvas.""" + + @classmethod + def setUpClass(cls): + cls.tmp = tempfile.TemporaryDirectory() + d = Path(cls.tmp.name) + cls.clean = d / "clean.mp4" + cls.bordered = d / "bordered.mp4" + r = subprocess.run( + ["ffmpeg", "-y", "-f", "lavfi", "-t", "2", "-i", + "testsrc2=size=320x180:rate=30", "-c:v", "libx264", + "-preset", "ultrafast", "-crf", "28", "-pix_fmt", "yuv420p", + str(cls.clean)], capture_output=True, text=True) + assert r.returncode == 0, r.stderr + # The Ecamm-style defect: content shrunk inside a flat black frame. + r = subprocess.run( + ["ffmpeg", "-y", "-f", "lavfi", "-t", "2", "-i", + "testsrc2=size=320x180:rate=30", "-vf", + "scale=240:134,pad=320:180:40:23:black", "-c:v", "libx264", + "-preset", "ultrafast", "-crf", "28", "-pix_fmt", "yuv420p", + str(cls.bordered)], capture_output=True, text=True) + assert r.returncode == 0, r.stderr + + @classmethod + def tearDownClass(cls): + cls.tmp.cleanup() + + def test_clean_source_passes(self): + r = run_cli([str(self.clean)]) + self.assertEqual(r.returncode, 0, r.stderr) + self.assertTrue(json.loads(r.stdout)["qc_ok"]) + + def test_bordered_source_halts_with_exit_3(self): + r = run_cli([str(self.bordered)]) + self.assertEqual(r.returncode, 3, r.stdout) + self.assertIn("SOURCE QC FAILED", r.stderr) + + def test_halt_reports_the_inferred_active_rectangle(self): + r = run_cli([str(self.bordered)]) + self.assertIn("inferred active content: crop=", r.stderr) + qc = json.loads(r.stdout)["files"][0]["qc"] + self.assertIsNotNone(qc["active_rect"]) + w, h = qc["active_rect"][2], qc["active_rect"][3] + self.assertLess(w, 320) + self.assertLess(h, 180) + + def test_halt_points_at_the_remedy(self): + r = run_cli([str(self.bordered)]) + self.assertIn("normalize_source.py", r.stderr) + + def test_allow_qc_defects_downgrades_the_halt(self): + r = run_cli([str(self.bordered), "--allow-qc-defects"]) + self.assertEqual(r.returncode, 0) + self.assertFalse(json.loads(r.stdout)["qc_ok"]) + + def test_no_qc_skips_the_pass(self): + r = run_cli([str(self.bordered), "--no-qc"]) + self.assertEqual(r.returncode, 0) + self.assertNotIn("qc", json.loads(r.stdout)["files"][0]) + + if __name__ == "__main__": unittest.main() diff --git a/skills/mc-cut/scripts/tests/test-render_preview.py b/skills/mc-cut/scripts/tests/test-render_preview.py index fb14c03..a5157c8 100644 --- a/skills/mc-cut/scripts/tests/test-render_preview.py +++ b/skills/mc-cut/scripts/tests/test-render_preview.py @@ -314,5 +314,567 @@ def test_preview_renders_mixed_dimensions_and_audioless(self): self.assertEqual(core.probe_dims(out), (expected_w, 108)) +def ov(oid, start, dur, path="g/x.mov", image=False): + return {"id": oid, "start": start, "dur": dur, "path": path, + "image": image} + + +class TestPlanOverlayLanes(unittest.TestCase): + """Lane count must equal max concurrent overlays: that number becomes the + depth of the final overlay stack, which is the whole optimization. The + real project went from a 56-deep stack to 2 lanes.""" + + def test_no_overlays_is_no_lanes(self): + self.assertEqual(core.plan_overlay_lanes([]), []) + + def test_sequential_overlays_share_one_lane(self): + lanes = core.plan_overlay_lanes( + [ov("a", 0.0, 1.0), ov("b", 1.0, 1.0), ov("c", 5.0, 1.0)]) + self.assertEqual(len(lanes), 1) + self.assertEqual([o["id"] for o in lanes[0]], ["a", "b", "c"]) + + def test_overlapping_overlays_split_into_lanes(self): + lanes = core.plan_overlay_lanes( + [ov("a", 0.0, 3.0), ov("b", 1.0, 3.0)]) + self.assertEqual(len(lanes), 2) + + def test_lane_count_equals_max_concurrency(self): + # Three at once in the middle, one at a time either side. + lanes = core.plan_overlay_lanes([ + ov("a", 0.0, 1.0), + ov("b", 5.0, 4.0), ov("c", 6.0, 4.0), ov("d", 7.0, 4.0), + ov("e", 20.0, 1.0)]) + self.assertEqual(len(lanes), 3) + + def test_fifty_six_sequential_overlays_still_collapse(self): + # The real shape: many overlays, almost none concurrent. + many = [ov(f"b{i:02d}", i * 10.0, 4.0) for i in range(56)] + self.assertEqual(len(core.plan_overlay_lanes(many)), 1) + + def test_every_overlay_lands_in_exactly_one_lane(self): + overlays = [ov(f"b{i}", i * 1.5, 2.0) for i in range(12)] + lanes = core.plan_overlay_lanes(overlays) + placed = [o["id"] for lane in lanes for o in lane] + self.assertEqual(sorted(placed), sorted(o["id"] for o in overlays)) + self.assertEqual(len(placed), len(set(placed))) + + def test_within_a_lane_overlays_never_overlap(self): + overlays = [ov(f"b{i}", i * 1.5, 2.0) for i in range(12)] + for lane in core.plan_overlay_lanes(overlays): + for a, b in zip(lane, lane[1:]): + self.assertLessEqual(a["start"] + a["dur"], b["start"] + 1e-6) + + def test_touching_overlays_can_share_a_lane(self): + lanes = core.plan_overlay_lanes( + [ov("a", 0.0, 2.0), ov("b", 2.0, 2.0)]) + self.assertEqual(len(lanes), 1) + + +class TestLaneCommands(unittest.TestCase): + def test_lane_filter_alternates_gaps_and_overlays(self): + inputs, graph = core.build_lane_filter( + [ov("a", 1.0, 2.0), ov("b", 5.0, 1.0)], 10.0, (320, 180), 30) + # gap, a, gap, b, trailing gap = 5 concat inputs + self.assertIn("concat=n=5", graph) + self.assertEqual(inputs.count("-i"), 5) + + def test_gaps_are_forced_fully_transparent(self): + _, graph = core.build_lane_filter( + [ov("a", 1.0, 2.0)], 5.0, (320, 180), 30) + # A colour source carries no usable alpha; aa=0 is load-bearing. + self.assertIn("colorchannelmixer=aa=0", graph) + + def test_overlay_starting_at_zero_has_no_leading_gap(self): + _, graph = core.build_lane_filter( + [ov("a", 0.0, 2.0)], 4.0, (320, 180), 30) + self.assertIn("concat=n=2", graph) # overlay + trailing gap only + + def test_lane_filling_the_timeline_has_no_gaps(self): + _, graph = core.build_lane_filter( + [ov("a", 0.0, 4.0)], 4.0, (320, 180), 30) + self.assertIn("concat=n=1", graph) + + def test_png_overlays_are_looped_for_their_duration(self): + inputs, _ = core.build_lane_filter( + [ov("a", 0.0, 2.0, "g/x.png", image=True)], 4.0, (320, 180), 30) + self.assertIn("-loop", inputs) + + def test_empty_lane_yields_no_command(self): + self.assertEqual( + core.build_lane_command([], 10.0, (320, 180), 30, "o.mov"), []) + + def test_lane_codec_default_is_alpha_capable(self): + cmd = core.build_lane_command( + [ov("a", 0.0, 2.0)], 4.0, (320, 180), 30, "o.mov") + self.assertIn("qtrle", cmd) + self.assertIn("argb", cmd) + + def test_lane_codec_is_selectable(self): + cmd = core.build_lane_command( + [ov("a", 0.0, 2.0)], 4.0, (320, 180), 30, "o.mov", + codec="prores") + self.assertIn("prores_ks", cmd) + + def test_composite_depth_is_the_lane_count_not_the_overlay_count(self): + cmd = core.build_lane_composite_command("base.mp4", + ["l1.mov", "l2.mov"], "o.mp4") + graph = cmd[cmd.index("-filter_complex") + 1] + self.assertEqual(graph.count("overlay="), 2) + + def test_composite_carries_the_base_audio(self): + cmd = core.build_lane_composite_command("base.mp4", ["l1.mov"], + "o.mp4") + self.assertIn("0:a?", cmd) + + def test_composite_with_no_lanes_is_a_passthrough_graph(self): + cmd = core.build_lane_composite_command("base.mp4", [], "o.mp4") + graph = cmd[cmd.index("-filter_complex") + 1] + self.assertEqual(graph.count("overlay="), 0) + + +class TestProxies(unittest.TestCase): + def test_proxy_path_is_named_by_stem_and_height(self): + p = core.proxy_path("/p/renders/proxy", "raw/cam.mp4", 720) + self.assertEqual(p.name, "cam-720p.mp4") + + def test_proxy_command_scales_to_the_height(self): + cmd = core.build_proxy_command("in.mp4", "out.mp4", 720) + self.assertIn("scale=-2:720", cmd) + + def test_missing_proxy_is_not_fresh(self): + self.assertFalse(core.proxy_is_fresh("/nope/p.mp4", "/nope/s.mp4")) + + def test_proxy_freshness_tracks_the_source_content(self): + with tempfile.TemporaryDirectory() as tmp: + src = Path(tmp) / "src.mp4" + proxy = Path(tmp) / "src-720p.mp4" + src.write_bytes(b"original") + proxy.write_bytes(b"proxy") + core.write_proxy_sidecar(proxy, src) + self.assertTrue(core.proxy_is_fresh(proxy, src)) + # A re-recorded take with the same filename must invalidate it. + src.write_bytes(b"re-recorded, different content entirely") + self.assertFalse(core.proxy_is_fresh(proxy, src)) + + def test_proxied_edl_swaps_sources_and_keeps_every_timecode(self): + edl = {"source": "raw/cam.mp4", "fade_ms": 30, "segments": [ + {"source": "raw/cam.mp4", "start": 1.0, "end": 2.5}, + {"source": "raw/screen.mp4", "start": 0.0, "end": 3.0}]} + out = core.proxied_edl(edl, {"raw/cam.mp4": "renders/proxy/cam.mp4"}) + self.assertEqual(out["segments"][0]["source"], + "renders/proxy/cam.mp4") + self.assertEqual(out["segments"][1]["source"], "raw/screen.mp4") + # A proxy is the same footage at a smaller size: times are untouched. + for a, b in zip(edl["segments"], out["segments"]): + self.assertEqual((a["start"], a["end"]), (b["start"], b["end"])) + + def test_proxied_edl_does_not_mutate_the_original(self): + edl = {"segments": [{"source": "raw/cam.mp4", "start": 0, "end": 1}]} + core.proxied_edl(edl, {"raw/cam.mp4": "proxy/cam.mp4"}) + self.assertEqual(edl["segments"][0]["source"], "raw/cam.mp4") + + +@unittest.skipUnless(FFMPEG, "ffmpeg/ffprobe not installed") +class TestLanePreviewEndToEnd(unittest.TestCase): + """The composited preview through the lane path: overlapping overlays + must still land at the right times on the edited timeline.""" + + @classmethod + def setUpClass(cls): + cls.tmp = tempfile.TemporaryDirectory() + proj = Path(cls.tmp.name) + for d in ("raw", "cut", "graphics", "beats", "renders"): + (proj / d).mkdir() + r = subprocess.run( + ["ffmpeg", "-y", "-f", "lavfi", "-t", "12", "-i", + "testsrc2=size=320x180:rate=30", "-f", "lavfi", "-t", "12", + "-i", "sine=frequency=440:sample_rate=48000", "-shortest", + "-c:v", "libx264", "-preset", "ultrafast", "-crf", "30", + "-pix_fmt", "yuv420p", "-c:a", "aac", + str(proj / "raw" / "cam.mp4")], capture_output=True, text=True) + assert r.returncode == 0, r.stderr + for bid in ("b01", "b02", "b03", "b04"): + r = subprocess.run( + ["ffmpeg", "-y", "-f", "lavfi", "-t", "2", "-i", + "color=c=red:s=320x180:r=30,format=rgba", "-c:v", "qtrle", + "-pix_fmt", "argb", str(proj / "graphics" / f"{bid}.mov")], + capture_output=True, text=True) + assert r.returncode == 0, r.stderr + # b01/b02 overlap, b03/b04 overlap: 2 lanes, with a clean gap at 4.5. + (proj / "beats" / "beats.md").write_text( + "| id | start | dur | end | type |\n|---|---|---|---|---|\n" + "| b01 | 0.5 | 2.0 | 2.5 | overlay |\n" + "| b02 | 2.0 | 2.0 | 4.0 | overlay |\n" + "| b03 | 5.0 | 2.0 | 7.0 | overlay |\n" + "| b04 | 5.5 | 2.0 | 7.5 | overlay |\n") + (proj / "cut" / "edl.json").write_text(json.dumps( + {"source": "raw/cam.mp4", "fade_ms": 30, "pad_ms": 60, + "segments": [ + {"source": "raw/cam.mp4", "start": 0.0, "end": 5.0}, + {"source": "raw/cam.mp4", "start": 6.0, "end": 11.0}]})) + cls.proj = proj + cls.out = proj / "renders" / "preview.mp4" + r = subprocess.run( + [sys.executable, str(SCRIPT), str(proj / "cut" / "edl.json"), + "-o", str(cls.out), "--height", "180", + "--beats", str(proj / "beats" / "beats.md"), + "--graphics-dir", str(proj / "graphics")], + capture_output=True, text=True) + assert r.returncode == 0, r.stderr + cls.summary = json.loads(r.stdout) + + @classmethod + def tearDownClass(cls): + cls.tmp.cleanup() + + def _centre_pixel(self, t): + r = subprocess.run( + ["ffmpeg", "-v", "error", "-ss", str(t), "-i", str(self.out), + "-frames:v", "1", "-vf", + "crop=40:40:(iw-40)/2:(ih-40)/2,scale=1:1", + "-f", "rawvideo", "-pix_fmt", "rgb24", "-"], + capture_output=True) + return tuple(r.stdout[:3]) + + def _is_red(self, rgb): + return rgb[0] > 150 and rgb[1] < 90 and rgb[2] < 90 + + def test_four_overlays_pack_into_two_lanes(self): + self.assertEqual(self.summary["overlays"], 4) + self.assertEqual(self.summary["overlay_lanes"], 2) + + def test_output_validates(self): + self.assertTrue(self.summary["validated"]) + ok, problems = core.validate_render(self.out) + self.assertTrue(ok, problems) + + def test_overlay_is_on_screen_inside_its_window(self): + self.assertTrue(self._is_red(self._centre_pixel(1.0))) + self.assertTrue(self._is_red(self._centre_pixel(3.0))) + self.assertTrue(self._is_red(self._centre_pixel(6.0))) + + def test_overlay_is_absent_in_the_gap_between_beats(self): + self.assertFalse(self._is_red(self._centre_pixel(4.5))) + + def test_overlay_is_absent_after_the_last_beat(self): + self.assertFalse(self._is_red(self._centre_pixel(8.0))) + + def test_duration_matches_the_edl(self): + self.assertAlmostEqual(self.summary["actual_duration_seconds"], 10.0, + delta=0.2) + + def test_a_proxy_was_built_and_is_reused(self): + self.assertEqual(self.summary["proxied_sources"], 1) + self.assertTrue((self.proj / "renders" / "proxy" / + "cam-180p.mp4").is_file()) + + def test_audio_survives_the_lane_composite(self): + self.assertTrue(core.probe_has_audio(self.out)) + + +class TestRenderKey(unittest.TestCase): + """Content identity: two renders share a key exactly when they would + produce the same bytes.""" + + EDL = {"source": "raw/a.mp4", "fade_ms": 30, + "segments": [{"source": "raw/a.mp4", "start": 0.0, "end": 1.0}]} + + def key(self, edl=None, params=None, overlays=None): + return core.render_key(edl or self.EDL, params or {"height": 720}, + overlays or {}, ffmpeg="8.1.2") + + def test_identical_inputs_give_identical_keys(self): + self.assertEqual(self.key(), self.key()) + + def test_key_ignores_dict_ordering(self): + a = core.render_key(self.EDL, {"height": 720, "mode": "preview"}, + {"b": "1", "a": "2"}, ffmpeg="8.1.2") + b = core.render_key(self.EDL, {"mode": "preview", "height": 720}, + {"a": "2", "b": "1"}, ffmpeg="8.1.2") + self.assertEqual(a, b) + + def test_changed_edl_changes_the_key(self): + other = json.loads(json.dumps(self.EDL)) + other["segments"][0]["end"] = 2.0 + self.assertNotEqual(self.key(), self.key(edl=other)) + + def test_changed_params_change_the_key(self): + self.assertNotEqual(self.key(), self.key(params={"height": 540})) + + def test_rerendered_overlay_changes_the_key(self): + self.assertNotEqual(self.key(overlays={"b01": "sha256:aaa"}), + self.key(overlays={"b01": "sha256:bbb"})) + + def test_ffmpeg_bump_changes_the_key(self): + a = core.render_key(self.EDL, {}, {}, ffmpeg="8.1.2") + b = core.render_key(self.EDL, {}, {}, ffmpeg="9.0.0") + self.assertNotEqual(a, b) + + +class TestTempRenderPath(unittest.TestCase): + def test_temp_path_is_beside_the_output(self): + tmp = core.temp_render_path(Path("/p/renders/preview.mp4"), "abc123") + self.assertEqual(tmp.parent, Path("/p/renders")) + self.assertTrue(tmp.name.endswith(".mp4")) + + def test_temp_path_is_not_the_output(self): + out = Path("/p/renders/preview.mp4") + self.assertNotEqual(core.temp_render_path(out, "abc123"), out) + + def test_temp_path_carries_the_pid(self): + # Two concurrent renders of the SAME edl must not share a temp path, + # or the interleaved-write corruption just moves one level down. + import os + tmp = core.temp_render_path(Path("/p/renders/preview.mp4"), "abc123") + self.assertIn(str(os.getpid()), tmp.name) + + +class TestCurrentRenderKey(unittest.TestCase): + def test_unchanged_edl_matches(self): + with tempfile.TemporaryDirectory() as tmp: + p = Path(tmp) / "edl.json" + edl = {"segments": [{"source": "a.mp4", "start": 0, "end": 1}]} + p.write_text(json.dumps(edl)) + key = core.render_key(edl, {}, {}, ffmpeg="8.1.2") + self.assertEqual( + core.current_render_key(p, {}, {}, ffmpeg="8.1.2"), key) + + def test_changed_edl_does_not_match(self): + with tempfile.TemporaryDirectory() as tmp: + p = Path(tmp) / "edl.json" + edl = {"segments": [{"source": "a.mp4", "start": 0, "end": 1}]} + p.write_text(json.dumps(edl)) + key = core.render_key(edl, {}, {}, ffmpeg="8.1.2") + edl["segments"][0]["end"] = 5 + p.write_text(json.dumps(edl)) + self.assertNotEqual( + core.current_render_key(p, {}, {}, ffmpeg="8.1.2"), key) + + def test_unreadable_edl_returns_none_not_a_false_supersede(self): + # An unreadable EDL is a different problem; treating it as a + # supersede would throw away a good render. + self.assertIsNone(core.current_render_key("/nope/edl.json")) + + +@unittest.skipUnless(FFMPEG, "ffmpeg/ffprobe not installed") +class TestValidateRender(unittest.TestCase): + """The check that was missing when a corrupt preview shipped: a truncated + mp4 still reports a plausible container duration, so only a full decode + pass can see it.""" + + @classmethod + def setUpClass(cls): + cls.tmp = tempfile.TemporaryDirectory() + cls.dir = Path(cls.tmp.name) + cls.good = cls.dir / "good.mp4" + r = subprocess.run( + ["ffmpeg", "-y", "-f", "lavfi", "-t", "3", "-i", + "testsrc2=size=160x90:rate=30", "-c:v", "libx264", + "-preset", "ultrafast", "-crf", "30", "-pix_fmt", "yuv420p", + str(cls.good)], capture_output=True, text=True) + assert r.returncode == 0, r.stderr + + @classmethod + def tearDownClass(cls): + cls.tmp.cleanup() + + def test_good_file_validates(self): + ok, problems = core.validate_render(self.good, 3.0) + self.assertTrue(ok, problems) + + def test_missing_file_fails(self): + ok, problems = core.validate_render(self.dir / "nope.mp4") + self.assertFalse(ok) + self.assertIn("not written", problems[0]) + + def test_empty_file_fails(self): + empty = self.dir / "empty.mp4" + empty.write_bytes(b"") + ok, problems = core.validate_render(empty) + self.assertFalse(ok) + self.assertIn("empty", problems[0]) + + def test_truncated_file_fails_decode(self): + # The real corruption shape: a partially-written mp4. + bad = self.dir / "truncated.mp4" + bad.write_bytes(self.good.read_bytes()[: int( + self.good.stat().st_size * 0.4)]) + ok, problems = core.validate_render(bad) + self.assertFalse(ok) + + def test_garbage_bytes_fail(self): + bad = self.dir / "garbage.mp4" + bad.write_bytes(b"\x00\x01\x02" * 5000) + ok, _ = core.validate_render(bad) + self.assertFalse(ok) + + def test_duration_mismatch_is_reported(self): + ok, problems = core.validate_render(self.good, 30.0) + self.assertFalse(ok) + self.assertTrue(any("duration" in p for p in problems)) + + def test_duration_within_tolerance_passes(self): + ok, _ = core.validate_render(self.good, 3.3, tolerance=0.5) + self.assertTrue(ok) + + +class TestPublishRender(unittest.TestCase): + def test_publish_moves_atomically_and_writes_the_key(self): + with tempfile.TemporaryDirectory() as tmp: + src = Path(tmp) / "part.mp4" + dst = Path(tmp) / "preview.mp4" + src.write_bytes(b"payload") + core.publish_render(src, dst, "deadbeef") + self.assertFalse(src.exists()) + self.assertEqual(dst.read_bytes(), b"payload") + self.assertEqual( + (Path(tmp) / "preview.mp4.key").read_text().strip(), + "deadbeef") + + def test_publish_overwrites_an_existing_output(self): + with tempfile.TemporaryDirectory() as tmp: + src = Path(tmp) / "part.mp4" + dst = Path(tmp) / "preview.mp4" + dst.write_bytes(b"old") + src.write_bytes(b"new") + core.publish_render(src, dst, "k") + self.assertEqual(dst.read_bytes(), b"new") + + def test_discard_removes_the_temp(self): + with tempfile.TemporaryDirectory() as tmp: + p = Path(tmp) / "part.mp4" + p.write_bytes(b"x") + core.discard_render(p) + self.assertFalse(p.exists()) + + def test_discard_tolerates_a_missing_file(self): + core.discard_render("/nope/never/part.mp4") # must not raise + + +@unittest.skipUnless(FFMPEG, "ffmpeg/ffprobe not installed") +class TestSupersedeEndToEnd(unittest.TestCase): + """The 2026-07-24 corruption, reproduced as a test: a render whose EDL + changed underneath it must discard its output, not publish over the + newer one.""" + + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.proj = Path(self.tmp.name) + (self.proj / "raw").mkdir() + (self.proj / "cut").mkdir() + (self.proj / "renders").mkdir() + r = subprocess.run( + ["ffmpeg", "-y", "-f", "lavfi", "-t", "5", "-i", + "testsrc2=size=160x90:rate=30", "-f", "lavfi", "-t", "5", + "-i", "sine=frequency=440:sample_rate=48000", "-shortest", + "-c:v", "libx264", "-preset", "ultrafast", "-crf", "30", + "-pix_fmt", "yuv420p", "-c:a", "aac", + str(self.proj / "raw" / "cam.mp4")], + capture_output=True, text=True) + assert r.returncode == 0, r.stderr + self.edl_path = self.proj / "cut" / "edl.json" + self.out = self.proj / "renders" / "preview.mp4" + + def tearDown(self): + self.tmp.cleanup() + + def _write_edl(self, end): + # Atomic, because a render already in flight may be reading this + # file: a plain write truncates first and a concurrent reader gets a + # torn EDL. Any writer of edl.json owes readers the same. + core.write_json_atomic(self.edl_path, { + "source": "raw/cam.mp4", "fade_ms": 30, "pad_ms": 60, + "segments": [{"source": "raw/cam.mp4", "start": 0.0, + "end": end}]}) + + def _render(self): + return subprocess.run( + [sys.executable, str(SCRIPT), str(self.edl_path), "-o", + str(self.out), "--height", "90"], + capture_output=True, text=True) + + def test_normal_render_publishes_and_validates(self): + self._write_edl(2.0) + r = self._render() + self.assertEqual(r.returncode, 0, r.stderr) + self.assertTrue(self.out.is_file()) + summary = json.loads(r.stdout) + self.assertTrue(summary["validated"]) + ok, problems = core.validate_render(self.out) + self.assertTrue(ok, problems) + + def test_published_output_records_its_key(self): + self._write_edl(2.0) + self._render() + keyfile = self.out.with_name(self.out.name + ".key") + self.assertTrue(keyfile.is_file()) + self.assertEqual(keyfile.read_text().strip(), + json.loads(self._render().stdout)["render_key"]) + + def test_no_temp_files_survive_a_successful_render(self): + self._write_edl(2.0) + self._render() + leftovers = [p for p in (self.proj / "renders").iterdir() + if ".part" in p.name] + self.assertEqual(leftovers, []) + + def test_concurrent_renders_never_produce_a_corrupt_output(self): + """The bug report's acceptance test, literally. + + Two renders of DIFFERENT EDLs racing to the same output path. That + is what produced an unplayable preview ("Invalid NAL unit size", + "missing picture in access unit") on the real project. Whichever + wins is not the assertion; the assertion is that the file left + behind always decodes cleanly.""" + self._write_edl(4.5) + first = subprocess.Popen( + [sys.executable, str(SCRIPT), str(self.edl_path), "-o", + str(self.out), "--height", "90"], + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + # Change the EDL underneath the running render, then start a second + # render against the new one, both aimed at the same output. + self._write_edl(1.5) + second = subprocess.Popen( + [sys.executable, str(SCRIPT), str(self.edl_path), "-o", + str(self.out), "--height", "90"], + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + first.communicate() + second.communicate() + + self.assertTrue(self.out.is_file()) + ok, problems = core.validate_render(self.out) + self.assertTrue(ok, f"published a corrupt render: {problems}") + # And nothing half-written is left lying around. + leftovers = [p for p in (self.proj / "renders").iterdir() + if ".part" in p.name] + self.assertEqual(leftovers, []) + + def test_a_stale_render_is_reported_as_superseded(self): + # A render whose EDL changes before it publishes must refuse to + # publish, and say why, rather than clobber the newer output. + self._write_edl(2.0) + edl = json.loads(self.edl_path.read_text()) + params = {"height": 90, "mode": "preview", "beats": False} + stale_key = core.render_key(edl, params, {}, core.ffmpeg_version()) + self._write_edl(3.0) + live_key = core.current_render_key(self.edl_path, params, {}, + core.ffmpeg_version()) + self.assertIsNotNone(live_key) + self.assertNotEqual(stale_key, live_key) + + def test_a_failed_render_leaves_the_previous_output_intact(self): + self._write_edl(2.0) + self.assertEqual(self._render().returncode, 0) + good = self.out.read_bytes() + # An EDL referencing a source that does not exist: the render fails + # and must not touch the published file. + self.edl_path.write_text(json.dumps({ + "source": "raw/gone.mp4", "fade_ms": 30, "pad_ms": 60, + "segments": [{"source": "raw/gone.mp4", "start": 0.0, + "end": 1.0}]})) + self.assertEqual(self._render().returncode, 1) + self.assertEqual(self.out.read_bytes(), good) + + if __name__ == "__main__": unittest.main() diff --git a/skills/mc-cut/scripts/tests/test-snap_spans.py b/skills/mc-cut/scripts/tests/test-snap_spans.py new file mode 100644 index 0000000..54e734b --- /dev/null +++ b/skills/mc-cut/scripts/tests/test-snap_spans.py @@ -0,0 +1,176 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.11" +# /// +"""Tests for snap_spans.py, the step 7a snapping mechanic. + +The load-bearing property is that snapping is DIRECTIONAL. An unconstrained +"nearest silence" can pull a span's end backwards past its own start and +annihilate it, which is a real bug that was caught once already in +cutplan.py's candidate path. This is the same guarantee for approved spans.""" +import importlib.util +import json +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +SCRIPT = Path(__file__).resolve().parent.parent / "snap_spans.py" +spec = importlib.util.spec_from_file_location("snap_spans", SCRIPT) +ss = importlib.util.module_from_spec(spec) +spec.loader.exec_module(ss) + + +def audio_map(silence, duration=10.0, media="m.mp4"): + return {"media": media, "duration": duration, + "silence": [{"start": s, "end": e, "dur": e - s} + for s, e in silence]} + + +def run_cli(spans, amap, extra=None, expect=None): + with tempfile.TemporaryDirectory() as tmp: + s = Path(tmp) / "spans.json" + a = Path(tmp) / "audio-map.json" + o = Path(tmp) / "snapped.json" + s.write_text(json.dumps(spans)) + a.write_text(json.dumps(amap)) + r = subprocess.run( + [sys.executable, str(SCRIPT), str(s), "--audio-map", str(a), + "-o", str(o), *(extra or [])], + capture_output=True, text=True) + if expect is not None: + assert r.returncode == expect, f"rc={r.returncode} {r.stderr}" + out = json.loads(o.read_text()) if o.is_file() else None + return out, r + + +class TestDirectional(unittest.TestCase): + + def test_start_moves_earlier_and_end_moves_later(self): + silence = [(0.0, 0.5), (2.0, 3.0)] + out = ss.snap_span({"start": 1.0, "end": 1.2}, silence, 1.0) + self.assertTrue(out["snapped"]) + self.assertAlmostEqual(out["start"], 0.5) + self.assertAlmostEqual(out["end"], 2.0) + + def test_an_end_is_never_pulled_back_past_its_own_start(self): + # The collapse bug. Unconstrained, end 1.2 would snap BACK to 0.5 + # (0.7s away) rather than forward to 2.0 (0.8s away), landing the span + # on 0.5-0.5 and deleting it. + silence = [(0.0, 0.5), (2.0, 3.0)] + out = ss.snap_span({"start": 1.0, "end": 1.2}, silence, 1.0) + self.assertGreater(out["end"], out["start"]) + self.assertAlmostEqual(out["end"], 2.0) + + def test_a_span_only_ever_widens(self): + silence = [(0.0, 0.5), (2.0, 3.0)] + out = ss.snap_span({"start": 1.0, "end": 1.2}, silence, 1.0) + self.assertLessEqual(out["start"], 1.0) + self.assertGreaterEqual(out["end"], 1.2) + + def test_edges_already_in_silence_do_not_move(self): + silence = [(1.0, 2.0), (5.0, 6.0)] + out = ss.snap_span({"start": 1.5, "end": 5.5}, silence, 1.0) + self.assertAlmostEqual(out["start"], 1.5) + self.assertAlmostEqual(out["end"], 5.5) + self.assertEqual(out["shift"], {"start": 0.0, "end": 0.0}) + + +class TestUnsnapped(unittest.TestCase): + + def test_an_edge_out_of_reach_leaves_the_span_unsnapped(self): + silence = [(0.0, 0.1)] + out = ss.snap_span({"start": 5.0, "end": 6.0}, silence, 0.25) + self.assertFalse(out["snapped"]) + + def test_unsnapped_spans_keep_their_original_times(self): + silence = [(0.0, 0.1)] + out = ss.snap_span({"start": 5.0, "end": 6.0}, silence, 0.25) + self.assertAlmostEqual(out["start"], 5.0) + self.assertAlmostEqual(out["end"], 6.0) + + def test_the_unreachable_edges_are_named(self): + silence = [(4.9, 5.05)] + out = ss.snap_span({"start": 5.0, "end": 9.0}, silence, 0.25) + self.assertEqual(out["unsnapped_edges"], ["end"]) + + def test_report_counts_snapped_and_unsnapped(self): + silence = [(0.0, 0.5), (2.0, 3.0)] + spans = ss.snap_all([{"start": 1.0, "end": 1.2}, + {"start": 7.0, "end": 8.0}], silence, 1.0) + report = ss.build_report(spans) + self.assertFalse(report["ok"]) + self.assertEqual(report["spans"], 2) + self.assertEqual(report["snapped"], 1) + self.assertEqual(len(report["unsnapped"]), 1) + + +class TestPassthrough(unittest.TestCase): + + def test_extra_keys_survive_the_round_trip(self): + silence = [(0.0, 0.5), (2.0, 3.0)] + span = {"start": 1.0, "end": 1.2, "id": "c7", "quote": "the thing", + "reason": "editorial: cut the aside"} + out = ss.snap_span(span, silence, 1.0) + self.assertEqual(out["id"], "c7") + self.assertEqual(out["quote"], "the thing") + self.assertEqual(out["reason"], "editorial: cut the aside") + + def test_the_input_span_is_not_mutated(self): + silence = [(0.0, 0.5), (2.0, 3.0)] + span = {"start": 1.0, "end": 1.2} + ss.snap_span(span, silence, 1.0) + self.assertEqual(span, {"start": 1.0, "end": 1.2}) + + def test_duration_is_recomputed_from_the_snapped_edges(self): + silence = [(0.0, 0.5), (2.0, 3.0)] + out = ss.snap_span({"start": 1.0, "end": 1.2}, silence, 1.0) + self.assertAlmostEqual(out["dur"], 1.5) + + +class TestParsing(unittest.TestCase): + + def test_a_bare_list_is_accepted(self): + self.assertEqual(ss.parse_spans([{"start": 0, "end": 1}]), + [{"start": 0, "end": 1}]) + + def test_an_object_with_a_spans_key_is_accepted(self): + self.assertEqual(ss.parse_spans({"spans": [{"start": 0, "end": 1}]}), + [{"start": 0, "end": 1}]) + + def test_anything_else_is_rejected(self): + self.assertIsNone(ss.parse_spans({"nope": 1})) + + +class TestCLI(unittest.TestCase): + + def test_all_snapped_exits_zero(self): + amap = audio_map([(0.0, 0.5), (2.0, 3.0)]) + out, r = run_cli([{"start": 1.0, "end": 1.2}], amap, + ["--snap-ms", "1000"], expect=0) + self.assertEqual(len(out), 1) + self.assertTrue(out[0]["snapped"]) + + def test_an_unsnapped_span_exits_one_and_says_to_check_by_ear(self): + amap = audio_map([(0.0, 0.1)]) + out, r = run_cli([{"start": 5.0, "end": 6.0}], amap, expect=1) + self.assertFalse(out[0]["snapped"]) + self.assertIn("by ear", r.stderr) + + def test_a_span_missing_end_is_a_usage_error(self): + amap = audio_map([(0.0, 0.5)]) + run_cli([{"start": 1.0}], amap, expect=2) + + def test_an_audio_map_without_silence_is_a_usage_error(self): + run_cli([{"start": 1.0, "end": 1.2}], {"media": "m.mp4"}, expect=2) + + def test_help_works(self): + r = subprocess.run([sys.executable, str(SCRIPT), "--help"], + capture_output=True, text=True) + self.assertEqual(r.returncode, 0) + self.assertIn("--snap-ms", r.stdout) + + +if __name__ == "__main__": + unittest.main() diff --git a/skills/mc-cut/scripts/tests/test-transcribe.py b/skills/mc-cut/scripts/tests/test-transcribe.py index e0f77f3..d9577e2 100644 --- a/skills/mc-cut/scripts/tests/test-transcribe.py +++ b/skills/mc-cut/scripts/tests/test-transcribe.py @@ -501,6 +501,156 @@ def test_help_exits_zero(self): self.assertEqual(r.returncode, 0) self.assertIn("--provider", r.stdout) + def test_overlap_not_smaller_than_window_is_a_usage_error(self): + r = run(["missing.mp4", "-o", "out.json", "--provider", "onnx-asr", + "--window", "20", "--overlap", "20"]) + self.assertEqual(r.returncode, 2) + self.assertIn("--overlap must be", r.stderr) + + def test_oversized_window_warns_about_silent_drops(self): + # Exits 2 on the missing media, but the warning must fire first. + r = run(["missing.mp4", "-o", "out.json", "--provider", "onnx-asr", + "--window", "120"]) + self.assertIn("silently drops speech", r.stderr) + self.assertIn("verify_transcript.py", r.stderr) + + +class TestWindowDefaults(unittest.TestCase): + """The validated windowing constants. These are not free parameters: + 20 s isolated windows are what makes parakeet lossless (module docstring's + chunking note), so a change here is a behavior regression, not a tweak.""" + + def test_window_is_twenty_seconds(self): + self.assertEqual(transcribe.CHUNK_WINDOW_S, 20.0) + + def test_overlap_is_three_seconds(self): + self.assertEqual(transcribe.CHUNK_OVERLAP_S, 3.0) + + def test_overlap_is_smaller_than_window(self): + self.assertLess(transcribe.CHUNK_OVERLAP_S, transcribe.CHUNK_WINDOW_S) + + +class TestMlxTokensToDicts(unittest.TestCase): + """The mlx lane's per-window token adapter. AlignedToken text already + carries the leading-space word boundary, so this is a field copy plus the + window clamp and the bare-boundary carry.""" + + class Tok: + def __init__(self, text, start, end, confidence=0.9): + self.text = text + self.start = start + self.end = end + self.confidence = confidence + + def test_field_copy_preserves_leading_space_boundary(self): + toks = [self.Tok(" Al", 0.0, 0.2), self.Tok("right", 0.2, 0.5)] + out = transcribe.mlx_tokens_to_dicts(toks) + self.assertEqual([t["text"] for t in out], [" Al", "right"]) + self.assertEqual(out[0]["start"], 0.0) + self.assertEqual(out[1]["end"], 0.5) + self.assertAlmostEqual(out[0]["confidence"], 0.9) + + def test_output_groups_into_words_through_the_shared_helper(self): + toks = [self.Tok(" Al", 0.0, 0.2), self.Tok("right", 0.2, 0.5), + self.Tok(",", 0.5, 0.52), self.Tok(" so", 0.9, 1.1)] + words = transcribe.group_subwords(transcribe.mlx_tokens_to_dicts(toks)) + self.assertEqual([w[0] for w in words], ["Alright,", "so"]) + + def test_clamp_caps_end_at_the_window_length(self): + out = transcribe.mlx_tokens_to_dicts([self.Tok(" word", 19.5, 21.0)], + clamp=20.0) + self.assertEqual(out[0]["end"], 20.0) + + def test_end_never_precedes_start(self): + out = transcribe.mlx_tokens_to_dicts([self.Tok(" word", 19.9, 21.0)], + clamp=19.0) + self.assertEqual(out[0]["end"], out[0]["start"]) + + def test_bare_boundary_token_carries_onto_the_next(self): + toks = [self.Tok(" ", 0.0, 0.01), self.Tok("hello", 0.01, 0.4)] + out = transcribe.mlx_tokens_to_dicts(toks) + self.assertEqual([t["text"] for t in out], [" hello"]) + + def test_empty_token_is_dropped_without_emitting_a_word(self): + toks = [self.Tok("", 0.0, 0.0), self.Tok(" hi", 0.1, 0.3)] + out = transcribe.mlx_tokens_to_dicts(toks) + self.assertEqual([t["text"] for t in out], [" hi"]) + + def test_accepts_dict_tokens_too(self): + toks = [{"text": " hi", "start": 0.0, "end": 0.3, "confidence": 0.5}] + out = transcribe.mlx_tokens_to_dicts(toks) + self.assertEqual(out[0]["text"], " hi") + self.assertAlmostEqual(out[0]["confidence"], 0.5) + + +class TestTranscribeWindowed(unittest.TestCase): + """The lane-agnostic driver. Both providers go through it, so a fake + recognizer proves the contract without either model dependency.""" + + def _driver(self, duration, window=20.0, overlap=3.0, tokens_for=None): + calls = [] + + def fake_recognize(wav, length): + calls.append((str(wav), length)) + return tokens_for(len(calls) - 1, length) if tokens_for else [] + + with mock.patch.object(transcribe, "probe_duration", + return_value=duration), \ + mock.patch.object(transcribe, "extract_chunk"), \ + contextlib.redirect_stderr(io.StringIO()): + text, merged, dur = transcribe.transcribe_windowed( + "fake.mp4", fake_recognize, window=window, overlap=overlap) + return text, merged, dur, calls + + def test_long_media_is_split_into_many_short_windows(self): + # 20.5 minutes, the take that OOM'd and then dropped speech. + _, _, _, calls = self._driver(1230.0) + self.assertGreater(len(calls), 60) + self.assertTrue(all(length <= 20.0 + 1e-6 for _, length in calls)) + + def test_no_window_ever_exceeds_the_configured_size(self): + _, _, _, calls = self._driver(97.3, window=20.0, overlap=3.0) + for _, length in calls: + self.assertLessEqual(length, 20.0 + 1e-6) + + def test_short_media_is_a_single_window(self): + _, _, _, calls = self._driver(12.0) + self.assertEqual(len(calls), 1) + self.assertAlmostEqual(calls[0][1], 12.0) + + def test_duration_is_reported_from_the_probe(self): + _, _, dur, _ = self._driver(1230.0) + self.assertEqual(dur, 1230.0) + + def test_tokens_are_offset_into_absolute_time_and_merged(self): + # One word per window, mid-window so it lands on its own side of both + # seam cuts (a word near a window edge is the merge's job, covered by + # TestMergeChunkTokens; this test is about the offsetting). + def tokens_for(i, length): + mid = length / 2 + return [{"text": f" w{i}", "start": mid, "end": mid + 0.4, + "confidence": 1.0}] + + text, merged, _, calls = self._driver(45.0, tokens_for=tokens_for) + self.assertEqual(len(calls), 3) # windows at 0, 17, 34 + starts = [t["start"] for t in merged] + self.assertEqual(starts, sorted(starts)) + self.assertAlmostEqual(starts[0], 10.0) # window 0 + 20/2 + self.assertAlmostEqual(starts[1], 27.0) # window 17 + 20/2 + self.assertAlmostEqual(starts[2], 39.5) # window 34 + 11/2 + self.assertEqual(text, "w0 w1 w2") + + def test_every_window_word_survives_the_merge(self): + def tokens_for(i, length): + mid = length / 2 + return [{"text": f" w{i}", "start": mid, "end": mid + 0.4, + "confidence": 1.0}] + + # The regression this whole workstream exists for: nothing is silently + # lost between the recognizer and the merged output. + _, merged, _, calls = self._driver(1230.0, tokens_for=tokens_for) + self.assertEqual(len(merged), len(calls)) + if __name__ == "__main__": unittest.main() diff --git a/skills/mc-cut/scripts/tests/test-verify_edl.py b/skills/mc-cut/scripts/tests/test-verify_edl.py new file mode 100644 index 0000000..d2e541d --- /dev/null +++ b/skills/mc-cut/scripts/tests/test-verify_edl.py @@ -0,0 +1,273 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.11" +# /// +"""Tests for verify_edl.py, the EDL gate. + +The headline case is the one that would break a naive implementation: a cut +correctly placed in verified silence that ALSO sits inside a pause-absorbed +word span must PASS. Failing it would reject correct cuts on exactly the +footage the two-source rule exists for.""" +import importlib.util +import json +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +SCRIPT = Path(__file__).resolve().parent.parent / "verify_edl.py" +spec = importlib.util.spec_from_file_location("verify_edl", SCRIPT) +ve = importlib.util.module_from_spec(spec) +spec.loader.exec_module(ve) + +DURATION = 20.0 +# Silence islands. Everything else is speech. +SILENCE = [(0.0, 1.0), (5.0, 6.0), (10.0, 11.0), (19.0, 20.0)] + + +def words_from(spans): + return [{"word": t, "start": s, "end": e} for t, s, e in spans] + + +def audio_map(duration=DURATION, silence=None, media="m.mp4"): + pairs = list(SILENCE if silence is None else silence) + speech, cursor = [], 0.0 + for s, e in pairs: + if s > cursor: + speech.append((cursor, s)) + cursor = max(cursor, e) + if duration > cursor: + speech.append((cursor, duration)) + return { + "media": media, "duration": duration, + "silence": [{"start": s, "end": e, "dur": e - s} for s, e in pairs], + "speech": [{"start": s, "end": e, "dur": e - s} for s, e in speech], + } + + +def seg(start, end, source="m.mp4", quote="words", reason="keep"): + return {"source": source, "start": start, "end": end, + "quote": quote, "reason": reason} + + +def edl(segments, duration=DURATION, source="m.mp4"): + return {"source": source, "source_duration": duration, "fade_ms": 30, + "pad_ms": 60, "segments": segments} + + +def report_for(segments, words=None, duration=DURATION, tolerance=0.0, + amap=None): + payload = edl(segments, duration) + maps = {None: ve.audio.to_pairs((amap or audio_map())["silence"])} + transcripts = {None: words or []} + return ve.build_report(payload, maps, transcripts, tolerance) + + +def run_cli(payload, amap, words, extra=None, expect=None): + with tempfile.TemporaryDirectory() as tmp: + e = Path(tmp) / "edl.json" + a = Path(tmp) / "audio-map.json" + w = Path(tmp) / "words.json" + o = Path(tmp) / "edl-check.json" + e.write_text(json.dumps(payload)) + a.write_text(json.dumps(amap)) + w.write_text(json.dumps(words)) + r = subprocess.run( + [sys.executable, str(SCRIPT), str(e), "--audio-map", str(a), + "--words", str(w), "-o", str(o), *(extra or [])], + capture_output=True, text=True) + if expect is not None: + assert r.returncode == expect, f"rc={r.returncode} {r.stderr}" + report = json.loads(o.read_text()) if o.is_file() else None + return report, r + + +class TestPauseAbsorption(unittest.TestCase): + """The reason word spans are context and never a verdict.""" + + def test_boundary_in_silence_passes_even_inside_an_absorbed_word(self): + # "about." swallowed the pause after it: the word's timestamp span + # reaches from 4.5 all the way to 5.8, straight through the silence + # at 5.0-6.0. This is the real 2026-07-24 shape. + words = words_from([("about.", 4.5, 5.8), ("next", 6.1, 6.5)]) + r = report_for([seg(0.0, 5.5), seg(10.5, DURATION)], words) + self.assertTrue(r["ok"], r["violations"]) + + def test_word_overlap_alone_is_never_a_violation(self): + words = words_from([("about.", 4.5, 5.8)]) + r = report_for([seg(0.0, 5.5), seg(10.5, DURATION)], words) + self.assertEqual([v for v in r["violations"] + if "inside_word" in v], []) + + def test_word_overlap_is_reported_as_detail_on_a_failing_boundary(self): + # 7.5 is in speech AND mid-word: the boundary fails on the audio, and + # the word tells the creator how bad the miss is. + words = words_from([("hello", 7.2, 7.9)]) + r = report_for([seg(0.0, 7.5), seg(10.5, DURATION)], words) + boundary = [v for v in r["violations"] if v["kind"] == "boundary"] + self.assertEqual(len(boundary), 1) + self.assertEqual(boundary[0]["inside_word"], "hello") + self.assertIn("inside the word", boundary[0]["reason"]) + + +class TestBoundaries(unittest.TestCase): + + def test_boundaries_resting_in_silence_pass(self): + r = report_for([seg(0.0, 5.5), seg(10.5, DURATION)]) + self.assertTrue(r["ok"], r["violations"]) + + def test_boundary_in_speech_fails(self): + r = report_for([seg(0.0, 5.5), seg(12.0, DURATION)]) + self.assertFalse(r["ok"]) + self.assertEqual(r["violations"][0]["kind"], "boundary") + self.assertEqual(r["violations"][0]["edge"], "start") + + def test_failure_reports_distance_to_the_nearest_silence(self): + r = report_for([seg(0.0, 5.5), seg(12.0, DURATION)]) + v = r["violations"][0] + self.assertAlmostEqual(v["nearest_silence"], 11.0) + self.assertAlmostEqual(v["distance"], 1.0) + + def test_tolerance_can_admit_a_near_miss(self): + r = report_for([seg(0.0, 5.5), seg(11.2, DURATION)], tolerance=0.5) + self.assertTrue(r["ok"], r["violations"]) + + def test_zero_tolerance_is_the_default_and_rejects_the_near_miss(self): + r = report_for([seg(0.0, 5.5), seg(11.2, DURATION)]) + self.assertFalse(r["ok"]) + + def test_a_source_with_no_silence_at_all_is_named_as_such(self): + amap = audio_map(silence=[]) + r = report_for([seg(2.0, 8.0)], amap=amap) + self.assertFalse(r["ok"]) + self.assertIn("no detected silence", r["violations"][0]["reason"]) + + +class TestExemptions(unittest.TestCase): + """Boundaries that are not cuts. Nothing was removed, so nothing can clip.""" + + def test_source_head_and_tail_are_not_cuts(self): + # 0.0 sits in silence here anyway, so use a map with no silence at the + # edges to prove the exemption is doing the work. + amap = audio_map(silence=[(5.0, 6.0)]) + r = report_for([seg(0.0, 5.5), seg(5.5, DURATION)], amap=amap) + self.assertTrue(r["ok"], r["violations"]) + + def test_contiguous_segments_share_a_boundary_that_is_not_a_cut(self): + amap = audio_map(silence=[]) + r = report_for([seg(0.0, 7.0), seg(7.0, DURATION)], amap=amap) + self.assertTrue(r["ok"], r["violations"]) + + def test_a_real_cut_between_two_segments_is_still_checked(self): + amap = audio_map(silence=[]) + r = report_for([seg(0.0, 7.0), seg(9.0, DURATION)], amap=amap) + self.assertFalse(r["ok"]) + + +class TestReordering(unittest.TestCase): + """Step 5 picks best takes and orders segments. A reordered EDL is correct.""" + + def test_segments_out_of_source_order_are_not_a_violation(self): + r = report_for([seg(10.5, DURATION), seg(0.0, 5.5)]) + self.assertTrue(r["ok"], r["violations"]) + + +class TestStructure(unittest.TestCase): + + def test_zero_length_segment_fails(self): + r = report_for([seg(5.5, 5.5)]) + self.assertFalse(r["ok"]) + self.assertEqual(r["violations"][0]["kind"], "structure") + + def test_inverted_segment_fails(self): + r = report_for([seg(10.5, 5.5)]) + self.assertFalse(r["ok"]) + self.assertIn("not before", r["violations"][0]["reason"]) + + def test_segment_past_the_source_duration_fails(self): + r = report_for([seg(0.0, 25.0)]) + self.assertFalse(r["ok"]) + self.assertTrue(any("past the source duration" in v["reason"] + for v in r["violations"])) + + def test_segment_without_a_source_fails(self): + bad = {"start": 0.0, "end": 5.5, "quote": "q", "reason": "r"} + r = report_for([bad]) + self.assertFalse(r["ok"]) + self.assertIn("no source", r["violations"][0]["reason"]) + + +class TestProvenance(unittest.TestCase): + """Every EDL segment records the words it carries and why.""" + + def test_missing_quote_fails(self): + r = report_for([seg(0.0, 5.5, quote=""), seg(10.5, DURATION)]) + self.assertFalse(r["ok"]) + self.assertTrue(any(v["kind"] == "provenance" for v in r["violations"])) + + def test_missing_reason_fails(self): + r = report_for([seg(0.0, 5.5, reason=" "), seg(10.5, DURATION)]) + self.assertFalse(r["ok"]) + + def test_provenance_and_boundary_failures_are_reported_together(self): + r = report_for([seg(0.0, 5.5), seg(12.0, DURATION, quote="")]) + kinds = {v["kind"] for v in r["violations"]} + self.assertEqual(kinds, {"boundary", "provenance"}) + + +class TestMultiSource(unittest.TestCase): + + def test_maps_are_matched_to_each_segment_source(self): + a = audio_map(media="a.mp4", silence=[(5.0, 6.0)]) + b = audio_map(media="b.mp4", silence=[(12.0, 13.0)]) + maps = {"a.mp4": ve.audio.to_pairs(a["silence"]), + "b.mp4": ve.audio.to_pairs(b["silence"])} + payload = edl([seg(0.0, 5.5, source="a.mp4"), + seg(12.5, DURATION, source="b.mp4")]) + r = ve.build_report(payload, maps, {None: []}, 0.0) + self.assertTrue(r["ok"], r["violations"]) + + def test_a_segment_whose_source_has_no_map_is_named(self): + maps = {"a.mp4": ve.audio.to_pairs(audio_map()["silence"])} + payload = edl([seg(0.0, 5.5, source="b.mp4")]) + r = ve.build_report(payload, maps, {None: []}, 0.0) + self.assertFalse(r["ok"]) + self.assertIn("no audio map covers", r["violations"][0]["reason"]) + + +class TestCLI(unittest.TestCase): + + def test_clean_edl_exits_zero(self): + payload = edl([seg(0.0, 5.5), seg(10.5, DURATION)]) + report, r = run_cli(payload, audio_map(), {"words": []}, expect=0) + self.assertTrue(report["ok"]) + + def test_bad_edl_exits_one_and_names_the_segment(self): + payload = edl([seg(0.0, 5.5), seg(12.0, DURATION)]) + report, r = run_cli(payload, audio_map(), {"words": []}, expect=1) + self.assertFalse(report["ok"]) + self.assertIn("EDL VERIFICATION FAILED", r.stderr) + self.assertIn("does not rest in an audio-verified silence", r.stderr) + + def test_an_edl_with_no_segments_is_a_usage_error(self): + run_cli(edl([]), audio_map(), {"words": []}, expect=2) + + def test_an_audio_map_without_silence_is_a_usage_error(self): + payload = edl([seg(0.0, 5.5)]) + run_cli(payload, {"media": "m.mp4", "duration": 20.0}, + {"words": []}, expect=2) + + def test_the_words_file_must_carry_a_words_key(self): + payload = edl([seg(0.0, 5.5)]) + run_cli(payload, audio_map(), {"tokens": []}, expect=2) + + def test_help_works(self): + r = subprocess.run([sys.executable, str(SCRIPT), "--help"], + capture_output=True, text=True) + self.assertEqual(r.returncode, 0) + self.assertIn("--audio-map", r.stdout) + + +if __name__ == "__main__": + unittest.main() diff --git a/skills/mc-cut/scripts/tests/test-verify_transcript.py b/skills/mc-cut/scripts/tests/test-verify_transcript.py new file mode 100644 index 0000000..5f7757e --- /dev/null +++ b/skills/mc-cut/scripts/tests/test-verify_transcript.py @@ -0,0 +1,553 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.11" +# /// +"""Tests for verify_transcript.py, the transcript completeness gate. + +The headline cases are the two from the real failure (2026-07-24): the broken +transcript must FAIL and the good one must PASS. Everything else here defends +the property that makes that work, namely that the check is computed from the +AUDIO side and so does not inherit the pause-absorption bug it exists to +catch.""" +import importlib.util +import json +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +SCRIPT = Path(__file__).resolve().parent.parent / "verify_transcript.py" +spec = importlib.util.spec_from_file_location("verify_transcript", SCRIPT) +vt = importlib.util.module_from_spec(spec) +spec.loader.exec_module(vt) + + +def words_from(spans): + """Word dicts from (text, start, end) triples.""" + return [{"word": t, "start": s, "end": e, "confidence": 1.0, "i": i, + "gap_before": 0.0, "gap_after": 0.0} + for i, (t, s, e) in enumerate(spans)] + + +def audio_map(duration, silence, media="m.mp4"): + pairs = [(s, e) for s, e in silence] + speech = [] + cursor = 0.0 + for s, e in pairs: + if s > cursor: + speech.append((cursor, s)) + cursor = max(cursor, e) + if duration > cursor: + speech.append((cursor, duration)) + return { + "media": media, + "duration": duration, + "noise_db": -30.0, + "min_silence": 0.3, + "silent_seconds": sum(e - s for s, e in pairs), + "speech_seconds": sum(e - s for s, e in speech), + "counts": {"silence": len(pairs), "speech": len(speech)}, + "silence": [{"start": s, "end": e, "dur": e - s} for s, e in pairs], + "speech": [{"start": s, "end": e, "dur": e - s} for s, e in speech], + } + + +def run_cli(transcript, amap, extra=None, expect=None): + with tempfile.TemporaryDirectory() as tmp: + w = Path(tmp) / "words.json" + a = Path(tmp) / "audio-map.json" + o = Path(tmp) / "report.json" + w.write_text(json.dumps(transcript)) + a.write_text(json.dumps(amap)) + r = subprocess.run( + [sys.executable, str(SCRIPT), str(w), "--audio-map", str(a), + "-o", str(o), *(extra or [])], + capture_output=True, text=True) + if expect is not None: + assert r.returncode == expect, f"rc={r.returncode} {r.stderr}" + report = json.loads(o.read_text()) if o.is_file() else None + return report, r + + +class TestUncoveredRegions(unittest.TestCase): + """Audible with no words. The complement of (silence + word spans).""" + + def test_speech_fully_covered_by_words_leaves_nothing(self): + words = words_from([("a", 1.0, 2.0), ("b", 2.0, 3.0)]) + silence = [(0.0, 1.0), (3.0, 5.0)] + self.assertEqual(vt.uncovered_regions(silence, words, 5.0), []) + + def test_audible_span_with_no_words_is_uncovered(self): + words = words_from([("a", 1.0, 2.0)]) + silence = [(0.0, 1.0), (2.0, 3.0)] + # 3.0 to 20.0 is neither silent nor transcribed: dropped speech. + self.assertEqual(vt.uncovered_regions(silence, words, 20.0), + [(3.0, 20.0)]) + + def test_silence_with_no_words_is_not_uncovered(self): + words = words_from([("a", 1.0, 2.0)]) + silence = [(0.0, 1.0), (2.0, 20.0)] + self.assertEqual(vt.uncovered_regions(silence, words, 20.0), []) + + def test_pause_absorbed_word_end_cannot_fake_coverage(self): + # The failure mode this design defends against: parakeet extends a + # word's end across the pause. Even if the last kept word's end + # reaches 2s into an 18s dropped region, 16s stay uncovered. + words = words_from([("about.", 30.0, 34.0)]) + silence = [(0.0, 30.0)] + regions = vt.uncovered_regions(silence, words, 52.0) + self.assertEqual(regions, [(34.0, 52.0)]) + + def test_words_outside_any_speech_still_count_as_coverage(self): + words = words_from([("a", 0.5, 1.5)]) + silence = [(0.0, 5.0)] + self.assertEqual(vt.uncovered_regions(silence, words, 5.0), []) + + +class TestCluster(unittest.TestCase): + def test_near_pieces_merge(self): + self.assertEqual(vt.cluster([(1.0, 2.0), (2.5, 3.0)], 1.0), + [(1.0, 3.0)]) + + def test_far_pieces_stay_apart(self): + self.assertEqual(vt.cluster([(1.0, 2.0), (9.0, 10.0)], 1.0), + [(1.0, 2.0), (9.0, 10.0)]) + + def test_empty_is_empty(self): + self.assertEqual(vt.cluster([], 1.0), []) + + +class TestFindDropped(unittest.TestCase): + def test_long_uncovered_region_is_reported(self): + words = words_from([("a", 1.0, 2.0)]) + silence = [(0.0, 1.0), (2.0, 3.0)] + dropped = vt.find_dropped(silence, words, 20.0) + self.assertEqual(len(dropped), 1) + self.assertEqual(dropped[0]["start"], 3.0) + self.assertEqual(dropped[0]["audible"], 17.0) + + def test_short_uncovered_region_is_below_threshold(self): + words = words_from([("a", 1.0, 2.0)]) + silence = [(0.0, 1.0), (2.5, 20.0)] + # Only 0.5s uncovered (2.0 to 2.5), under the 1.0s default. + self.assertEqual(vt.find_dropped(silence, words, 20.0), []) + + def test_threshold_applies_to_audible_not_the_cluster_span(self): + # Two 1.0s uncovered pieces separated by a 0.5s covered stretch + # cluster into a 2.5s span, but only 2.0s of it is audible, so + # clustering must not push it over a 2.5s threshold. + words = words_from([("a", 0.0, 1.0), ("b", 2.0, 2.5)]) + silence = [] + dropped = vt.find_dropped(silence, words, 3.5, min_drop=2.5, + cluster_gap=1.0) + self.assertEqual(dropped, []) + + def test_reported_region_carries_a_readable_timecode(self): + words = words_from([("a", 1.0, 2.0)]) + silence = [(0.0, 1.0), (2.0, 89.0)] + dropped = vt.find_dropped(silence, words, 120.0) + self.assertEqual(dropped[0]["at"], "1:29.0") + + def test_min_drop_is_configurable(self): + words = words_from([("a", 1.0, 2.0)]) + silence = [(0.0, 1.0), (4.0, 20.0)] # 2.0s uncovered + self.assertEqual(vt.find_dropped(silence, words, 20.0, min_drop=2.5), + []) + self.assertEqual( + len(vt.find_dropped(silence, words, 20.0, min_drop=1.5)), 1) + + +class TestWordRate(unittest.TestCase): + def test_speech_rate_excludes_dead_air(self): + rate = vt.word_rate([0] * 100, duration=120.0, speech_seconds=60.0) + self.assertEqual(rate["wall_wpm"], 50.0) + self.assertEqual(rate["speech_wpm"], 100.0) + + def test_below_floor_fails(self): + rate = vt.word_rate([0] * 50, duration=120.0, speech_seconds=60.0, + wpm=200.0, floor_ratio=0.6) + self.assertFalse(rate["ok"]) + self.assertEqual(rate["floor_wpm"], 120.0) + + def test_above_floor_passes(self): + rate = vt.word_rate([0] * 200, duration=120.0, speech_seconds=60.0, + wpm=200.0, floor_ratio=0.6) + self.assertTrue(rate["ok"]) + + def test_no_configured_wpm_skips_the_check(self): + rate = vt.word_rate([0] * 5, duration=120.0, speech_seconds=60.0) + self.assertTrue(rate["ok"]) + self.assertIn("skipped", rate["note"]) + + def test_zero_duration_does_not_divide_by_zero(self): + rate = vt.word_rate([], duration=0.0, speech_seconds=0.0, wpm=200.0) + self.assertEqual(rate["wall_wpm"], 0.0) + self.assertEqual(rate["speech_wpm"], 0.0) + + +class TestTheRealFailure(unittest.TestCase): + """The 2026-07-24 take: 20.5 minutes, 3 dropped paragraphs. + + Reduced to the essential shape. The good transcript covers all the + speech; the broken one is missing three audible regions.""" + + DURATION = 1230.0 + # Three dropped regions at the reported timecodes: 0:49, 1:27, 9:37. + DROPS = [(49.0, 58.0), (87.0, 105.0), (577.0, 585.0)] + + def _silence(self): + # Dead air spread through the take, none of it overlapping a drop. + sil = [(0.0, 1.0)] + t = 200.0 + while t < 560.0: + sil.append((t, t + 2.0)) + t += 20.0 + t = 600.0 + while t < 1200.0: + sil.append((t, t + 2.0)) + t += 20.0 + sil.append((1225.0, 1230.0)) + return sil + + WORD_S = 0.33 # about 180 wpm over speech time, a plausible read + + def _speech(self): + silence = self._silence() + out = [] + cursor = 0.0 + for s, e in silence: + if s > cursor: + out.append((cursor, s)) + cursor = max(cursor, e) + if self.DURATION > cursor: + out.append((cursor, self.DURATION)) + return out + + def _words(self, include_drops): + """Words tiling every speech interval end to end. + + Contiguous by construction, so the only uncovered audio in the good + case is nothing at all: any region this gate reports is a region the + fixture deliberately dropped, never a tiling artifact.""" + spans = [] + idx = 0 + for s, e in self._speech(): + t = s + while t < e - 1e-9: + end = min(t + self.WORD_S, e) + # Absorb a final sliver into the previous word so speech is + # covered exactly. + if e - end < self.WORD_S / 2: + end = e + if include_drops or not any(ds <= t < de + for ds, de in self.DROPS): + spans.append((f"w{idx}", round(t, 3), round(end, 3))) + idx += 1 + t = end + return words_from(spans) + + def _transcript(self, words): + return {"provider": "parakeet-mlx", "media": "m.mp4", + "duration": self.DURATION, "text": "", "words": words} + + def test_good_transcript_passes(self): + amap = audio_map(self.DURATION, self._silence()) + report = vt.build_report(self._transcript(self._words(True)), amap, + wpm=198) + self.assertTrue(report["ok"], report["dropped_regions"]) + self.assertEqual(report["dropped_regions"], []) + + def test_broken_transcript_fails(self): + amap = audio_map(self.DURATION, self._silence()) + report = vt.build_report(self._transcript(self._words(False)), amap, + wpm=198) + self.assertFalse(report["ok"]) + self.assertEqual(len(report["dropped_regions"]), 3) + + def test_broken_transcript_names_the_right_regions(self): + amap = audio_map(self.DURATION, self._silence()) + report = vt.build_report(self._transcript(self._words(False)), amap, + wpm=198) + starts = sorted(d["start"] for d in report["dropped_regions"]) + for expected, got in zip([49.0, 87.0, 577.0], starts): + self.assertAlmostEqual(got, expected, delta=1.5) + + def test_word_rate_alone_would_NOT_have_caught_it(self): + """The documented caveat, pinned as a test. + + Three missing paragraphs are a ~3 percent word deficit in a 20 minute + take, so the rate check passes on the BROKEN transcript. If this test + ever starts failing, the rate check has become sensitive enough to + matter and the docstring's caveat needs revisiting; until then, the + coverage scan is the only thing standing between a Swiss-cheese + transcript and a shipped cut.""" + amap = audio_map(self.DURATION, self._silence()) + broken = vt.build_report(self._transcript(self._words(False)), amap, + wpm=198) + self.assertTrue(broken["checks"]["word_rate"]["ok"]) + self.assertFalse(broken["checks"]["coverage"]["ok"]) + + def test_catastrophic_truncation_does_trip_the_rate_check(self): + amap = audio_map(self.DURATION, self._silence()) + half = self._words(True)[:len(self._words(True)) // 4] + report = vt.build_report(self._transcript(half), amap, wpm=198) + self.assertFalse(report["checks"]["word_rate"]["ok"]) + + +class TestLowConfidence(unittest.TestCase): + def test_run_of_low_confidence_words_is_reported(self): + words = words_from([(f"w{i}", i * 1.0, i * 1.0 + 0.5) + for i in range(8)]) + for w in words[2:7]: + w["confidence"] = 0.1 + runs = vt.low_confidence_runs(words) + self.assertEqual(len(runs), 1) + self.assertEqual(runs[0]["words"], 5) + + def test_short_run_is_ignored(self): + words = words_from([(f"w{i}", i * 1.0, i * 1.0 + 0.5) + for i in range(8)]) + words[3]["confidence"] = 0.1 + self.assertEqual(vt.low_confidence_runs(words), []) + + def test_confidence_never_fails_the_gate(self): + words = words_from([(f"w{i}", i * 1.0, i * 1.0 + 0.9) + for i in range(20)]) + for w in words: + w["confidence"] = 0.0 + # The map must carry the 0.1s inter-word gaps, as a real one built at + # the 0.1s default does; omitting them would make ordinary pauses + # read as untranscribed audio. + amap = audio_map(20.0, [(i * 1.0 + 0.9, (i + 1) * 1.0) + for i in range(19)] + [(19.9, 20.0)]) + report = vt.build_report({"words": words}, amap) + self.assertTrue(report["ok"], report["dropped_regions"]) + self.assertTrue(report["checks"]["confidence"]["ok"]) + self.assertEqual(len(report["low_confidence"]), 1) + + +class TestMapGranularityGuard(unittest.TestCase): + """A coarse audio map omits natural inter-word gaps, which then read as + untranscribed audio. Explain that rather than failing a good transcript + with no reason given.""" + + def test_fine_map_produces_no_warning(self): + self.assertIsNone(vt.map_too_coarse({"min_silence": 0.1})) + + def test_boundary_granularity_is_acceptable(self): + self.assertIsNone(vt.map_too_coarse({"min_silence": 0.2})) + + def test_coarse_map_warns(self): + warning = vt.map_too_coarse({"min_silence": 0.5}) + self.assertIsNotNone(warning) + self.assertIn("--map-granularity", warning) + + def test_warning_names_the_remedy(self): + self.assertIn("Regenerate the map", + vt.map_too_coarse({"min_silence": 0.5})) + + def test_missing_granularity_does_not_warn(self): + self.assertIsNone(vt.map_too_coarse({})) + + def test_warning_rides_on_the_report(self): + words = words_from([("a", 1.0, 2.0)]) + amap = audio_map(5.0, [(0.0, 1.0), (2.0, 5.0)]) + amap["min_silence"] = 0.9 + self.assertIsNotNone(vt.build_report({"words": words}, + amap)["map_warning"]) + + +class TestCli(unittest.TestCase): + def _good(self): + words = words_from([(f"w{i}", i * 1.0, i * 1.0 + 0.9) + for i in range(10)]) + return {"duration": 11.0, "words": words}, audio_map(11.0, + [(10.0, 11.0)]) + + def test_passing_transcript_exits_zero(self): + t, a = self._good() + report, r = run_cli(t, a, expect=0) + self.assertTrue(report["ok"]) + + def test_failing_transcript_exits_one_and_names_regions(self): + t, a = self._good() + a = audio_map(60.0, [(50.0, 60.0)]) # 40s of speech with no words + _, r = run_cli(t, a, expect=1) + self.assertIn("TRANSCRIPT INCOMPLETE", r.stderr) + self.assertIn("dropped speech at", r.stderr) + + def test_failure_message_points_at_the_remediation(self): + t, _ = self._good() + a = audio_map(60.0, [(50.0, 60.0)]) + _, r = run_cli(t, a, expect=1) + self.assertIn("--window", r.stderr) + + def test_missing_audio_map_is_usage_error(self): + with tempfile.TemporaryDirectory() as tmp: + w = Path(tmp) / "words.json" + w.write_text(json.dumps({"duration": 1.0, "words": []})) + r = subprocess.run( + [sys.executable, str(SCRIPT), str(w), "--audio-map", + str(Path(tmp) / "nope.json")], + capture_output=True, text=True) + self.assertEqual(r.returncode, 2) + + def test_audio_map_missing_keys_is_usage_error(self): + t, _ = self._good() + with tempfile.TemporaryDirectory() as tmp: + w = Path(tmp) / "words.json" + a = Path(tmp) / "audio-map.json" + w.write_text(json.dumps(t)) + a.write_text(json.dumps({"duration": 1.0})) + r = subprocess.run( + [sys.executable, str(SCRIPT), str(w), "--audio-map", str(a)], + capture_output=True, text=True) + self.assertEqual(r.returncode, 2) + self.assertIn("analyze_audio.py", r.stderr) + + def test_help_exits_zero(self): + r = subprocess.run([sys.executable, str(SCRIPT), "--help"], + capture_output=True, text=True) + self.assertEqual(r.returncode, 0) + self.assertIn("--wpm", r.stdout) + + +class TestParseRegion(unittest.TestCase): + + def test_plain_region(self): + self.assertEqual(vt.parse_region("12.5-18.0"), (12.5, 18.0)) + + def test_integers(self): + self.assertEqual(vt.parse_region("3-9"), (3.0, 9.0)) + + def test_whitespace_is_tolerated(self): + self.assertEqual(vt.parse_region(" 4.0-5.0 "), (4.0, 5.0)) + + def test_backwards_region_is_rejected(self): + with self.assertRaises(ValueError): + vt.parse_region("9-3") + + def test_zero_length_region_is_rejected(self): + with self.assertRaises(ValueError): + vt.parse_region("5-5") + + def test_garbage_is_rejected(self): + with self.assertRaises(ValueError): + vt.parse_region("sometime") + + +class TestAcceptances(unittest.TestCase): + """The escape valve: blocking by default, past it only with a reason.""" + + def setUp(self): + self.dropped = [{"start": 10.0, "end": 14.0, "audible": 4.0, + "at": "0:10.0"}] + + def test_no_acceptances_leaves_everything_failing(self): + remaining, cleared = vt.apply_acceptances(self.dropped, []) + self.assertEqual(len(remaining), 1) + self.assertEqual(cleared, []) + + def test_a_fully_covering_acceptance_clears_the_region(self): + accepted = [{"start": 9.0, "end": 15.0, "reason": "audience laugh"}] + remaining, cleared = vt.apply_acceptances(self.dropped, accepted) + self.assertEqual(remaining, []) + self.assertEqual(len(cleared), 1) + self.assertEqual(cleared[0]["accepted_by"]["reason"], "audience laugh") + + def test_an_exactly_covering_acceptance_clears_the_region(self): + accepted = [{"start": 10.0, "end": 14.0, "reason": "music bed"}] + remaining, _ = vt.apply_acceptances(self.dropped, accepted) + self.assertEqual(remaining, []) + + def test_partial_coverage_does_not_clear(self): + # The creator listened to 10-12. The region runs to 14. What is in + # 12-14 is still unaccounted for, so it keeps failing. + accepted = [{"start": 10.0, "end": 12.0, "reason": "laugh"}] + remaining, cleared = vt.apply_acceptances(self.dropped, accepted) + self.assertEqual(len(remaining), 1) + self.assertEqual(cleared, []) + + def test_an_unrelated_acceptance_does_not_clear(self): + accepted = [{"start": 100.0, "end": 200.0, "reason": "elsewhere"}] + remaining, _ = vt.apply_acceptances(self.dropped, accepted) + self.assertEqual(len(remaining), 1) + + def test_only_the_covered_region_clears(self): + dropped = self.dropped + [{"start": 40.0, "end": 44.0, + "audible": 4.0, "at": "0:40.0"}] + accepted = [{"start": 9.0, "end": 15.0, "reason": "laugh"}] + remaining, cleared = vt.apply_acceptances(dropped, accepted) + self.assertEqual(len(remaining), 1) + self.assertEqual(remaining[0]["start"], 40.0) + self.assertEqual(len(cleared), 1) + + +class TestAcceptanceCLI(unittest.TestCase): + """A run that fails must pass once the region is signed off, and only then.""" + + def broken(self): + # 3 seconds of audible speech with no words over it. + words = words_from([("hello", 0.0, 0.33), ("there", 0.33, 0.66)]) + amap = audio_map(10.0, [(0.66, 3.0), (6.0, 10.0)]) + return words, amap + + def test_the_region_fails_without_an_acceptance(self): + words, amap = self.broken() + report, r = run_cli({"words": words}, amap, expect=1) + self.assertFalse(report["ok"]) + self.assertEqual(report["checks"]["coverage"]["dropped_regions"], 1) + + def test_accepting_the_region_passes_the_gate(self): + words, amap = self.broken() + report, r = run_cli( + {"words": words}, amap, + extra=["--accept-region", "2.9-6.1", "--reason", "music bed"], + expect=0) + self.assertTrue(report["ok"]) + self.assertEqual(report["checks"]["coverage"]["accepted_regions"], 1) + + def test_the_acceptance_and_its_reason_are_recorded(self): + words, amap = self.broken() + report, _ = run_cli( + {"words": words}, amap, + extra=["--accept-region", "2.9-6.1", "--reason", "music bed"], + expect=0) + self.assertEqual(report["accepted"][0]["accepted_by"]["reason"], + "music bed") + + def test_acceptance_without_a_reason_is_a_usage_error(self): + words, amap = self.broken() + _, r = run_cli({"words": words}, amap, + extra=["--accept-region", "2.9-6.1"], expect=2) + self.assertIn("requires --reason", r.stderr) + + def test_an_empty_reason_is_a_usage_error(self): + words, amap = self.broken() + run_cli({"words": words}, amap, + extra=["--accept-region", "2.9-6.1", "--reason", " "], + expect=2) + + def test_a_reason_with_no_region_is_a_usage_error(self): + words, amap = self.broken() + run_cli({"words": words}, amap, extra=["--reason", "why"], expect=2) + + def test_a_malformed_region_is_a_usage_error(self): + words, amap = self.broken() + _, r = run_cli({"words": words}, amap, + extra=["--accept-region", "nope", "--reason", "x"], + expect=2) + self.assertIn("bad --accept-region", r.stderr) + + def test_a_partial_acceptance_still_fails_the_gate(self): + words, amap = self.broken() + report, _ = run_cli( + {"words": words}, amap, + extra=["--accept-region", "3.0-4.0", "--reason", "partial"], + expect=1) + self.assertFalse(report["ok"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/skills/mc-cut/scripts/transcribe.py b/skills/mc-cut/scripts/transcribe.py index 7fd2783..72ceb0c 100644 --- a/skills/mc-cut/scripts/transcribe.py +++ b/skills/mc-cut/scripts/transcribe.py @@ -34,9 +34,19 @@ times round to 2 decimals, confidence to 4. "i" is the word index. gap_before = start - previous word's end (first word: its start). gap_after = next word's start - end (last word: duration - end). - gaps never go negative (clamped to 0.0). The shape is identical - across providers; downstream consumers (cutplan.py) never need to - know which lane produced the file. + gaps never go negative (clamped to 0.0). "window_s"/"overlap_s" + record the windowing this transcript was produced with. + + The JSON SHAPE is identical across providers, but gap VALUES are + not comparable between lanes and neither lane's gaps are a + timing source of truth. parakeet absorbs a pause into the + preceding token's end, so an mlx-lane gap across real silence + can read 0.0; the onnx lane caps derived word ends + (WORD_END_CAP_FRAMES) precisely to avoid that, so its gaps are + closer to real but still only advisory. Consumers deciding WHERE + TO CUT must use the audio silence map from analyze_audio.py, not + these gaps (see mc-cut/SKILL.md). Gaps remain useful as a weak + signal for sentence-start detection and nothing else. provider from the studio config [transcription] table; this script is the switch. "auto" (the default) picks per platform: macOS Apple Silicon -> parakeet-mlx (the reference lane) @@ -68,15 +78,38 @@ automatically; otherwise it runs on CPU, and if nvidia-smi is on PATH it prints a loud warning that the GPU escalation is needed (or failed) instead of silently running slow. - chunking parakeet-mlx chunks long audio internally. The onnx-asr lane - caps around 20-30 s per call, so this script extracts fixed - 20 s windows with 2 s overlap via ffmpeg (16 kHz mono wav), - offsets each chunk's timestamps by its window start, and merges - at the overlap midpoint on word boundaries with a seam-repair - pass (the two chunks time boundary words independently, so a - take kept by both sides deduplicates and a take kept by - neither restores from the nearer chunk): no word is split, - duplicated, or dropped across chunks. + chunking EVERY lane windows. Both providers extract fixed 20 s windows + with 3 s overlap via ffmpeg (16 kHz mono wav), recognize each + window in ISOLATION, offset each chunk's timestamps by its + window start, and merge at the overlap midpoint on word + boundaries with a seam-repair pass (the two chunks time boundary + words independently, so a take kept by both sides deduplicates + and a take kept by neither restores from the nearer chunk): no + word is split, duplicated, or dropped across chunks. + + This is NOT an onnx-asr accommodation, and short windows are not + optional on either lane. parakeet's decoder SILENTLY DROPS whole + spans inside long windows, worst around the many pauses of a + teleprompter read: no error, no warning, just missing + paragraphs. Measured on one 20.5 min take (2026-07-24): + chunk_duration=120 -> 3,446 words, 3 paragraphs lost + 90 s isolated windows -> 3,308 words, still lossy + 20 s isolated windows -> 3,546 words, complete + An earlier version of this docstring claimed parakeet-mlx chunks + long audio internally and the mlx lane called + model.transcribe() with no chunk_duration. That was + false twice over: the call OOMs on Apple Silicon above roughly + 15 min of 4K source (Metal buffer ceiling), and working around + the OOM with a large chunk_duration silently loses speech. Do + not "optimize" this back into fewer, longer windows, and do not + rely on model.transcribe(..., chunk_duration=...) as a + substitute for isolated windows. + + --window/--overlap tune the windowing; the defaults are the + validated values and the recorded window_s/overlap_s in the + output say what a given transcript actually used. Whatever the + setting, verify_transcript.py is the gate that proves the + result is complete. confidence parakeet-mlx reports per-token confidence natively. The onnx-asr lane maps per-token scores when the runtime exposes them (logprobs are exponentiated into probabilities, values @@ -128,9 +161,13 @@ # this many frames past its start; otherwise every pause would be absorbed # into the preceding word and the gap data cutting depends on would read 0. WORD_END_CAP_FRAMES = 3 -# onnx-asr caps most models around 20-30 s per call; fixed windows + overlap. +# Fixed transcription windows, applied on EVERY lane. 20 s is the empirically +# validated ceiling for lossless parakeet decoding (see the `chunking` note in +# the module docstring); longer windows silently drop speech on the mlx lane +# and exceed most onnx models' per-call cap. 3 s of overlap gives the seam +# repair in merge_chunk_tokens enough context to resolve boundary words. CHUNK_WINDOW_S = 20.0 -CHUNK_OVERLAP_S = 2.0 +CHUNK_OVERLAP_S = 3.0 # SentencePiece word-boundary marker used by the ONNX tokenizer. SP_MARK = "▁" @@ -231,18 +268,95 @@ def probe_duration(media): return float(out.stdout.strip()) -def transcribe_parakeet(media, model_id): - """Run parakeet-mlx and return (full_text, tokens, duration). +def mlx_tokens_to_dicts(tokens, clamp=None): + """Convert parakeet-mlx AlignedTokens into the shared token dict shape (pure). - tokens are AlignedToken objects (text/start/end/confidence). Import is - local so the pure helpers stay importable without the model dependency. + AlignedToken carries text/start/end/confidence, and its raw text already + leads with a space at a word boundary, which is the same marker + group_subwords and _word_runs key on. So the conversion is a straight + field copy: no grouping, no timestamp derivation, no confidence mapping. + clamp, when given, caps end times at the window length. A bare + whitespace token carries its word boundary onto the next token rather + than emitting an empty word (mirroring onnx_tokens_to_parakeet). + """ + out = [] + pending_space = False + for t in tokens: + text = str(_get(t, "text")) + if text.strip() == "": + if text.startswith(" "): + pending_space = True + continue + if pending_space: + if not text.startswith(" "): + text = " " + text + pending_space = False + start = float(_get(t, "start")) + end = float(_get(t, "end")) + if clamp is not None: + end = min(end, float(clamp)) + if end < start: + end = start + out.append({ + "text": text, + "start": start, + "end": end, + "confidence": float(_get(t, "confidence")), + }) + return out + + +def transcribe_windowed(media, recognize_window, window=CHUNK_WINDOW_S, + overlap=CHUNK_OVERLAP_S): + """Drive any per-window recognizer over isolated windows and merge. + + The lane-agnostic transcription path: plan fixed windows, extract each to + a 16 kHz mono wav via ffmpeg (so video containers work on every lane), + hand it to `recognize_window(wav_path, length) -> token dicts` with times + relative to the window, offset those into absolute time, and merge at the + overlap midpoints. Returns (full_text, tokens, duration). + + Every provider goes through here. A lane that recognized the whole file + in one pass would reintroduce the silent-drop bug documented in the + module docstring's `chunking` note. + """ + duration = probe_duration(media) + chunks = plan_chunks(duration, window=window, overlap=overlap) + + per_chunk = [] + with tempfile.TemporaryDirectory(prefix="mc-transcribe-") as tmp: + for i, (start, length) in enumerate(chunks): + print( + f"chunk {i + 1}/{len(chunks)}: {start:.2f}s +{length:.2f}s", + file=sys.stderr, + ) + wav = Path(tmp) / f"chunk{i:04d}.wav" + extract_chunk(media, start, length, wav) + tokens = recognize_window(wav, length) + per_chunk.append((start, start + length, offset_tokens(tokens, start))) + + merged = merge_chunk_tokens(per_chunk) + text = "".join(t["text"] for t in merged).strip() + return text, merged, duration + + +def transcribe_parakeet(media, model_id, window=CHUNK_WINDOW_S, + overlap=CHUNK_OVERLAP_S): + """Run parakeet-mlx over isolated windows and return (text, tokens, duration). + + The model loads ONCE and is reused across windows; only the audio handed + to it is short. Import is local so the pure helpers stay importable + without the model dependency. """ import parakeet_mlx model = parakeet_mlx.from_pretrained(model_id) - result = model.transcribe(str(media)) - duration = probe_duration(media) - return result.text, result.tokens, duration + + def recognize(wav, length): + result = model.transcribe(str(wav)) + return mlx_tokens_to_dicts(result.tokens, clamp=length) + + return transcribe_windowed(media, recognize, window=window, overlap=overlap) # --- onnx-asr lane ----------------------------------------------------------- @@ -571,39 +685,25 @@ def _result_scores(result): def transcribe_onnx(media, model_id, window=CHUNK_WINDOW_S, overlap=CHUNK_OVERLAP_S): - """Run onnx-asr over fixed windows and return (full_text, tokens, duration). + """Run onnx-asr over isolated windows and return (text, tokens, duration). - Every window is extracted to a 16 kHz mono wav via ffmpeg (so video - containers work exactly like the mlx lane), recognized with timestamps, - converted to parakeet-shaped token dicts, offset by the window start, and - merged at overlap midpoints. Imports are local so the pure helpers stay - importable without the onnx-asr dependency. + Same driver as the mlx lane (transcribe_windowed); this lane supplies only + the per-window recognizer, which converts onnx-asr's token/timestamp + arrays into the shared token dict shape. Imports are local so the pure + helpers stay importable without the onnx-asr dependency. """ - duration = probe_duration(media) - chunks = plan_chunks(duration, window=window, overlap=overlap) model = _load_onnx_model(model_id) - per_chunk = [] - with tempfile.TemporaryDirectory(prefix="mc-transcribe-") as tmp: - for i, (start, length) in enumerate(chunks): - print( - f"chunk {i + 1}/{len(chunks)}: {start:.2f}s +{length:.2f}s", - file=sys.stderr, - ) - wav = Path(tmp) / f"chunk{i:04d}.wav" - extract_chunk(media, start, length, wav) - result = model.with_timestamps().recognize(str(wav)) - tokens = onnx_tokens_to_parakeet( - list(result.tokens), - list(result.timestamps), - logprobs=_result_scores(result), - clamp=length, - ) - per_chunk.append((start, start + length, offset_tokens(tokens, start))) + def recognize(wav, length): + result = model.with_timestamps().recognize(str(wav)) + return onnx_tokens_to_parakeet( + list(result.tokens), + list(result.timestamps), + logprobs=_result_scores(result), + clamp=length, + ) - merged = merge_chunk_tokens(per_chunk) - text = "".join(t["text"] for t in merged).strip() - return text, merged, duration + return transcribe_windowed(media, recognize, window=window, overlap=overlap) def main(argv=None): @@ -619,8 +719,27 @@ def main(argv=None): help="model id (default per provider: " f"{DEFAULT_MODELS[PROVIDER_MLX]} for parakeet-mlx, " f"{DEFAULT_MODELS[PROVIDER_ONNX]} for onnx-asr)") + parser.add_argument("--window", type=float, default=CHUNK_WINDOW_S, + help=f"transcription window seconds (default " + f"{CHUNK_WINDOW_S}; longer windows silently drop " + "speech, see the chunking note in this script)") + parser.add_argument("--overlap", type=float, default=CHUNK_OVERLAP_S, + help=f"window overlap seconds (default {CHUNK_OVERLAP_S})") args = parser.parse_args(argv) + if args.overlap < 0 or args.overlap >= args.window: + print("error: --overlap must be >= 0 and smaller than --window", + file=sys.stderr) + return 2 + if args.window > CHUNK_WINDOW_S: + print( + f"WARNING: --window {args.window}s exceeds the validated " + f"{CHUNK_WINDOW_S}s ceiling. parakeet silently drops speech inside " + "long windows. Run verify_transcript.py on the result before " + "cutting against it.", + file=sys.stderr, + ) + provider = args.provider if provider == PROVIDER_AUTO: provider = default_provider() @@ -645,9 +764,11 @@ def main(argv=None): try: if provider == PROVIDER_MLX: - text, tokens, duration = transcribe_parakeet(media, model_id) + text, tokens, duration = transcribe_parakeet( + media, model_id, window=args.window, overlap=args.overlap) else: - text, tokens, duration = transcribe_onnx(media, model_id) + text, tokens, duration = transcribe_onnx( + media, model_id, window=args.window, overlap=args.overlap) except ImportError as exc: print( f"error: provider {provider} dependencies unavailable on this " @@ -666,6 +787,8 @@ def main(argv=None): "model": model_id, "media": args.media, "duration": round(float(duration), 2), + "window_s": round(float(args.window), 3), + "overlap_s": round(float(args.overlap), 3), "text": text.strip(), "words": words, } @@ -679,8 +802,11 @@ def main(argv=None): "model": model_id, "media": args.media, "duration": payload["duration"], + "window_s": payload["window_s"], + "overlap_s": payload["overlap_s"], "words": len(words), "output": str(output), + "next": "verify_transcript.py (completeness gate) before cutplan.py", })) return 0 diff --git a/skills/mc-cut/scripts/verify_edl.py b/skills/mc-cut/scripts/verify_edl.py new file mode 100644 index 0000000..93fac08 --- /dev/null +++ b/skills/mc-cut/scripts/verify_edl.py @@ -0,0 +1,393 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.11" +# /// +"""EDL gate: prove no cut lands inside a word before anything renders. + +Usage: + uv run {skill-root}/scripts/verify_edl.py {projects-path}//cut/edl.json \ + --audio-map {projects-path}//cut/audio-map.json \ + --words {projects-path}//transcript/words.json \ + [--tolerance 0.0] [-o {projects-path}//cut/edl-check.json] + +Why this exists: + mc-cut's cutting rules have always said "never cut inside a word" and its + checklist has always claimed "edl times checked against word timestamps + AND resting in audio silences". Nothing enforced either one. cutplan.py + snaps CANDIDATES into silence, but the EDL is written by hand at step 6 + and rewritten at step 7a, and until this script nothing ever read it back. + + So the stage's own deliverable was the one artifact with no gate on it, + while its siblings all got one: preflight asserts on source QC, + verify_transcript asserts on transcript completeness, verify_anchors + asserts on beat placement. This is mc-cut's twin of verify_anchors.py. + A check the pipeline claims to perform must be a script that exits + non-zero (AGENTS.md); this is that script for the cut itself. + +Why silence is the authority and word spans are only context: + The obvious implementation is "fail any boundary that falls inside a word + span". That is WRONG here, and shipping it would fail correct cuts. + parakeet absorbs a pause into the preceding word's end, so the word + "about." can be timestamped 32.16 -> 34.64: a single "word" covering 2.5 + seconds because it swallowed the silence after it. A cut correctly placed + in that silence sits inside the word's timestamp span while sitting in + real, audible silence. + + So the check follows the two-source rule the module already commits to. + The AUDIO decides: a boundary resting inside an audio-verified silence + CANNOT clip a word, whatever the transcript timestamps say. The word span + is reported only as extra detail on a boundary that ALREADY failed the + audio test, where "and it is mid-word" tells the creator how bad the miss + is. A boundary is never failed on word overlap alone. + +What is exempt, and why: + - The head of the source (start 0.0) and its tail (end == source_duration) + are not cuts. No material was removed there, so there is nothing to + clip. + - A boundary shared by two segments that are contiguous in the same source + (prev.end == next.start) is not a cut either: the timeline is continuous + across it and nothing was removed. + Segments are NOT required to be in source order. Step 5 explicitly picks + best takes and orders segments, so a reordered EDL is correct by design + and this script must not fail it. + +Checks, per segment: + 1. Structure: start < end, both inside [0, source_duration], source set. + 2. Provenance: quote and reason are both non-empty (the EDL's own contract + is that every segment records what was said and why it was kept). + 3. Boundaries: every non-exempt boundary rests inside an audio-verified + silence, within --tolerance. Failures report the distance to the + nearest silence and whether the boundary is mid-word. + +Contract: + input cut/edl.json, one or more audio-map.json, one or more + words.json. With a single map (or single transcript) it applies + to every segment; with several they are matched to each + segment's `source`. + output optional -o report JSON: {"ok", "segments", "boundaries", + "violations": [...], "exempt": [...]} + summary json.dumps on stdout either way. + +Exit codes: 0 every segment verified, 1 one or more violations (do NOT +render, do not export a timeline, do not present this cut), 2 usage error. + +STATUS: implemented (pure logic covered by scripts/tests/test-verify_edl.py). +""" + +import argparse +import json +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent)) # noqa: E402 +import analyze_audio as audio # noqa: E402 + +# How far a boundary may sit from a silence and still pass. Zero by default: +# cutplan snaps edges INTO silence, so a boundary that is merely near one was +# not derived the way the pipeline derives them. Raise it only to triage an +# inherited EDL. +DEFAULT_TOLERANCE_S = 0.0 +# Float slop when comparing times against duration and against each other. +EPS = 1e-6 + + +def r3(x): + return round(float(x), 3) + + +def word_at(words, t): + """The word whose span strictly contains t, or None (pure). + + Context only. See the module docstring: a pause-absorbed word end reaches + past the sound, so this answer alone never fails a boundary. + """ + for w in words: + if float(w["start"]) < t < float(w["end"]): + return w + return None + + +def natural_edges(segments, duration): + """Boundary times that are not cuts (pure). + + The source head and tail, plus any boundary where two segments run + contiguously in the same source so nothing was removed between them. + """ + # Malformed segments are reported by verify_segment, so this pass must + # tolerate them rather than raising: a gate that crashes on bad input + # tells the creator nothing about what is wrong. + exempt = set() + for seg in segments: + source, start, end = _edges(seg) + if start is None or end is None: + continue + if abs(start) <= EPS: + exempt.add((source, r3(start))) + if duration is not None and abs(end - duration) <= EPS: + exempt.add((source, r3(end))) + for a, b in zip(segments, segments[1:]): + a_source, _, a_end = _edges(a) + b_source, b_start, _ = _edges(b) + if a_source != b_source or a_end is None or b_start is None: + continue + if abs(a_end - b_start) <= EPS: + exempt.add((a_source, r3(a_end))) + return exempt + + +def _edges(seg): + """(source, start, end) with non-numeric times as None (pure).""" + if not isinstance(seg, dict): + return None, None, None + try: + return seg.get("source"), float(seg["start"]), float(seg["end"]) + except (KeyError, TypeError, ValueError): + return seg.get("source"), None, None + + +def check_boundary(t, silence, words, tolerance=DEFAULT_TOLERANCE_S): + """Verify one boundary time against the audio (pure). + + Returns None when the boundary is safe, or a dict describing the miss. + """ + if audio.enclosing(silence, t) is not None: + return None + landing = audio.nearest_silence(silence, t, max_shift=float("inf")) + distance = None if landing is None else abs(landing - t) + if distance is not None and distance <= tolerance + EPS: + return None + detail = {"time": r3(t)} + if distance is not None: + detail["nearest_silence"] = r3(landing) + detail["distance"] = r3(distance) + w = word_at(words, t) + if w is not None: + detail["inside_word"] = str(w["word"]) + detail["word_span"] = [r3(w["start"]), r3(w["end"])] + return detail + + +def verify_segment(index, seg, silence, words, duration, exempt, + tolerance=DEFAULT_TOLERANCE_S): + """Verify one EDL segment. Returns a list of violation dicts (pure).""" + out = [] + sid = seg.get("id", index) + source = seg.get("source") + if not source: + out.append({"segment": sid, "kind": "structure", + "reason": "segment has no source"}) + return out + try: + start = float(seg["start"]) + end = float(seg["end"]) + except (KeyError, TypeError, ValueError): + out.append({"segment": sid, "kind": "structure", + "reason": "segment start/end missing or not a number"}) + return out + + if end - start <= EPS: + out.append({"segment": sid, "kind": "structure", "time": r3(start), + "reason": f"segment start {start:.3f} is not before end " + f"{end:.3f}"}) + return out + if start < -EPS: + out.append({"segment": sid, "kind": "structure", "time": r3(start), + "reason": f"segment starts before zero ({start:.3f})"}) + if duration is not None and end > duration + EPS: + out.append({"segment": sid, "kind": "structure", "time": r3(end), + "reason": f"segment ends at {end:.3f}, past the source " + f"duration {duration:.3f}"}) + + for field in ("quote", "reason"): + value = seg.get(field) + if value is None or not str(value).strip(): + out.append({"segment": sid, "kind": "provenance", + "reason": f"segment has no {field}; every EDL segment " + "records the words it carries and why"}) + + for edge, t in (("start", start), ("end", end)): + if (source, r3(t)) in exempt: + continue + miss = check_boundary(t, silence, words, tolerance) + if miss is None: + continue + detail = dict(miss) + detail["segment"] = sid + detail["kind"] = "boundary" + detail["edge"] = edge + where = "" + if "distance" in detail: + where = (f"; nearest silence is {detail['distance']:.3f}s away at " + f"{detail['nearest_silence']:.3f}") + else: + where = "; this source has no detected silence at all" + word = "" + if "inside_word" in detail: + word = (f', and it lands inside the word "{detail["inside_word"]}"' + f" ({detail['word_span'][0]:.3f}-" + f"{detail['word_span'][1]:.3f})") + detail["reason"] = ( + f"segment {edge} {t:.3f} does not rest in an audio-verified " + f"silence{where}{word}. Snap it with snap_spans.py instead of " + "placing it by hand.") + out.append(detail) + return out + + +def build_report(edl, maps, transcripts, tolerance=DEFAULT_TOLERANCE_S): + """Verify every segment and assemble the verdict (pure). + + maps and transcripts are {source_key: value} indexes; a single entry + keyed None applies to every segment. + """ + segments = edl["segments"] + duration = edl.get("source_duration") + duration = None if duration is None else float(duration) + exempt = natural_edges(segments, duration) + + violations = [] + boundaries = 0 + for i, seg in enumerate(segments): + source = seg.get("source") + silence = _pick(maps, source) + words = _pick(transcripts, source) + if silence is None: + violations.append({ + "segment": seg.get("id", i), "kind": "structure", + "reason": f"no audio map covers source {source!r}; run " + "analyze_audio.py on it"}) + continue + for edge in ("start", "end"): + if (source, r3(seg.get(edge, 0))) not in exempt: + boundaries += 1 + violations.extend(verify_segment(i, seg, silence, words or [], + duration, exempt, tolerance)) + + return { + "ok": not violations, + "segments": len(segments), + "boundaries": boundaries, + "exempt": sorted({r3(t) for _, t in exempt}), + "tolerance": tolerance, + "violations": violations, + } + + +def _pick(index, source): + """Look a source up in a {key: value} index (pure). + + A lone entry keyed None applies to everything, which is the single-source + project. Otherwise match the resolved path, then the basename. + """ + if None in index: + return index[None] + if source in index: + return index[source] + try: + resolved = str(Path(source).resolve()) + except (TypeError, OSError): + resolved = None + if resolved and resolved in index: + return index[resolved] + base = Path(source).name if source else None + return index.get(base) + + +def _load(path, label): + try: + with open(path, encoding="utf-8") as f: + return json.load(f) + except (OSError, json.JSONDecodeError) as e: + print(f"verify_edl: cannot read {label} {path}: {e}", file=sys.stderr) + return None + + +def _index(payloads, extract, single): + """Index loaded payloads by their media path, or by None when there is one.""" + if len(payloads) == 1 and single: + return {None: extract(payloads[0])} + out = {} + for p in payloads: + media = p.get("media") or p.get("source") + value = extract(p) + if media: + out[str(media)] = value + try: + out[str(Path(media).resolve())] = value + except (TypeError, OSError): + pass + out[Path(str(media)).name] = value + return out + + +def main(argv=None): + p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + p.add_argument("edl", help="path to cut/edl.json") + p.add_argument("--audio-map", required=True, action="append", + help="path to cut/audio-map.json from analyze_audio.py; " + "repeat once per source on a multi-source project") + p.add_argument("--words", required=True, action="append", + help="path to transcript/words.json; repeat once per " + "source on a multi-source project") + p.add_argument("-o", "--output", default=None, + help="optional path for the full report JSON") + p.add_argument("--tolerance", type=float, default=DEFAULT_TOLERANCE_S, + help=f"seconds a boundary may sit outside a silence " + f"(default {DEFAULT_TOLERANCE_S}; raise only to " + "triage an inherited EDL)") + args = p.parse_args(argv) + + edl = _load(args.edl, "edl") + if edl is None: + return 2 + if not edl.get("segments"): + print("verify_edl: edl has no segments", file=sys.stderr) + return 2 + + raw_maps = [_load(m, "audio map") for m in args.audio_map] + if any(m is None for m in raw_maps): + return 2 + for m, path in zip(raw_maps, args.audio_map): + if "silence" not in m: + print(f"verify_edl: audio map {path} has no 'silence' key; " + "regenerate it with analyze_audio.py", file=sys.stderr) + return 2 + raw_words = [_load(w, "transcript") for w in args.words] + if any(w is None for w in raw_words): + return 2 + for w, path in zip(raw_words, args.words): + if "words" not in w: + print(f"verify_edl: transcript {path} has no 'words' key", + file=sys.stderr) + return 2 + + maps = _index(raw_maps, lambda m: audio.to_pairs(m["silence"]), + single=len(raw_maps) == 1) + transcripts = _index(raw_words, lambda w: w["words"], + single=len(raw_words) == 1) + + report = build_report(edl, maps, transcripts, args.tolerance) + + if args.output: + out = Path(args.output) + out.parent.mkdir(parents=True, exist_ok=True) + out.write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n", + encoding="utf-8") + + print(json.dumps({k: report[k] for k in + ("ok", "segments", "boundaries", "tolerance")}, indent=2)) + + if report["ok"]: + return 0 + + print(f"\nEDL VERIFICATION FAILED: {len(report['violations'])} problem(s). " + "Do NOT render, export a timeline, or present this cut.", + file=sys.stderr) + for v in report["violations"]: + print(f" segment {v['segment']} [{v['kind']}]: {v['reason']}", + file=sys.stderr) + return 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/mc-cut/scripts/verify_transcript.py b/skills/mc-cut/scripts/verify_transcript.py new file mode 100644 index 0000000..4df5a18 --- /dev/null +++ b/skills/mc-cut/scripts/verify_transcript.py @@ -0,0 +1,504 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.11" +# /// +"""Transcript completeness gate: prove the transcript is not missing speech. + +Usage: + uv run {skill-root}/scripts/verify_transcript.py \ + {projects-path}//transcript/words.json \ + --audio-map {projects-path}//cut/audio-map.json \ + [--wpm 198] [-o {projects-path}//cut/transcript-check.json] + +Why this exists: + On the first real project (2026-07-24) the transcriber silently dropped + three whole paragraphs of clearly-spoken content. No error, no warning. + Everything downstream ran on the hole: candidates, EDL, cutplan, render, + and a passed gate 2. The dropped regions looked like dead air to the + cutter, which nearly deleted three paragraphs of the actual video. + + Nothing in the pipeline ever checked. This script is that check, and it + RUNS BEFORE CANDIDATE DETECTION. It exits non-zero, and a non-zero exit + means stop: do not cut against this transcript. + +The checks, in order of what actually catches things: + + 1. COVERAGE SCAN (the decisive one, exit 1 on failure) + Audio that is above the silence floor but has no transcribed words is + DROPPED SPEECH. Computed as the complement of (silence + word spans): + anything left over is audible, un-transcribed, and therefore missing. + Contiguous leftovers are clustered and any cluster at or above + --min-drop seconds fails the gate, reported with timecodes so the + region can be re-transcribed in isolation and spliced. + + This check is deliberately built from the AUDIO side, not from word + gaps. Word gaps cannot be trusted: parakeet absorbs pauses into the + preceding word's end, so a gap across real silence reads about 0.0 (see + analyze_audio.py). A check that scanned word gaps would inherit exactly + the bug it is meant to catch. + + 2. WORD-RATE SANITY (exit 1 on failure, but read the caveat) + Effective words per minute over SPEECH time (dead air excluded, which + wall-clock rate does not do) against the creator's configured wpm. + + CAVEAT, measured on the real failure: this check would NOT have caught + it. The broken transcript had 3,446 words against a good 3,546, a 3 + percent deficit, because three missing paragraphs are a small fraction + of a 20 minute take. Against speech time it reads 224 wpm versus 231 + for the good one; against wall-clock, 168 versus 173. Either way it + sails past any sane floor. The original bug report proposed a 55 + percent floor and that would have passed the broken transcript too. + + So this check is real but narrow: it catches CATASTROPHIC failure (a + lane that returned half a file, a wrong-language model, a silent + truncation) and nothing subtler. Check 1 is the one that catches + dropped paragraphs. Do not let this check's presence create confidence + that check 1 is optional. + + 3. CONFIDENCE CLUSTERS (advisory, never fails the gate) + A run of near-zero-confidence words is a soft signal to re-check that + region by ear. Reported, never blocking. On the onnx lane confidence + degrades to 1.0 when the runtime exposes no scores, so absence of this + signal means nothing. + +Contract: + input words.json from transcribe.py, plus audio-map.json from + analyze_audio.py (both must describe the same media). + output optional -o report JSON: {"ok", "checks": {...}, + "dropped_regions": [...], "accepted": [...], "word_rate": {...}, + "low_confidence": [...]} + summary json.dumps on stdout either way. + +Exit codes: + 0 transcript passes; safe to build candidates against + 1 transcript INCOMPLETE or implausible; do not proceed + 2 usage error (missing file, bad arguments, media mismatch) + +Remediation when this fails: re-run the take (see transcribe.py's `chunking` +note for the measured window sizes), then re-verify. + +When the finding is a FALSE POSITIVE, the escape valve: + Not every audible span with no words is lost speech. A laugh, a music + bed, an off-mic aside, or a long breath above the noise floor all read + the same way to a coverage scan. So the creator can listen to a flagged + region and sign it off: + + --accept-region 612.4-618.9 --reason "audience laugh, no speech" + + The acceptance must FULLY cover the flagged region, --reason is + mandatory, and both land in the report, so the override is a recorded + decision rather than a silent one. This is the same shape as preflight's + --allow-qc-defects: the gate stays blocking by default, and the only way + past it is a human saying why. A gate with no acknowledged override gets + worked around in ways that leave no trace, which is worse. + +STATUS: implemented (pure logic covered by +scripts/tests/test-verify_transcript.py). +""" + +import argparse +import json +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent)) +import analyze_audio as audio + +# A cluster of audible-but-untranscribed audio at or above this many seconds +# is dropped speech. +# +# Calibrated against the real 20.5 minute take from 2026-07-24, sweeping this +# value over its known-good transcript (3,546 words) and its known-broken one +# (3,446 words, three paragraphs lost): +# +# min-drop false positives on good real drops caught in broken +# 0.75 0 5 +# 1.00 0 5 +# 1.50 0 4 +# 2.50 0 4 +# 3.00 0 3 +# +# The good transcript produces ZERO false positives all the way down to +# 0.75s, because audio under the -30 dB floor (breaths, lip noise, room +# tone) is already classified as silence and never reaches this check. So +# the conservative 2.5s this shipped with bought no safety and cost +# sensitivity: it missed a real drop that 1.0s catches. 1.0s keeps a little +# headroom over the measured noise margin. +# +# Raise it for a noisy recording environment where room tone sits above the +# silence floor; the right fix there is usually a lower --noise in +# analyze_audio.py instead. +DEFAULT_MIN_DROP = 1.0 +# Uncovered pieces closer together than this are one region for reporting; a +# dropped paragraph shows up as many short uncovered runs separated by the +# speaker's own natural pauses, and reporting them individually is noise. +DEFAULT_CLUSTER_GAP = 1.0 +# Effective speech-time wpm must reach this fraction of the configured wpm. +# See the caveat in the module docstring: this is a catastrophe detector. +DEFAULT_WPM_FLOOR_RATIO = 0.6 +DEFAULT_LOW_CONFIDENCE = 0.35 +DEFAULT_LOW_CONFIDENCE_RUN = 5 + + +def r2(x): + return round(float(x), 2) + + +def tc(seconds): + """Seconds to m:ss for human-readable region reports (pure).""" + seconds = max(0.0, float(seconds)) + return f"{int(seconds // 60)}:{seconds % 60:04.1f}" + + +def word_spans(words): + """Word intervals as (start, end) pairs (pure).""" + return [(float(w["start"]), float(w["end"])) for w in words] + + +def uncovered_regions(silence, words, duration): + """Audible spans with no transcribed words (pure). + + The complement of (silence + word spans) over [0, duration]. What is left + is audio above the noise floor that produced no words, which is the + definition of dropped speech. Merging the two lists first is what makes + this immune to pause-absorbed word ends: an absorbed end simply covers a + bit more, it cannot manufacture coverage of a genuinely empty region. + """ + covered = audio._clean(list(silence) + word_spans(words), duration) + return audio.complement(covered, duration) + + +def cluster(intervals, gap): + """Merge intervals separated by less than `gap` (pure).""" + out = [] + for start, end in intervals: + if out and start - out[-1][1] < gap: + out[-1][1] = max(out[-1][1], end) + else: + out.append([start, end]) + return [(s, e) for s, e in out] + + +def find_dropped(silence, words, duration, min_drop=DEFAULT_MIN_DROP, + cluster_gap=DEFAULT_CLUSTER_GAP): + """Dropped-speech regions at or above min_drop seconds (pure). + + Returns a list of {start, end, dur, audible} records. `audible` is the + seconds of actual uncovered audio inside the clustered region, which is + less than `dur` when the cluster spans a short pause; the threshold is + applied to `audible`, not to the cluster's wall span, so clustering can + never inflate a region past the gate. + """ + pieces = uncovered_regions(silence, words, duration) + out = [] + for start, end in cluster(pieces, cluster_gap): + audible = audio.overlap_seconds(pieces, start, end) + if audible >= min_drop: + out.append({ + "start": r2(start), + "end": r2(end), + "dur": r2(end - start), + "audible": r2(audible), + "at": tc(start), + }) + return out + + +def word_rate(words, duration, speech_seconds, wpm=None, + floor_ratio=DEFAULT_WPM_FLOOR_RATIO): + """Effective word rates and the pass/fail verdict (pure). + + speech_wpm excludes dead air, which is why it is the one gated on: a take + with five minutes of pauses is not a slow speaker. wall_wpm is reported + for continuity with hand calculations but never gated. + """ + n = len(words) + wall_wpm = (n / (duration / 60.0)) if duration > 0 else 0.0 + speech_wpm = (n / (speech_seconds / 60.0)) if speech_seconds > 0 else 0.0 + result = { + "words": n, + "wall_wpm": r2(wall_wpm), + "speech_wpm": r2(speech_wpm), + "configured_wpm": wpm, + "floor_ratio": floor_ratio, + } + if not wpm: + result["ok"] = True + result["note"] = ("no configured wpm supplied; rate check skipped " + "(pass --wpm from [owner] wpm in the studio config)") + return result + floor = wpm * floor_ratio + result["floor_wpm"] = r2(floor) + result["ok"] = speech_wpm >= floor + return result + + +def low_confidence_runs(words, threshold=DEFAULT_LOW_CONFIDENCE, + run_min=DEFAULT_LOW_CONFIDENCE_RUN): + """Runs of consecutive low-confidence words (pure, advisory only).""" + out = [] + run = [] + for w in words: + if float(w.get("confidence", 1.0)) < threshold: + run.append(w) + continue + if len(run) >= run_min: + out.append(run) + run = [] + if len(run) >= run_min: + out.append(run) + return [{ + "start": r2(r[0]["start"]), + "end": r2(r[-1]["end"]), + "words": len(r), + "at": tc(r[0]["start"]), + "text": " ".join(w["word"] for w in r)[:120], + } for r in out] + + +# An audio map coarser than this cannot be trusted by the coverage scan. +# See map_too_coarse. +MAX_USABLE_MAP_GRANULARITY = 0.2 + + +def map_too_coarse(audio_map, min_drop=DEFAULT_MIN_DROP): + """Warning text when the audio map is too coarse to scan against (pure). + + The coverage scan calls anything that is neither silence nor a word + "dropped speech". Natural inter-word gaps are 0.1 to 0.2s, so they are + only correctly classified when the audio map actually CONTAINS silences + that small. A map generated with a coarse --map-granularity omits them, + they read as uncovered, and enough of them in a row cluster into a + phantom dropped region. The gate would then fail a good transcript. + + analyze_audio.py defaults fine enough for this; the check exists because + someone passing --map-granularity to speed that step up would otherwise + get a confusing false failure rather than an explanation. + + `min_silence` is read as a fallback for maps written before the flag was + renamed to say what it actually controls. + """ + granularity = audio_map.get("map_granularity", + audio_map.get("min_silence")) + if granularity is None or granularity <= MAX_USABLE_MAP_GRANULARITY: + return None + return ( + f"WARNING: the audio map was built with --map-granularity " + f"{granularity}s, which is coarser than natural inter-word gaps. " + f"Silences below {granularity}s are missing from it, so ordinary " + "pauses between words can register as untranscribed audio and cluster " + "into phantom dropped regions. Regenerate the map with " + f"--map-granularity {MAX_USABLE_MAP_GRANULARITY} or finer before " + "trusting a failure from this gate." + ) + + +def parse_region(text): + """Parse a "START-END" acceptance region into (start, end) (pure). + + Raises ValueError on anything malformed, because a mistyped acceptance + that silently covered nothing would look like the gate passing. + """ + raw = str(text).strip() + body = raw[1:] if raw.startswith("-") else raw + parts = body.rsplit("-", 1) + if len(parts) != 2: + raise ValueError(f"expected START-END, got {text!r}") + head = ("-" + parts[0]) if raw.startswith("-") else parts[0] + start, end = float(head), float(parts[1]) + if end <= start: + raise ValueError(f"region {text!r} does not end after it starts") + return start, end + + +def apply_acceptances(dropped, accepted): + """Split dropped regions into (still failing, accepted) (pure). + + A dropped region is only cleared when an acceptance FULLY covers it. + Partial overlap leaves it failing: the creator signed off on what they + listened to, and a region that extends past that is something else. + """ + if not accepted: + return list(dropped), [] + remaining, cleared = [], [] + for d in dropped: + cover = next((a for a in accepted + if a["start"] <= d["start"] and d["end"] <= a["end"]), + None) + if cover is None: + remaining.append(d) + else: + record = dict(d) + record["accepted_by"] = {"start": cover["start"], + "end": cover["end"], + "reason": cover["reason"]} + cleared.append(record) + return remaining, cleared + + +def build_report(transcript, audio_map, wpm=None, min_drop=DEFAULT_MIN_DROP, + cluster_gap=DEFAULT_CLUSTER_GAP, + floor_ratio=DEFAULT_WPM_FLOOR_RATIO, accepted=None): + """Run every check and assemble the verdict (pure).""" + words = transcript.get("words", []) + duration = float(audio_map["duration"]) + silence = audio.to_pairs(audio_map.get("silence", [])) + speech_seconds = float(audio_map.get("speech_seconds", 0.0)) + + found = find_dropped(silence, words, duration, min_drop, cluster_gap) + dropped, cleared = apply_acceptances(found, accepted) + rate = word_rate(words, duration, speech_seconds, wpm, floor_ratio) + low_conf = low_confidence_runs(words) + + checks = { + "coverage": { + "ok": not dropped, + "dropped_regions": len(dropped), + "dropped_seconds": r2(sum(d["audible"] for d in dropped)), + "accepted_regions": len(cleared), + }, + "word_rate": {"ok": rate["ok"]}, + "confidence": {"ok": True, "low_confidence_runs": len(low_conf)}, + } + return { + "ok": checks["coverage"]["ok"] and checks["word_rate"]["ok"], + "map_warning": map_too_coarse(audio_map, min_drop), + "media": audio_map.get("media"), + "duration": r2(duration), + "speech_seconds": r2(speech_seconds), + "checks": checks, + "dropped_regions": dropped, + "accepted": cleared, + "word_rate": rate, + "low_confidence": low_conf, + } + + +def _load(path, label): + try: + with open(path, encoding="utf-8") as f: + return json.load(f) + except (OSError, json.JSONDecodeError) as e: + print(f"verify_transcript: cannot read {label} {path}: {e}", + file=sys.stderr) + return None + + +def main(argv=None): + p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + p.add_argument("words", help="path to transcript/words.json") + p.add_argument("--audio-map", required=True, + help="path to cut/audio-map.json (from analyze_audio.py)") + p.add_argument("-o", "--output", default=None, + help="optional path to write the full report JSON") + p.add_argument("--wpm", type=float, default=None, + help="creator's configured words per minute ([owner] wpm)") + p.add_argument("--min-drop", type=float, default=DEFAULT_MIN_DROP, + help=f"seconds of audible-but-untranscribed audio that " + f"count as dropped speech (default {DEFAULT_MIN_DROP})") + p.add_argument("--cluster-gap", type=float, default=DEFAULT_CLUSTER_GAP, + help=f"merge uncovered runs closer than this (default " + f"{DEFAULT_CLUSTER_GAP}s)") + p.add_argument("--wpm-floor-ratio", type=float, + default=DEFAULT_WPM_FLOOR_RATIO, + help=f"fraction of configured wpm the speech-time rate " + f"must reach (default {DEFAULT_WPM_FLOOR_RATIO})") + p.add_argument("--accept-region", action="append", metavar="START-END", + help="acknowledge a flagged region the creator has " + "LISTENED TO and confirmed carries no lost speech " + "(a laugh, a music bed, an off-mic aside). Repeatable. " + "Requires --reason. The acceptance must fully cover " + "the flagged region, and is recorded in the report") + p.add_argument("--reason", default=None, + help="why the accepted regions are not dropped speech; " + "required with --accept-region so the override is " + "auditable rather than silent") + args = p.parse_args(argv) + + accepted = [] + if args.accept_region: + if not args.reason or not args.reason.strip(): + print("verify_transcript: --accept-region requires --reason. An " + "unexplained override of the completeness gate is exactly " + "the failure this gate exists to catch.", file=sys.stderr) + return 2 + for text in args.accept_region: + try: + start, end = parse_region(text) + except ValueError as e: + print(f"verify_transcript: bad --accept-region: {e}", + file=sys.stderr) + return 2 + accepted.append({"start": start, "end": end, + "reason": args.reason.strip()}) + elif args.reason: + print("verify_transcript: --reason given with no --accept-region", + file=sys.stderr) + return 2 + + transcript = _load(args.words, "transcript") + if transcript is None: + return 2 + audio_map = _load(args.audio_map, "audio map") + if audio_map is None: + return 2 + if "words" not in transcript: + print("verify_transcript: transcript has no 'words' key", + file=sys.stderr) + return 2 + if "silence" not in audio_map or "duration" not in audio_map: + print("verify_transcript: audio map missing 'silence'/'duration'; " + "regenerate it with analyze_audio.py", file=sys.stderr) + return 2 + + report = build_report(transcript, audio_map, wpm=args.wpm, + min_drop=args.min_drop, + cluster_gap=args.cluster_gap, + floor_ratio=args.wpm_floor_ratio, + accepted=accepted) + + if args.output: + out = Path(args.output) + out.parent.mkdir(parents=True, exist_ok=True) + out.write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n", + encoding="utf-8") + + print(json.dumps({k: report[k] for k in + ("ok", "duration", "speech_seconds", "checks")}, + indent=2)) + + if report["map_warning"]: + print(report["map_warning"], file=sys.stderr) + + if report["low_confidence"]: + print(f"note: {len(report['low_confidence'])} low-confidence run(s); " + "advisory only, listed in the report", file=sys.stderr) + + for a in report["accepted"]: + print(f"note: flagged region at {a['at']} accepted by the creator " + f"({a['accepted_by']['reason']})", file=sys.stderr) + + if report["ok"]: + return 0 + + print("\nTRANSCRIPT INCOMPLETE. Do NOT build candidates against it.", + file=sys.stderr) + for d in report["dropped_regions"]: + print(f" dropped speech at {d['at']} " + f"({d['start']}s to {d['end']}s, {d['audible']}s audible, " + "no transcribed words)", file=sys.stderr) + if not report["word_rate"]["ok"]: + wr = report["word_rate"] + print(f" word rate {wr['speech_wpm']} wpm over speech time is below " + f"the {wr['floor_wpm']} floor " + f"({wr['configured_wpm']} configured x {wr['floor_ratio']})", + file=sys.stderr) + print("\nRemediation: re-transcribe the named regions in isolation and " + "splice, or re-run the take with a smaller transcribe.py --window " + "(20s is the validated default; anything larger drops speech).", + file=sys.stderr) + return 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/mc-graphics/SKILL.md b/skills/mc-graphics/SKILL.md index 432a453..82b26ff 100644 --- a/skills/mc-graphics/SKILL.md +++ b/skills/mc-graphics/SKILL.md @@ -1,33 +1,63 @@ --- name: mc-graphics -description: Execute the approved beat table in HyperFrames/OGraf/HTML, render, frame-verify, and deliver alpha overlays plus a HANDOFF for the creator's editor. Use at the graphics stage, only after gate 3 (beats) is approved. +description: Render the beat table into alpha overlays. Use at the graphics stage after gate 3, or when the user says "build the graphics" or "render the overlays". --- # mc-graphics -## Steps +The approved beat table comes in; rendered overlays go out. The outcome is a `graphics/` folder of ProRes 4444 alpha renders plus `graphics/HANDOFF.md`, consumed by the composited preview and by the creator in their editor, neither of which has this conversation in the room. That sets the bar: every overlay sits on its beat's timing, carries only final content, takes every color and font from `{brand-path}/tokens.json`, and has passed both `render_verify.py` and your own eyes before the creator sees it. This is the expensive stage, so nothing here is a draft. -1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there) and this skill's own surface (`uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}`; run `{workflow.activation_steps_prepend}` now, `{workflow.activation_steps_append}` after this step, and hold `{workflow.persistent_facts}` as standing context). Resolve `paths` values against `{project-root}`. Read `project.json` (confirm `approvals.beats` is a date, stage `graphics`), `beats/beats.md`, `beats/STORYBOARD.md`, `{brand-path}/tokens.json`, `{brand-path}/production-bible.md` (the styling contract beyond tokens.json: overlay aesthetic, motion feel, image-type policy, placement rules), the format profile, and `{skill-root}/engines/.md` for each engine the table names. Beats marked OGraf route through the mc-ograf skill, and ONLY if `[editor] ograf-editable = true`; otherwise build them as baked alpha overlays like everything else (baked alpha works in every editor). -2. Engine workspaces live at `{engines-path}//`; initialize on first use per the engine README, installing the latest published version at that moment and recording what it resolved. For HyperFrames, refresh its Agent Skills now (`npx hyperframes init`); mc-setup installs them (step 2b) so their knowledge is live from the beats stage, but install them here if setup was skipped or predates them (`npx skills add heygen-com/hyperframes --all --full-depth`, or `npx hyperframes skills update` for the core set). The skills are the agent's current, self-refreshing knowledge of the engine's authoring patterns and full capability surface, so nothing here transcribes a catalog that would go stale. -3. Source before authoring, always: for each beat, reach first for a fitting HyperFrames block or skill (`npx hyperframes add`, the installed skills, existing brand-themed blocks in the engine workspace) across the whole catalog (captions, transitions, lower thirds, social cards, data viz, VFX, device mockups) and its footage-facing effects (color grading, background removal, HTML-in-Canvas), per `{skill-root}/engines/hyperframes.md`; for simple moves on a finished still (fly-in and fly-out, staged builds), prefer the ffmpeg recipes in `{skill-root}/references/motion-recipes.md`; author from scratch via the html lane (`{skill-root}/engines/html.md`) or the design-prompting loop (`{skill-root}/engines/design-prompting.md`) only when nothing fits. Everything themes through tokens.json, no hardcoded colors or fonts. -4. Build per engine in the project's `graphics/` folder. Follow the loop: edit, lint, preview, draft render (CRF 28), single-frame verify, final render. The shipped toolkit does the mechanical parts: `{skill-root}/scripts/html_to_png.py` (exact-size HTML render, separate `--guides` pass, alpha verify) and `{skill-root}/scripts/snug_frame.py` (native-aspect photo framing). Sound: when a composition calls for a whoosh, hit, chime, or bed (the animation-feel conventions in the Production Bible say when), route through the mc-audio service skill (never reach into its folder) and deliver the wav into `graphics/` next to the overlay it belongs to, with its timing noted in HANDOFF.md. -5. Verify every final render with `uv run {skill-root}/scripts/render_verify.py`, passing expectations explicitly: `--pixfmt` per the delivery target, `--expect-dur` from the beat's dur, `--expect-fps` and `--expect-res` from the format profile, or `--meta` pointing at the comp's meta.json render contract carrying the same keys (extracted frames visually checked over checkerboard for alpha). A render without checked frames is not done. -6. Self-review gate before presenting any batch: zoom-inspect every asset (read every string, check edges and alpha fringes), check each against the Production Bible's aesthetic language, and ask of each one "is this the best you could do?". Fix what fails before the creator sees anything. -7. Write `graphics/HANDOFF.md`: per beat, the rendered file, its timeline position (from the beat table), track suggestion, and any editor notes (OGraf items go in as editable graphics, not MOVs). -8. Trigger the composited preview: `graphics/` now holds rendered overlays, so `renders/preview.mp4` must re-render with them composited. That render belongs to mc-cut, and this skill never runs another skill's scripts: hand back to mc-pipeline, which routes through mc-cut's composited preview re-render (its "Composited preview (after graphics)" section) before the next stage skill runs. The same hand-back applies whenever a later overlay fix re-renders anything in `graphics/`. -9. Update project.json artifacts, advance stage per the profile (usually `assets`, or `package` where assets is absent), and report, naming the composited preview hand-back from step 8 so it is not skipped. +## Resolution rules -## Craft rules +- Bare paths resolve against `{video-path}`, the current video project at `{projects-path}//`. +- `{skill-root}` → this skill's installed directory; files in it always carry it (`{skill-root}/references/motion-recipes.md`). +- `{project-root}` → the project working directory. +- `{skill-name}` → the skill directory's basename. -- Relevance gate: every popup must visibly connect to the words being spoken at its anchor. A viewer pausing at the anchor frame should see WHY this graphic is on screen; a graphic that needs the storyboard to explain it fails the gate. -- Transcript fact-check: any claim-bearing graphic (numbers, dates, quotes, names, titles, announcements) is checked against the transcript verbatim before render. Never invent an announcement, statistic, or quote the speaker did not say. -- Never guess external references: video IDs, channel names, people's names, and spellings are verified against a source or asked about, never guessed. Prefer the project's people-glossary when one exists; when a reference cannot be verified, ask the creator instead of rendering a guess. -- Deliverable images never contain placeholder or helper text: no lorem ipsum, no "TODO", no safe-zone markers, no annotation arrows. Guides go in a separate `--guides` render plus a written spec; the deliverable contains only final content. -- QC sweep: when the creator corrects one asset, treat the correction as a defect CLASS, not a one-off. Audit the whole library for the same defect and fix every instance before re-rendering anything. -- Compositing defaults: never shrink or letterbox the source video to make room for graphics. Composite over the full frame in detected safe zones (find the talking-head region and place around it); photos get snug native-aspect frames, never uniform letterboxed panels. +## On Activation + +1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there). Resolve `paths` values against `{project-root}`. +2. Read `project.json` (confirm `approvals.beats` is a date, stage `graphics`), `beats/beats.md`, `beats/STORYBOARD.md`, the format profile, and `{skill-root}/engines/.md` for each engine the table names. +3. Read `{brand-path}/tokens.json`. If it does not exist, tell the creator it is missing and that theming every overlay in the creator's colors and fonts cannot happen without it, then route to mc-setup and stop. +4. Read `{brand-path}/production-bible.md`. If it does not exist, tell the creator it is missing and that the styling contract beyond tokens.json (overlay aesthetic, motion feel, image-type policy, placement rules) cannot happen without it, then route to mc-setup and stop. +5. Confirm the beat table passed its anchor placement gate: `beats/anchor-check.json` exists and reports `"ok": true`. Missing or failing is a stop, hand back to mc-beats. Every overlay here is positioned by a beat time, so authoring against unverified times spends the expensive stage on graphics that land off their phrases. This skill never runs mc-beats' scripts; it only checks the artifact. + +## Engine workspaces + +Engine workspaces live at `{engines-path}//`; initialize on first use per the engine README, installing the latest published version at that moment and recording what it resolved. For HyperFrames, refresh its Agent Skills now (`npx hyperframes init`); mc-setup installs them, but install them here if setup was skipped or predates them (`npx skills add heygen-com/hyperframes --all --full-depth`, or `npx hyperframes skills update` for the core set). + +## Source before authoring + +For each beat, reach first for a fitting HyperFrames block or installed skill across the whole catalog and its footage-facing effects (`npx hyperframes add`, existing brand-themed blocks in the engine workspace), per `{skill-root}/engines/hyperframes.md`. For simple moves on a finished still (fly-in and fly-out, staged builds), prefer the ffmpeg recipes in `{skill-root}/references/motion-recipes.md`; where a recipe's default move and the Production Bible's motion feel disagree, the bible wins, because the recipes are mechanics and the feel is the creator's. Author from scratch via the html lane (`{skill-root}/engines/html.md`) or the design-prompting loop (`{skill-root}/engines/design-prompting.md`) only when nothing fits. Everything themes through tokens.json; no hardcoded colors or fonts. + +## Build and verify + +Build per engine in the project's `graphics/` folder, running the loop: edit, lint, preview, draft render (CRF 28), single-frame verify, final render. The shipped toolkit does the mechanical parts: `{skill-root}/scripts/html_to_png.py` (exact-size HTML render, separate `--guides` pass, alpha verify) and `{skill-root}/scripts/snug_frame.py` (native-aspect photo framing). + +Verify every final render with `uv run {skill-root}/scripts/render_verify.py`, passing expectations explicitly: `--pixfmt` per the delivery target, `--expect-dur` from the beat's dur, `--expect-fps` and `--expect-res` from the format profile, or `--meta` pointing at the comp's meta.json render contract carrying the same keys. Then look at the extracted frames over checkerboard; a render without checked frames is not done. + +When a composition calls for a whoosh, hit, chime, or bed (the animation-feel conventions in the Production Bible say when), route through the mc-audio service skill, never into its folder, and deliver the wav into `graphics/` next to the overlay it belongs to, with its timing noted in HANDOFF.md. + +## Self-review before the creator sees anything + +Zoom-inspect every asset in the batch: read every string, check edges and alpha fringes, and hold each one against the Production Bible's aesthetic language. Fix what fails before presenting. + +## Handoff and advance + +Write `graphics/HANDOFF.md`: per beat, the rendered file, its timeline position (from the beat table), track suggestion, and any editor notes. + +`graphics/` now holds rendered overlays, so `renders/preview.mp4` must re-render with them composited. That render belongs to mc-cut, and this skill never runs another skill's scripts: hand back to mc-pipeline, which routes through mc-cut's composited preview re-entry before the next stage skill runs. The same hand-back applies whenever a later overlay fix re-renders anything in `graphics/`. + +Then update project.json artifacts, advance stage per the profile (usually `assets`, or `package` where assets is absent), and report, naming the composited preview hand-back so it is not skipped. ## Rules - The beat table is law. A composition that wants different timing goes back through the creator, not silently changed. -- Overlay exports are ProRes 4444 with alpha; anything else is a bug (except OGraf, which ships as its own editable format). +- Overlay exports are ProRes 4444 with alpha; anything else is a bug. +- Relevance gate: a viewer pausing at the anchor frame must see WHY this graphic is on screen. One that needs the storyboard to explain it fails. +- Transcript fact-check: any claim-bearing graphic (numbers, dates, quotes, names, titles, announcements) is checked against the transcript verbatim before render. Never invent an announcement, statistic, or quote the speaker did not say. +- Never guess external references. Video IDs, channel names, people's names, and spellings come from a source, from the project's people-glossary when one exists, or from asking the creator. +- Deliverable images never contain placeholder or helper text: no lorem ipsum, no "TODO", no safe-zone markers, no annotation arrows. Guides go in a separate `--guides` render plus a written spec. +- QC sweep: when the creator corrects one asset, treat the correction as a defect CLASS. Audit the whole library for the same defect and fix every instance before re-rendering anything. +- Never shrink or letterbox the source video to make room for graphics. Composite over the full frame in detected safe zones (find the talking-head region and place around it); photos get snug native-aspect frames, never uniform letterboxed panels. - New reusable compositions get promoted to the engine workspace and noted in the format profile's Templates section. diff --git a/skills/mc-graphics/customize.toml b/skills/mc-graphics/customize.toml deleted file mode 100644 index 8d6d0c4..0000000 --- a/skills/mc-graphics/customize.toml +++ /dev/null @@ -1,20 +0,0 @@ -# DO NOT EDIT -- overwritten on every update. -# -# Customization surface for mc-graphics. -# Override files (not edited here): -# {project-root}/_bmad/custom/mc-graphics.toml (team) -# {project-root}/_bmad/custom/mc-graphics.user.toml (personal) -# -# Studio-wide config (owner, paths, editor, transcription, assets, tools) -# lives in [modules.manticore] in {project-root}/_bmad/custom/config.toml, -# maintained by mc-setup. This file holds only this skill's own surface. - -[workflow] - -# Steps to run before / after the standard activation. -activation_steps_prepend = [] -activation_steps_append = [] - -# Persistent facts held for the whole run: literal sentences, or -# "file:{project-root}/..." paths whose contents are loaded as facts. -persistent_facts = [] diff --git a/skills/mc-graphics/engines/design-prompting.md b/skills/mc-graphics/engines/design-prompting.md index e9f899f..16abf59 100644 --- a/skills/mc-graphics/engines/design-prompting.md +++ b/skills/mc-graphics/engines/design-prompting.md @@ -1,18 +1,14 @@ # Engine: Design Prompting -The authoring path when no registry block or brand template fits a beat: hand the beat to a capable design model as a structured design brief, iterate on the look in a review surface, then render the agreed look deterministically. This pattern works with any design-capable model or surface; Claude (Claude Design, Artifacts, and Claude Code) is used as the worked example throughout because it is the reference implementation the pattern was proven on. +The authoring path when no registry block or brand template fits a beat: hand the beat to a capable design model as a structured design brief, iterate on the look in a review surface, then render the agreed look deterministically. Any design-capable model or surface works; Claude (Claude Design, Artifacts, and Claude Code) is the worked example throughout. ## The core rule -Design surfaces handle look and iteration, but pixels always come from deterministic frame-stepped rendering (HyperFrames render, or headless-Chrome frame stepping plus ffmpeg to ProRes 4444 alpha), never realtime screen recording. - -Realtime capture gives no alpha channel, a variable frame rate, and dropped frames. Every deliverable render walks frames deterministically: +Design surfaces handle look and iteration, but pixels always come from deterministic frame-stepped rendering, never realtime screen recording. Realtime capture gives no alpha channel, a variable frame rate, and dropped frames. A design surface's live preview, a hosted review page, and any screen recording are review surfaces only, never deliverables. - HyperFrames path: `npx hyperframes render` to ProRes 4444 (yuva444p10le) MOV for the editor lane, and a second render of the same comp to VP9 (yuva420p) WebM for the OBS/live lane. One comp, two renders, never two source comps. - HTML path: a headless-Chrome harness seeks the animation to frame N, screenshots with a transparent background (`omitBackground: true` gives PNG alpha), then `ffmpeg -framerate {fps} -i frame_%05d.png -c:v prores_ks -profile:v 4444 -pix_fmt yuva444p10le overlay.mov`. -A design surface's live preview, a hosted review page, and any screen recording are review surfaces only, never deliverables. - ## The determinism contract (required in every brief) Designs authored in a chat or design surface default to wall-clock CSS animations that cannot be seeked. The contract that makes an HTML comp renderable: @@ -26,7 +22,7 @@ HyperFrames beats satisfy the same contract natively: the comp is plain HTML/CSS ## The design brief -The brief is a file (`graphics/briefs/.md`), reviewable and diffable. It packs everything the design model needs so nothing is left to guessing. Omissions are where renders fail. Template: +The brief is a file (`graphics/briefs/.md`), reviewable and diffable. Omissions are where renders fail. Template: ```markdown # Design brief: (one-line description) @@ -61,7 +57,8 @@ No paraphrasing, no added copy, no lorem ipsum, no watermarks. - Feel: {3-5 adjectives from the bible's animation section} - Overlay aesthetic: {surface treatment, radius, shadow/glow per the bible} -- Ease: tokens motion easeDefault for entrances, easeEmphasis for the hit. +- Ease and duration: tokens motion `easeDefault` for entrances, `easeEmphasis` + for the hit, `durationBaseMs` for the main move, `durationFastMs` for accents. - In/out: animate in over {x}ms, hold, resolve out over {y}ms fully inside the beat duration (or hold the last frame to a hard cut; pick one). @@ -77,15 +74,15 @@ No paraphrasing, no added copy, no lorem ipsum, no watermarks. - ffprobe: {fps} fps, {WxH}, alpha present after render. ``` -Sources for the brief: the approved beat row (timing, anchor, composition), the transcript excerpt verbatim, `{brand-path}/tokens.json` inlined, the Production Bible's aesthetic and motion language, the format profile's safe zones, and the alpha requirement. Timing comes from the beat table and is never invented or stretched by the design; timing changes route back through the creator as beat-table changes. +Sources: the approved beat row (timing, anchor, composition), the transcript excerpt verbatim, `{brand-path}/tokens.json` inlined, the Production Bible's aesthetic and motion language, the format profile's safe zones, and the alpha requirement. Timing comes from the beat table and is never invented or stretched by the design; timing changes route back through the creator as beat-table changes. ## The iterate loop 1. Brief: generate the brief file from the beat row, tokens, and the Production Bible. 2. Propose: the design model produces a candidate comp (self-contained HTML honoring the seek contract, authored directly as a HyperFrames comp when it is headed for the engine workspace). -3. Render one frame: seek to the anchor frame and render it. Cheap, fast, and catches most misses before any video render. +3. Render one frame: seek to the anchor frame and render it, before any video render. 4. Critique against the bible: check the frame against the Production Bible's aesthetic language, the safe zones, the verbatim text, and alpha over checkerboard. Revise and repeat. -5. Review with the creator: the review surface may be a hosted page (a Claude Artifact is the worked example) showing the animation looping, a scrub slider driving the same `seek(frame)` the renderer will use, a checkerboard toggle to prove alpha, and a composite over a still frame extracted from the actual footage at the beat's start time so safe zones are checked against reality. The creator gives frame-referenced notes ("at 0:00.8 the underline overshoots"); each note feeds the next revision of the local source file, which stays the single source of truth. +5. Review with the creator on a review surface (a Claude Artifact is the worked example) showing the animation looping, a scrub slider driving the same `seek(frame)` the renderer will use, a checkerboard toggle to prove alpha, and a composite over a still frame extracted from the actual footage at the beat's start time, so safe zones are checked against reality. Frame-referenced notes ("at 0:00.8 the underline overshoots") feed the next revision of the local source file, which stays the single source of truth. 6. Render and verify: draft render, then final ProRes 4444, then `{skill-root}/scripts/render_verify.py` (ffprobe pixel format, duration, fps, resolution, extracted frames checked over checkerboard). The verified ProRes 4444 render is the deliverable. A render without checked frames is not done. ## Translating the agreed look into engine code @@ -93,138 +90,21 @@ Sources for the brief: the approved beat row (timing, anchor, composition), the Once the creator approves the look, it becomes durable engine code rather than a one-off: - HyperFrames: port the comp into a themed block in the HyperFrames workspace, all colors and fonts read from tokens, timing parameterized so the block can be reused at other durations. -- OGraf: only when the target supports it (editor lane per `[editor] ograf-editable`, always for the live lane); rebuild the approved look as an OGraf graphic via mc-ograf rather than wrapping the HTML. -- Foreign HTML (exported from a design surface) is sanitized before entering a workspace: strip or inline every external reference, replace hardcoded colors and fonts with token references (a grep for hex literals not present in tokens is the lint), retrofit the seek contract, and double-render a frame to verify determinism. +- Foreign HTML exported from a design surface is sanitized before entering a workspace: strip or inline every external reference, replace hardcoded colors and fonts with token references (a grep for hex literals not present in tokens is the lint), retrofit the seek contract, and double-render a frame to verify determinism. - Record promoted blocks in the format profile's Templates section so future beats assemble them instead of redesigning. Reusable ffmpeg motion primitives (fly-in and fly-out with optional whoosh, staged infographic builds) live at `{skill-root}/references/motion-recipes.md`; prefer them for simple moves before invoking the full design loop. -## Optional pattern: brand look-dev with a design-system surface - -There is a second, upstream lane this module documents but does not depend on: pushing the brand kit (tokens plus preview cards for the reusable overlay family) into a design-system surface so the creator iterates on the brand's motion language visually, outside any specific video. With Claude the mechanism is DesignSync from Claude Code into a claude.ai/design design-system project; exports come back through the same sanitize-on-import pass above. This look-dev lane is an optional pattern only: the shipped 1.0 workflow is the per-beat design loop, and nothing in this engine requires a design-system surface to exist. - -## Failure modes and guardrails - -- Timing drift: designs love to breathe longer than the beat. The brief states timing is law; the verifier checks duration; changes route through the creator. -- Alpha leaks: subtle full-frame gradients kill overlays. The checkerboard toggle and extracted-frame checks catch this. -- Token drift: hardcoded colors from a design tool. Grep-lint on import. -- Text mutation: models paraphrase. Briefs mark text verbatim; the anchor-frame check includes reading the text. -- Review-vs-render confusion: the hosted preview is never the deliverable; the local frame-stepped render is. - -## Worked example briefs - -### Example 1: keyword callout (lower-third emphasis) - -Transcript moment: at 03:12.4 the speaker says "the transcript IS the timeline: every cut and every graphic anchors to a word." - -```markdown -# Design brief: b07-transcript-is-timeline - -## Beat - -start 03:12.0 / dur 4.5s / end 03:16.5 (135 frames @ 30fps, 1920x1080) -anchor word: "timeline" at 03:13.1; underline hit lands ON "timeline" -spoken phrase: "the transcript IS the timeline" -## Exact text - -1. "THE TRANSCRIPT IS THE TIMELINE" -## Canvas and alpha - -Transparent. Speaker occupies the right third; graphic lives lower-left, -inside 5% title-safe. Nothing crosses x > 60% of frame width. -## Brand - - Accent underline uses the accent color; text in the -primary text color on a surface-color chip at 85% opacity. -## Look and motion feel - -Confident, snappy, engineered (per the bible). Chip slides up 24px and -fades in over durationBaseMs (easeDefault); accent underline draws -left-to-right over durationFastMs, timed to hit full width at the anchor -(easeEmphasis); the whole unit resolves out (slide down and fade) in the -final 400ms. -## Determinism contract / Acceptance - - -``` - -### Example 2: animated diagram that builds as the speaker names each stage - -Transcript moment: 07:40-07:58, the speaker walks the pipeline: "brain dump... outline... script... cut... graphics." Word timestamps from the transcript give one reveal per named stage. - -```markdown -# Design brief: b11-pipeline-build - -## Beat - -start 07:40.0 / dur 18.0s / end 07:58.0 (540 frames @ 30fps, 1920x1080) -anchors: "brain dump" 07:41.2, "outline" 07:44.8, "script" 07:48.1, -"cut" 07:51.9, "graphics" 07:55.0; each node and connector reveals ON its word -## Exact text - -Nodes, in order: "BRAIN DUMP", "OUTLINE", "SCRIPT", "CUT", "GRAPHICS" -## Canvas and alpha - -Transparent. Diagram occupies the upper 55% of frame, centered; the -speaker is lower-center. Inside title-safe. -## Brand - - Nodes: surface fill, border stroke, primary text -labels; connectors and the active-node glow use the accent color. -## Look and motion feel - -Calm, systematic, additive. Each node scales 0.92 to 1.0 and fades in over -durationBaseMs (easeDefault); the connector draws toward the next node over -durationFastMs; previously revealed nodes dim to 70% so the current one -reads. The fully built diagram holds from 07:56 and holds the last frame -to the cut. -## Determinism contract / Acceptance - - -``` - -### Example 3: cold-open title stinger (HyperFrames, dual render) +## Three traps this lane sets -Transcript moment: the approved hook line opens the video; the stinger also serves as the live scene transition. +- A subtle full-frame gradient reads as a design flourish and kills the overlay, because it + makes every pixel opaque. The checkerboard toggle and the extracted-frame check catch it. +- Models paraphrase text they are asked to render, so read the strings back off the + anchor frame. +- The hosted preview is never the deliverable. The local frame-stepped render is, and the + two can disagree. -```markdown -# Design brief: b00-title-stinger (engine: HyperFrames) - -## Beat - -start 00:00.0 / dur 1.6s / end 00:01.6 (48 frames @ 30fps, 1920x1080) -(keep stingers 1 to 2 seconds; transparent WebM renders slowly) -## Exact text - -1. "{approved video title, verbatim from the packaging promise}" -## Canvas and alpha - -Transparent throughout; logo and title only, no backdrop. Center-weighted, -inside 5% title-safe, clear of the lower-third caption zone. -## Brand +## Two variants of the brief template - Logo asset inlined as a data URI; title in the -heading font at weight 700, primary text color; accent sweep in the -accent color. -## Look and motion feel - -Kinetic, premium, over fast. The logo mark snaps in with an easeEmphasis -scale-settle (no bounce past 1.02); the accent sweep wipes behind the -title as it tracks in; everything exits with a fast upward wipe in the -last 12 frames so the footage is revealed clean. -## Determinism contract - -HyperFrames comp, 48 frames at 30fps; one GSAP timeline created -{ paused: true } and registered on window.__timelines; no unseeded -randomness. -## Deliverables - -ONE comp, TWO renders: ProRes 4444 (yuva444p10le) MOV for the editor, -VP9 yuva420p WebM for the live lane. Never two source comps. -## Acceptance - -Frame 0 and frame 47 fully transparent; ffprobe confirms alpha in both -outputs; title text verified verbatim at the hold frame. -``` +- Multi-anchor beats (a diagram that builds as the speaker names each stage) list every anchor with its timestamp in the Beat block, and their acceptance is a per-anchor frame check: exactly k elements visible at anchor k. +- A HyperFrames stinger brief adds a Deliverables block (ONE comp, TWO renders: ProRes 4444 MOV for the editor, VP9 yuva420p WebM for the live lane) and keeps to 1 to 2 seconds, because transparent WebM renders slowly. diff --git a/skills/mc-graphics/engines/html.md b/skills/mc-graphics/engines/html.md index 5905943..4a324af 100644 --- a/skills/mc-graphics/engines/html.md +++ b/skills/mc-graphics/engines/html.md @@ -1,25 +1,25 @@ # Engine: HTML -The html-design lane: model-authored HTML/CSS, themed from `{brand-path}/tokens.json` and the Production Bible, rendered to pixels by `{skill-root}/scripts/html_to_png.py`, screenshot-reviewed, iterated. This is the fastest path for static and single-frame graphics (popups, callout cards, framed imagery, infographic frames). Animated comps use the same authoring surface but must honor the `window.seek(frame)` determinism contract in `engines/design-prompting.md` and render frame-stepped to ProRes 4444. +The html-design lane: model-authored HTML/CSS, themed from `{brand-path}/tokens.json` and the Production Bible, rendered to pixels by `{skill-root}/scripts/html_to_png.py`, screenshot-reviewed, iterated. The fastest path for static and single-frame graphics (popups, callout cards, framed imagery, infographic frames). Animated comps use the same authoring surface but must honor the `window.seek(frame)` determinism contract in `{skill-root}/engines/design-prompting.md` and render frame-stepped to ProRes 4444. ## The loop -1. Author a single self-contained HTML file in the project's `graphics/` folder: no external requests of any kind (CDN scripts, fonts, images); inline or data-URI everything. Every color and font comes from `tokens.json`; the surface treatment, radius, shadow, and placement come from the Production Bible. Exact text is verbatim from the beat row or transcript. +1. Author a single self-contained HTML file in the project's `graphics/` folder: no external requests of any kind (CDN scripts, fonts, images); inline or data-URI everything. Every color and font comes from `{brand-path}/tokens.json`; surface treatment, radius, shadow, and placement come from the Production Bible. Exact text is verbatim from the beat row or transcript. 2. Render at the exact target size: `uv run {skill-root}/scripts/html_to_png.py graphics/.html --out graphics/.png --width {W} --height {H} --verify-alpha`. The script fails if the PNG is not exactly the requested pixel size. -3. Review the rendered PNG itself (not the HTML in your head): open it, zoom, read every string, check alpha over the `_checker.png` composite. Critique against the Production Bible's aesthetic language. +3. Review the rendered PNG itself, not the HTML in your head: open it, zoom, read every string, check alpha over the `_checker.png` composite, critique against the Production Bible's aesthetic language. 4. When placement matters, render a SEPARATE guides pass with `--guides graphics/_guides.png` (safe-zone insets plus crosshair over a checkerboard). Guides and helper text never appear in the deliverable image. -5. Revise the HTML and re-render until it passes the self-review gate, then hand the PNG (or the frame-stepped video render, for animated comps) to the compositing step. +5. Revise and re-render until it passes the self-review gate, then hand the PNG (or the frame-stepped video render, for animated comps) to the compositing step. ## Rendering rules -- Deliverables are rendered transparent (the default): unpainted pixels carry alpha 0 so the graphic composites over full-frame video. Pass `--no-transparent` only for opaque cards that are meant to fill their canvas. +- Deliverables are rendered transparent (the default): unpainted pixels carry alpha 0 so the graphic composites over full-frame video. Pass `--no-transparent` only for opaque cards meant to fill their canvas. - `--scale 2` doubles the device pixel ratio for crisp downstream scaling; the output size check accounts for it. - Photos inside HTML comps still obey the snug-frame rule: size the frame to the photo's native aspect (use `{skill-root}/scripts/snug_frame.py` for standalone framed photos), never a uniform letterboxed panel. - A render is not done until the PNG has been visually checked, and video renders additionally pass `{skill-root}/scripts/render_verify.py`. ## Brand fonts: the fontconfig shim pattern -The portable lane is to inline brand fonts into the HTML as data-URI `@font-face` rules (this also satisfies the self-contained rule, and it is the only reliable path for Chromium on macOS, which resolves fonts through CoreText, not fontconfig). +Inline brand fonts into the HTML as data-URI `@font-face` rules. This also satisfies the self-contained rule, and it is the only reliable path for Chromium on macOS, which resolves fonts through CoreText, not fontconfig. For renderers that resolve fonts through fontconfig (rsvg-convert, ffmpeg drawtext, Chromium on Linux) and brand fonts that are not system-installed, use the shim: write a `fonts.conf` beside the brand fonts and point `FONTCONFIG_FILE` at it for the render invocation. `html_to_png.py` inherits the variable from its environment. diff --git a/skills/mc-graphics/engines/hyperframes.md b/skills/mc-graphics/engines/hyperframes.md index e9c012c..f378ed0 100644 --- a/skills/mc-graphics/engines/hyperframes.md +++ b/skills/mc-graphics/engines/hyperframes.md @@ -1,42 +1,35 @@ # Engine: HyperFrames -Default engine for everything the pipeline renders as motion: per-video overlay beats, stingers, karaoke captions, and footage-facing effects. Apache 2.0, fully local, free. `npx hyperframes render` drives HTML/CSS/GSAP frame-by-frame in headless Chrome and exports overlay-only ProRes 4444 MOV with alpha (their docs recommend exactly this for Resolve workflows). Everything the pipeline uses is the local CLI and render path. The hosted conveniences (HeyGen cloud-render credits, the Studio web app, Claude Design, Figma import, AWS Lambda / Cloud Run) are never a dependency, and no HeyGen account is required. +Default engine for everything the pipeline renders as motion: per-video overlay beats, stingers, karaoke captions, and footage-facing effects. Apache 2.0, fully local, free. `npx hyperframes render` drives HTML/CSS/GSAP frame-by-frame in headless Chrome and exports overlay-only ProRes 4444 MOV with alpha. Everything the pipeline uses is the local CLI and render path; the hosted conveniences (HeyGen cloud-render credits, the Studio web app, Claude Design, Figma import, AWS Lambda / Cloud Run) are never a dependency, and no HeyGen account is required. -## The skills are the source of truth (install them, favor them) +## The skills are the source of truth -HyperFrames is pre-1.0 and moves fast, so this file never transcribes its full capability list or pins a version, which would only ship stale knowledge to every creator. Instead the agent loads HyperFrames' own Agent Skills, which teach the current authoring patterns (the `data-*` attributes, GSAP timeline registration, the component vocabulary, every effect) and refresh themselves: +HyperFrames is pre-1.0 and moves fast, so anything transcribed here ships stale. Load HyperFrames' own Agent Skills instead, which teach the current authoring patterns (the `data-*` attributes, GSAP timeline registration, the component vocabulary, every effect) and refresh themselves. -- Installed at setup (mc-setup step 2b), so the whole capability surface is known from the beats stage onward, not just at graphics time: `npx skills add heygen-com/hyperframes --all --full-depth` loads the whole catalog (or `npx hyperframes skills update` for the maintained core set). Favor them: the installed skills plus the block catalog are the FIRST place to look for any beat, ahead of authoring anything by hand. -- Refreshed on every graphics run, never blindly expanded: `npx hyperframes init` refreshes the core set plus whatever is already installed; `npx hyperframes skills check` reports drift and `npx hyperframes skills update` applies it. If setup was skipped, mc-graphics installs them on first use. The engine WORKSPACE itself still initializes lazily on the first graphics run (below); only the lightweight skill knowledge lands at setup. The harness loads the skills by its own skill resolution; nothing here is specific to one agent. -- The authoritative, always-current capability index is https://hyperframes.heygen.com/llms.txt. When a beat needs something, consult the installed skills and that index rather than this file. +- mc-setup installs them (`npx skills add heygen-com/hyperframes --all --full-depth` for the whole catalog, or `npx hyperframes skills update` for the maintained core set), so the capability surface is known from the beats stage onward. mc-graphics installs them on first use if setup was skipped. +- `npx hyperframes init` refreshes the core set plus whatever is already installed; `npx hyperframes skills check` reports drift and `npx hyperframes skills update` applies it. Refresh, never blindly expand. +- When a beat needs something, read the installed skills rather than this file, and fetch https://hyperframes.heygen.com/llms.txt for the always-current capability index when they do not cover it. -## What it can do (reach for these before building from scratch) +The engine WORKSPACE still initializes lazily on the first graphics run (below); only the lightweight skill knowledge lands at setup. -All of the following run locally and free. Names are examples, not the live list (the skills and the index above are canonical); the point is that these categories exist and the agent should reach for them: +## What it can do -- Block catalog (`npx hyperframes add`, 100+ blocks): code animations (typing, diff, morph, 3D extrude, particle assemble), transitions including WebGL shaders (whip pan, glitch, light leak, iris, vortex, burn), caption styles (karaoke, kinetic slam, neon, gradient, texture-mask), lower thirds, social cards (X, TikTok, Instagram, Reddit, Spotify), data viz (bar and line charts, flowcharts, US/world/choropleth maps), VFX (liquid glass, portal, shatter, news ticker, logo outro), and 3D device showcases (GLTF iPhone/MacBook with live HTML screens). Theme every block through tokens.json. -- Footage-facing media effects (these apply to the video and stills, not only to overlays): color grading with presets, project-local `.cube` LUTs, vignette, grain, blur, and pixelate via a `data-color-grading` attribute; background removal to a transparent overlay; HTML-in-Canvas to run WebGL shaders and 3D geometry over the DOM. These open real pipeline moves without leaving the local renderer: propose a few graded looks over the actual footage and let the creator pick, or cut a subject off its background. -- Delivery reach: renders to ProRes 4444 MOV, VP9 alpha WebM, MP4, GIF, and PNG sequences; HDR10 (BT.2020 PQ or HLG, 10-bit H.265) when the sources are HDR; 4K via the Chrome device scale factor. +All local and free. These are categories to reach for, not the live list (the skills and the index are canonical): + +- Block catalog (`npx hyperframes add`, 100+ blocks): code animations, WebGL shader transitions, caption styles including karaoke, lower thirds, social cards, data viz and maps, VFX, and 3D device showcases with live HTML screens. Theme every block through tokens.json. +- Footage-facing media effects apply to the video and stills, not only to overlays: color grading with presets, project-local `.cube` LUTs, vignette, grain, blur, and pixelate via a `data-color-grading` attribute; background removal to a transparent overlay; HTML-in-Canvas to run WebGL shaders and 3D geometry over the DOM. These open real pipeline moves without leaving the local renderer: propose a few graded looks over the actual footage and let the creator pick, or cut a subject off its background. +- Delivery reach: ProRes 4444 MOV, VP9 alpha WebM, MP4, GIF, and PNG sequences; HDR10 (BT.2020 PQ or HLG, 10-bit H.265) when the sources are HDR; 4K via the Chrome device scale factor. ## Setup -- The engine workspace lives at the creator's `{engines-path}/hyperframes/`, initialized on the first graphics run: install the latest published version at that moment (`npm install hyperframes@latest`) and install the skills (above). Never carry a version number in module docs. Record the resolved engine version in the workspace package.json and upgrade deliberately from there rather than floating mid-project. +- The engine workspace lives at the creator's `{engines-path}/hyperframes/`, initialized on the first graphics run: `npm install hyperframes@latest`, plus the skills above. Never carry a version number in module docs. Record the resolved engine version in the workspace package.json and upgrade deliberately rather than floating mid-project. - Pull the blocks a beat needs before authoring: `npx hyperframes add`. Theme every block through `{brand-path}/tokens.json`. -- `@hyperframes/studio` is the local timeline GUI with real bidirectional HTML sync (drag a beat in the GUI, the code updates; hand-edit the code, the GUI hot-reloads). Use it for timing nudges after mc-graphics gets close. The hosted Studio preview and Claude Design are optional and never required. - -## Why this engine and not Remotion - -Remotion is not used. Two reasons, recorded here so the decision is not relitigated per video: - -- License. Remotion is free only for companies of up to 3 people; past that it needs a paid Company License (per-seat, or per-render with a monthly minimum). Manticore is a distributed module, so shipping Remotion would hand every creator at a 4+ person company a licensing obligation they did not opt into. HyperFrames is Apache 2.0 with no commercial-use threshold. -- React bought nothing. Remotion's remaining justification was "anything React-stateful", but in a frame-deterministic renderer state IS a function of frame index. Every job it might hold here (the dual-render brand stinger, word-level karaoke captions, plain HTML/SVG comps) is a paused GSAP timeline in HyperFrames, and the beat table drives it identically. One engine means one authoring model and one thing for the agent to know. - -Remotion remains the stronger pick for React shops rendering at massive scale. That is not this pipeline. +- `@hyperframes/studio` is the local timeline GUI with real bidirectional HTML sync (drag a beat in the GUI, the code updates; hand-edit the code, the GUI hot-reloads). Use it for timing nudges after mc-graphics gets close. ## Jobs - Per-video overlay beats: the default lane, delivered as ProRes 4444 alpha MOVs. -- Brand stinger and transitions: ONE composition rendered twice, VP9 yuva420p WebM for OBS and ProRes 4444 MOV for the editor lane. Keep it 1 to 2 seconds (transparent WebM renders slowly). HyperFrames captures each frame as PNG with alpha and encodes through either alpha-capable codec, so the dual target is one comp and two renders. +- Brand stinger and transitions: ONE composition rendered twice, VP9 yuva420p WebM for OBS and ProRes 4444 MOV for the editor lane. Keep it 1 to 2 seconds; transparent WebM renders slowly. - Shorts karaoke captions: word-level highlight driven by the transcript's word timestamps, built from a registry caption block themed through tokens.json. - Footage-facing effects: color-graded looks and background removal on the source video and stills, when a beat or the Production Bible calls for them. diff --git a/skills/mc-graphics/references/motion-recipes.md b/skills/mc-graphics/references/motion-recipes.md index 3cdca21..37ca75c 100644 --- a/skills/mc-graphics/references/motion-recipes.md +++ b/skills/mc-graphics/references/motion-recipes.md @@ -9,9 +9,8 @@ The input stills come from wherever the graphic was authored: `{skill-root}/scri - Input PNGs must carry real alpha. A PNG whose alpha channel is all zero or all opaque full-frame produces an invisible or frame-covering overlay; `html_to_png.py` verifies this at export. - Every `overlay` filter in these recipes passes `format=auto`. The overlay filter's default working format is `yuv420`, which silently drops alpha; the encoder then re-adds a fully opaque alpha plane and the "overlay" covers the whole frame. - The filtergraph ends with `format=yuva444p10le` before `prores_ks`, and the output is always ProRes 4444 (`-c:v prores_ks -profile:v 4444 -pix_fmt yuva444p10le`), per this skill's deliverable rule. -- Parameters come from the project, never from these recipes: canvas size and fps from the format profile, the beat duration and timeline position from the approved beat table, motion durations and easing feel from the `tokens.json` motion values and the Production Bible. Timing is law; a move that wants a longer beat routes back through the creator. +- Parameters come from the project, never from these recipes: canvas size and fps from the format profile, the beat duration and timeline position from the approved beat table, motion durations and easing feel from the `{brand-path}/tokens.json` motion values and the Production Bible. Timing is law; a move that wants a longer beat routes back through the creator. - Easing uses cubic curves in the overlay position expression: ease-out is `1-pow(1-p,3)` and ease-in is `pow(p,3)`, where `p` is the normalized progress `(t-start)/dur` of that move. -- Verify every render with `render_verify.py` before calling it done (see the last section). ## Recipe 1: fly-in, hold, fly-out @@ -27,7 +26,7 @@ ffmpeg -y \ -t {dur} -r {fps} -c:v prores_ks -profile:v 4444 -pix_fmt yuva444p10le graphics/.mov ``` -Worked example, validated end to end (1920x1080 at 30 fps, 4.5 s beat, 0.6 s fly-in from the left, 0.4 s fly-out to the right, rest at 120,780): +Worked example (1920x1080 at 30 fps, 4.5 s beat, 0.6 s fly-in from the left, 0.4 s fly-out to the right, rest at 120,780): ```bash ffmpeg -y \ @@ -37,14 +36,9 @@ ffmpeg -y \ -t 4.5 -r 30 -c:v prores_ks -profile:v 4444 -pix_fmt yuva444p10le graphics/.mov ``` -Reading the x expression: before `{in}` the PNG travels from fully off-screen left (`-w`) to `{X}` on an ease-out; until `{hold-end}` it rests at `{X}`; then it accelerates to `W` (off-screen right) on an ease-in. Frame 0 is fully transparent because the graphic starts entirely off-canvas. +Frame 0 is fully transparent because the graphic starts entirely off-canvas (`-w`). -Variants: - -- Fly in from the right: swap the entrance target arithmetic, `x='if(lt(t,{in}), W-(W-{X})*(1-pow(1-t/{in},3)), ...)'`. -- Vertical moves: put the motion expression on `y` and fix `x`, entering from `-h` (top) or `H` (bottom). -- Enter and exit on the same side: reuse the entrance arithmetic with the ease-in curve for the exit. -- Hold to a hard cut: drop the third branch and let the graphic rest until `{dur}` ends. +Variants keep the same three-branch shape: enter from the right by starting off-screen right, `x='if(lt(t,{in}), W-(W-{X})*(1-pow(1-t/{in},3)), ...)'`; move vertically by putting the expression on `y` and fixing `x`, entering from `-h` or `H`; exit on the entrance side by reusing the entrance arithmetic with the ease-in curve; hold to a hard cut by dropping the third branch. ## Recipe 2: staged infographic build @@ -52,7 +46,7 @@ An infographic that assembles as the speaker names each part, one reveal per anc Parameters per layer `k`: `{Tk}` the layer's anchor time in seconds measured from the beat's start (anchor ts minus beat start, from the approved beat table). -Worked example, validated end to end (three layers at 1.2 s, 4.8 s, 8.1 s inside a 12 s beat, 1920x1080 at 30 fps): +Worked example (three layers at 1.2 s, 4.8 s, 8.1 s inside a 12 s beat, 1920x1080 at 30 fps): ```bash ffmpeg -y \ @@ -62,7 +56,7 @@ ffmpeg -y \ -t 12 -r 30 -c:v prores_ks -profile:v 4444 -pix_fmt yuva444p10le graphics/.mov ``` -Per layer, the pattern is `fade=t=in:st={Tk}:d=0.3:alpha=1` (fully transparent before its anchor, faded in 0.3 s after) plus a y expression that eases the layer from a 24 px offset to 0 over the same 0.3 s. Add or remove `[k] ... [lk]` chains and matching `overlay` links to change the layer count. Verify one extracted frame per anchor: exactly k layers visible at anchor k. +Per layer the pattern is `fade=t=in:st={Tk}:d=0.3:alpha=1` plus a y expression easing the layer from a 24 px offset to 0 over the same 0.3 s. Change the layer count by adding or removing matching `[k] ... [lk]` chains and `overlay` links. ## Recipe 3: whoosh SFX sidecar @@ -76,7 +70,7 @@ ffmpeg -y -i whoosh.wav \ -ar 48000 -c:a pcm_s16le graphics/-sfx.wav ``` -Validated example: a fly-in starting 1.2 s into a 12 s beat used `adelay=1200|1200,apad=whole_dur=12` and produced an exactly 12.0 s wav with the whoosh landing at 1.2 s. +Worked example: a fly-in starting 1.2 s into a 12 s beat uses `adelay=1200|1200,apad=whole_dur=12` and produces an exactly 12.0 s wav with the whoosh landing at 1.2 s. ## Verify before done diff --git a/skills/mc-new/SKILL.md b/skills/mc-new/SKILL.md index 42dc051..0a0ff42 100644 --- a/skills/mc-new/SKILL.md +++ b/skills/mc-new/SKILL.md @@ -1,33 +1,53 @@ --- name: mc-new -description: Scaffold a new Manticore video project from a format profile, idea-first or footage-first. Use when the creator greenlights an idea ("new video", "start a project", "let's make the X video") or wants a video built from existing footage ("cut this VOD", "make a video from this recording"). +description: Scaffold a video project from a format profile. Use when the user says "new video", "start a project", "cut this VOD", or "make a video from this recording". --- # mc-new +Scaffold the project every later stage runs against: `{projects-path}//` holding `project.json` and `brief.md`. `project.json` is the contract every downstream skill reads without this conversation in the room, so the stage list, the mode, and the entry point have to be right at creation. `brief.md` carries the creator's own words and the links back to where the idea came from; nothing later recovers them. + +## Resolution rules + +- Bare paths resolve against `{video-path}`, the current video project at `{projects-path}//`. +- `{skill-root}` → this skill's installed directory; files in it always carry it (`{skill-root}/scripts/new_project.py`). +- `{project-root}` → the project working directory. + +## On Activation + +1. Load the studio config: `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`. Empty means mc-setup has not run: stop and route the creator there. Resolve `paths` values against `{project-root}`. + ## Entry points -A project starts one of two ways; establish which before anything else. +Settle which of the two this is before anything else: it decides the format profile and the next stage. + +- Idea-first (default): the project runs the full pipeline from braindump. Any format profile works. +- Footage-first: the footage already exists (a livestream VOD, a recorded talk), ideation is skipped entirely, the source file is registered in `sources` in `project.json` at creation, and the next stage is cut. This needs a footage-first profile, one whose stage list contains no ideation stages (e.g. livestream-vod). If the studio has none, route the creator to mc-setup to add one rather than forcing an ideation profile. + +## Scaffold -- Idea-first (default): the creator greenlights an idea and the project runs the full pipeline from braindump. Any format profile works. -- Footage-first: the footage already exists (a livestream VOD, a recorded talk, a conference session). The project skips ideation entirely and goes straight to post-production; the source file is registered at creation and the next stage is cut. This requires a footage-first format profile, one whose stage list contains no ideation stages (e.g. livestream-vod). +Collect the slug (kebab-case), the format (a profile in `{formats-path}/`), a working title, and for footage-first the absolute path to the source file. Then ask the two things the creator will not volunteer: -## Steps +- Series: is this an episode of one? An episode lives in a series folder beside a shared `common/` for evergreen assets (chrome, stingers, recurring graphics), and that layout is fixed at creation. +- Deadline: does an external event gate delivery (a conference, a launch)? A date puts the project in deadline mode, where downstream stages cap iteration loops in favor of good-enough delivery. An aspirational date is not a deadline; leave it unset. -1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there) and this skill's own surface (`uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}`; run `{workflow.activation_steps_prepend}` now, `{workflow.activation_steps_append}` after this step, and hold `{workflow.persistent_facts}` as standing context). Resolve `paths` values against `{project-root}`. -2. Establish the entry point, slug (kebab-case), format (must match a profile in `{formats-path}/`), and working title. For footage-first, also get the absolute path to the source footage file; the format must be a footage-first profile, and if none exists yet, route the creator to mc-setup to add one (e.g. livestream-vod) rather than forcing an ideation profile. -3. Ask two scoping questions before scaffolding: - - Series: is this an episode of a series? If yes, get the series slug (kebab-case). Episodes live in a series folder under `{projects-path}` with a shared `common/` folder for evergreen assets (chrome, stingers, recurring graphics), and stages that read brand templates apply per-series packaging. - - Deadline: does an external event gate delivery (a conference, a launch, a scheduled premiere)? If yes, get the date (ISO, YYYY-MM-DD). Recording it puts the project in deadline mode: downstream stages order deliverables by hard external gates and cap iteration loops in favor of good-enough delivery. If nothing external gates delivery, leave it unset; an aspirational date is not a deadline. -4. Run: `uv run {skill-root}/scripts/new_project.py --format --title "" --projects-dir {projects-path} --formats-dir {formats-path}`, adding the flags the answers call for: `--parent <slug>` for a short cut from a long-form parent, `--series <series-slug>` for an episode, `--deadline YYYY-MM-DD` for an event-gated project, and `--ingest <absolute-footage-path>` for footage-first (with `--source-id` and `--source-role primary|interview|screen` when the defaults do not fit). -5. Fill `brief.md`. Idea-first: one paragraph of the idea in the creator's words, why now, and links to source material (idea notes, prior material). Footage-first: what the footage is, what the finished video should become, and any moments the creator already knows matter. Do not invent content; ask if the brief is thin. -6. Report the created project and hand off: idea-first goes to `braindump` (mc-braindump); footage-first goes to `cut` (mc-cut). Trust `stage` in `project.json` either way. +Then run: + +`uv run {skill-root}/scripts/new_project.py <slug> --format <format> --title "<title>" --projects-dir {projects-path} --formats-dir {formats-path}` + +adding the flags the answers call for: `--parent <slug>` for a short cut from a long-form parent, `--series <series-slug>` for an episode, `--deadline YYYY-MM-DD` for an event-gated project, and `--ingest <absolute-footage-path>` for footage-first (with `--source-id` and `--source-role primary|interview|screen` when the defaults do not fit). + +## Brief + +Fill `brief.md` from what the creator gives you, never invented; ask if it is thin. Idea-first: the idea in their words, why now, and links to the source material (idea notes, prior material). Footage-first: what the footage is, what the finished video should become, and any moments they already know matter. + +## Handoff + +Report the created project and route to the skill named by `stage` in `project.json`. Never assume the master stage list. ## Checklist -- Slug is kebab-case and not already taken (within its series folder, if any). -- Format profile exists and its stage list landed in `project.json`. -- brief.md links back to wherever the idea came from so its history stays findable. -- Footage-first: the source file exists on disk, is registered in `sources` in `project.json`, and the stage list contains no ideation stages. -- Series: the project sits in the series folder beside `common/`, and the `series` field is set in `project.json`. -- Deadline: only set when a real external event gates delivery, and it is an ISO date. +- `brief.md` is in the creator's language and links back to wherever the idea came from. +- The deadline field, if set, names a real external gate. + +`new_project.py` exits non-zero on the rest: a non-kebab slug or series, an existing project path, a missing or malformed format profile, an ideation-bearing profile under `--ingest`, a missing footage file, and a non-ISO deadline. diff --git a/skills/mc-new/customize.toml b/skills/mc-new/customize.toml deleted file mode 100644 index 4596a8b..0000000 --- a/skills/mc-new/customize.toml +++ /dev/null @@ -1,20 +0,0 @@ -# DO NOT EDIT -- overwritten on every update. -# -# Customization surface for mc-new. -# Override files (not edited here): -# {project-root}/_bmad/custom/mc-new.toml (team) -# {project-root}/_bmad/custom/mc-new.user.toml (personal) -# -# Studio-wide config (owner, paths, editor, transcription, assets, tools) -# lives in [modules.manticore] in {project-root}/_bmad/custom/config.toml, -# maintained by mc-setup. This file holds only this skill's own surface. - -[workflow] - -# Steps to run before / after the standard activation. -activation_steps_prepend = [] -activation_steps_append = [] - -# Persistent facts held for the whole run: literal sentences, or -# "file:{project-root}/..." paths whose contents are loaded as facts. -persistent_facts = [] diff --git a/skills/mc-ograf/SKILL.md b/skills/mc-ograf/SKILL.md deleted file mode 100644 index 570da71..0000000 --- a/skills/mc-ograf/SKILL.md +++ /dev/null @@ -1,58 +0,0 @@ ---- -name: mc-ograf -description: Design and build broadcast OGraf HTML graphics packages (lower thirds, title cards, straps, bugs) that load correctly in DaVinci Resolve 21+ and OBS/SPX-GC. Use when the user asks for an ograf graphic, an editable lower third for Resolve, or when mc-graphics/mc-stream-pack route a beat here. Only for editors/targets that support OGraf; check config first. ---- - -# mc-ograf - -Act as a broadcast motion-graphics engineer who also designs. The user brings the idea (a lower third, a title card, a news strap, a logo bug); you bring the craft: propose a strong look, decide what should be operator-editable, and ship an OGraf package that a renderer loads on the first try. - -The outcome is a FOLDER (`*.ograf.json` manifest + Web Component `.mjs` + assets + `preview.html`) that loads in DaVinci Resolve 21+, Fusion, CasparCG, or SPX-GC without errors, with correct alpha, and with its reusable values exposed as editable schema fields. - -## Gating (check before doing anything) - -1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there) and this skill's own surface (`uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}`; run `{workflow.activation_steps_prepend}` now, `{workflow.activation_steps_append}` after this step, and hold `{workflow.persistent_facts}` as standing context). Resolve `paths` values against `{project-root}`. -2. OGraf is only the right output when the target supports it: - - Editor lane: `[editor] ograf-editable = true` (DaVinci Resolve 21+). If false, STOP and say so; the same graphic should be built as a baked alpha overlay by mc-graphics instead (baked HyperFrames alpha works in every editor). - - Live lane: OBS/SPX-GC stream graphics (mc-stream-pack). Editor-independent; always allowed. - -## Design the graphic - -When a beat arrives routed from mc-graphics or mc-stream-pack, design from the approved `beats.md`/`STORYBOARD.md` and the format profile instead of re-eliciting; the "get a nod, then build" confirmation below is the only interaction point. Otherwise open the floor: ask what they are making and where it plays, then PROPOSE, don't just transcribe. Offer a concrete look (layout, motion, palette, type) and one or two alternatives, grounded in `{brand-path}/tokens.json` and `{brand-path}/production-bible.md` (the styling contract beyond tokens: overlay aesthetic, motion feel, placement rules; read both before authoring anything). Never invent brand colors from memory; if no tokens exist, ask. Lower thirds anchor bottom-left/center; bugs anchor a corner; full-frame titles center. Suggest entrance/hold/exit timing (a 10s lower third: in over ~1.3s, hold, out over ~0.8s). Sketch the animation in words, get a nod, then build. - -## Decide the config surface - -Walk the design and ask which values an operator should change per use (name, title, strap text, maybe an accent color or logo swap) and which are fixed brand. Each editable value becomes a property in the manifest `schema` (with `title` and `default`), surfaced as an Inspector field in Resolve or an SPX-GC control. Keep the surface tight; every field is operator cognitive load. Record the decided design in a short `design-notes.md` next to the package so a revisit resumes instead of re-eliciting. - -## Build - -Generate with `uv run {skill-root}/scripts/scaffold_ograf.py` rather than hand-writing (it bakes in the structural standards below; the standards that live in your own edits, like deterministic `render(tMs)` and hardened asset loads, remain yours to keep as you build). Pass the palette and font args from `{brand-path}/tokens.json`; the scaffold warns in its JSON output when any are left at the placeholder defaults. Then edit the generated `.mjs` to realize the design (the deterministic `render(tMs)` body), reading `{skill-root}/references/ograf-spec.md` for the manifest and Web Component contract. Inline the logo SVG and embed fonts so the package is self-contained. Output goes to the project's `graphics/ograf/<id>/` (or the stream pack's scenes folder). - -## Verify before handoff - -Run `uv run {skill-root}/scripts/verify_ograf.py <package-dir>`, passing the same `--width`, `--height`, and `--duration` used at scaffold time (defaults 1920x1080, 10000ms): it serves the folder, simulates exactly what Resolve does (register class once, instantiate, `load({renderType:"nonrealtime"})`, `goToTime` across the timeline), saves a transparent screenshot for eyeball review, and fails on any console error or empty DOM render. A skipped run is not a verified package. Fix and re-run until clean. The manual verification steps it prints are per-OS (open on macOS, start on Windows, xdg-open on Linux, `uv run python -m http.server 8771`), and when run from a human terminal with a `preview.html` present it serves the package and opens the preview in the default browser itself (tty-gated; agent runs are unaffected). At handoff, point the user at `{skill-root}/references/resolve-workflow.md` for import steps, and remind them: serve `preview.html` over HTTP, never open it from disk. - -## Non-negotiable standards - -Each of these, when violated, produces a silent black clip: - -- Never call `customElements.define()` in the `.mjs`; export the class as `default` only. The renderer registers it. -- An OGraf graphic is a folder, not a file. Import the `.json` from inside the intact folder. -- Render deterministically from the timeline clock: all visual state derives from `goToTime`'s timestamp via one `render(tMs)`. `requestAnimationFrame` only inside real-time `playAction`/`stopAction`. -- Never size by reading `clientHeight` (it can be 0 offline). Fill the host; design at target resolution. -- `actionDurations` entries key on `type`, not `id`. -- Transparency = no background anywhere on host or page. -- Harden every asset/font load in try/catch with a fallback; an unguarded throw blanks the graphic. -- Implement the full non-real-time API (`load`, `dispose`, `updateAction`, `playAction`, `stopAction`, `customAction`, `goToTime`, `setActionsSchedule`) and declare `supportsNonRealTime: true`. -- Resolve caches a failed clip: after any fix, delete the old clip from the Media Pool before re-importing. OGraf needs Resolve 21+. - -## Files - -| File | When | -|---|---| -| `references/ograf-spec.md` | Authoring the manifest or `.mjs` | -| `references/resolve-workflow.md` | Handoff: Resolve import steps + black-screen troubleshooting | -| `scripts/scaffold_ograf.py` | Generating a new package | -| `scripts/verify_ograf.py` | Verifying against the renderer code path | -| `assets/*.template.*` | Templates the scaffold fills (not edited directly) | -| `references/engine-rationale.md` | Why OGraf earns a slot at all (dual-target rationale) | diff --git a/skills/mc-ograf/assets/graphic.template.mjs b/skills/mc-ograf/assets/graphic.template.mjs deleted file mode 100644 index 38184e9..0000000 --- a/skills/mc-ograf/assets/graphic.template.mjs +++ /dev/null @@ -1,152 +0,0 @@ -// {{NAME}} — OGraf v1 Graphic (EBU spec). -// A Web Component that renders DETERMINISTICALLY from a timeline clock, so it works -// both real-time (playAction/stopAction) and non-real-time / scrubbable (goToTime) — -// e.g. DaVinci Resolve 21. -// -// Generated by BMad Manticore (mc-ograf). Edit the render(tMs) body to realize your design; the -// lifecycle, registration rules, and hardening below are correct as-is — do not -// reintroduce customElements.define() here (the renderer registers the class). - -// ---- Design tokens ---- -const ACCENT = "{{ACCENT}}"; -const SURFACE = "{{SURFACE}}"; -const TEXT = "{{TEXT}}"; -const MUTED = "{{MUTED}}"; -const FONT = "{{FONT_STACK}}"; - -// ---- Timeline (ms) ---- -const DURATION = {{DURATION}}; -const IN = [200, 1300]; // entrance window -const EXIT_DUR = 800; -const EXIT_END = DURATION - 130; -const EXIT_START = EXIT_END - EXIT_DUR; - -// ---- Editable fields (manifest schema). Order drives layout. ---- -const FIELD_KEYS = {{FIELD_KEYS_JS}}; -const DATA_DEFAULTS = {{DEFAULTS_JS}}; - -// ---- Easing ---- -const easeOut = cubicBezier(0.16, 1, 0.3, 1); -function cubicBezier(p1x, p1y, p2x, p2y) { - const A = (a, b) => 1 - 3 * b + 3 * a, B = (a, b) => 3 * b - 6 * a, C = (a) => 3 * a; - const calc = (t, a, b) => ((A(a, b) * t + B(a, b)) * t + C(a)) * t; - const slope = (t, a, b) => 3 * A(a, b) * t * t + 2 * B(a, b) * t + C(a); - return (x) => { - if (x <= 0) return 0; if (x >= 1) return 1; - let t = x; - for (let i = 0; i < 6; i++) { - const cx = calc(t, p1x, p2x) - x, d = slope(t, p1x, p2x); - if (Math.abs(cx) < 1e-5 || d === 0) break; - t -= cx / d; - } - return calc(t, p1y, p2y); - }; -} -function interp(t, [t0, t1], [v0, v1], ease) { - if (t <= t0) return v0; if (t >= t1) return v1; - const x = (t - t0) / (t1 - t0); - return v0 + (v1 - v0) * (ease ? ease(x) : x); -} - -class Graphic extends HTMLElement { - constructor() { - super(); - this._data = { ...DATA_DEFAULTS }; - this._raf = null; - this.attachShadow({ mode: "open" }); - } - - // ---------- OGraf lifecycle ---------- - async load(params = {}) { - if (params.data) Object.assign(this._data, params.data); - this._build(); - this._renderAt(0, 0); - return undefined; - } - - async updateAction(params = {}) { - if (params.data) Object.assign(this._data, params.data); - this._bindText(); - return undefined; - } - - async playAction() { - this._stopRaf(); - const start = perfNow(); - const tick = () => { const e = perfNow() - start; this._renderAt(e, 0); this._raf = requestAnimationFrame(tick); }; - tick(); - return { currentStep: 1 }; - } - - async stopAction() { - this._stopRaf(); - const settled = IN[1] + 200, start = perfNow(); - return new Promise((resolve) => { - const tick = () => { - const p = Math.min(1, (perfNow() - start) / EXIT_DUR); - this._renderAt(settled, easeOut(p)); - if (p >= 1) { this._stopRaf(); resolve(undefined); return; } - this._raf = requestAnimationFrame(tick); - }; - tick(); - }); - } - - async goToTime(params = {}) { - this._stopRaf(); - const ts = Math.max(0, Math.min(DURATION, params.timestamp ?? 0)); - const exit = interp(ts, [EXIT_START, EXIT_END], [0, 1], easeOut); - this._renderAt(ts, exit); - return undefined; - } - - async setActionsSchedule() { return undefined; } - async customAction() { return undefined; } - async dispose() { this._stopRaf(); if (this.shadowRoot) this.shadowRoot.innerHTML = ""; return undefined; } - - // ---------- internals ---------- - _stopRaf() { if (this._raf != null) { cancelAnimationFrame(this._raf); this._raf = null; } } - - _build() { - const root = this.shadowRoot; - root.innerHTML = ` - <style> - :host { position:absolute; inset:0; display:block; overflow:hidden; width:100%; height:100%; } - /* Designed at {{WIDTH}}x{{HEIGHT}}; the renderer sizes the host to the canvas. */ - .stage { position:absolute; inset:0; font-family:${FONT}; } - .panel { position:absolute; left:96px; bottom:104px; min-width:520px; padding:26px 46px 26px 30px; - background:${SURFACE}; border-left:7px solid ${ACCENT}; border-radius:0 12px 12px 0; - box-shadow:0 14px 40px rgba(0,0,0,0.45); overflow:hidden; } - .headline { color:${TEXT}; font-size:60px; font-weight:700; letter-spacing:-1px; line-height:1; } - .subline { color:${MUTED}; font-size:24px; font-weight:600; letter-spacing:1px; margin-top:10px; } - </style> - <div class="stage"><div class="panel"></div></div>`; - this._stage = root.querySelector(".stage"); - this._panel = root.querySelector(".panel"); - this._bindText(); - } - - _bindText() { - if (!this._panel) return; - const [head, ...subs] = FIELD_KEYS; - const esc = (s) => String(s ?? "").replace(/[&<>]/g, (c) => ({ "&": "&", "<": "<", ">": ">" }[c])); - this._panel.innerHTML = - `<div class="headline">${esc(this._data[head])}</div>` + - subs.map((k) => `<div class="subline">${esc(this._data[k])}</div>`).join(""); - } - - // tMs drives the entrance; exit is 0..1. ALL visual state derives from these. - _renderAt(tMs, exit) { - if (!this._panel) return; - const reveal = interp(tMs, IN, [0, 100], easeOut); - const enter = interp(tMs, [0, 700], [0, 1], easeOut); - this._panel.style.clipPath = `inset(0 ${100 - reveal}% 0 0)`; - this._stage.style.opacity = 1 - exit; - this._stage.style.transform = `translateX(${-50 * (1 - enter) - 70 * exit}px)`; - } -} - -const perfNow = () => (typeof performance !== "undefined" && performance.now ? performance.now() : 0); - -// Do NOT call customElements.define() here — the renderer registers this class. -export default Graphic; diff --git a/skills/mc-ograf/assets/manifest.template.json b/skills/mc-ograf/assets/manifest.template.json deleted file mode 100644 index f631c9d..0000000 --- a/skills/mc-ograf/assets/manifest.template.json +++ /dev/null @@ -1,21 +0,0 @@ -{ - "$schema": "https://ograf.ebu.io/v1/specification/json-schemas/graphics/schema.json", - "id": "{{ID}}", - "name": "{{NAME}}", - "version": "1.0.0", - "description": "{{DESCRIPTION}}", - "author": { "name": "{{AUTHOR}}" }, - "main": "{{MAIN}}", - "supportsRealTime": true, - "supportsNonRealTime": true, - "stepCount": 1, - "actionDurations": [ - { "type": "playAction", "duration": 1300 }, - { "type": "updateAction", "duration": 400 }, - { "type": "stopAction", "duration": 800 } - ], - "schema": { - "type": "object", - "properties": {{SCHEMA_PROPERTIES}} - } -} diff --git a/skills/mc-ograf/assets/preview.template.html b/skills/mc-ograf/assets/preview.template.html deleted file mode 100644 index d944be4..0000000 --- a/skills/mc-ograf/assets/preview.template.html +++ /dev/null @@ -1,51 +0,0 @@ -<!doctype html> -<html lang="en"> - <head> - <meta charset="utf-8" /> - <title>{{NAME}} — OGraf preview - - - -
    -
    - - - - -
    - - - diff --git a/skills/mc-ograf/customize.toml b/skills/mc-ograf/customize.toml deleted file mode 100644 index 44e22e2..0000000 --- a/skills/mc-ograf/customize.toml +++ /dev/null @@ -1,20 +0,0 @@ -# DO NOT EDIT -- overwritten on every update. -# -# Customization surface for mc-ograf. -# Override files (not edited here): -# {project-root}/_bmad/custom/mc-ograf.toml (team) -# {project-root}/_bmad/custom/mc-ograf.user.toml (personal) -# -# Studio-wide config (owner, paths, editor, transcription, assets, tools) -# lives in [modules.manticore] in {project-root}/_bmad/custom/config.toml, -# maintained by mc-setup. This file holds only this skill's own surface. - -[workflow] - -# Steps to run before / after the standard activation. -activation_steps_prepend = [] -activation_steps_append = [] - -# Persistent facts held for the whole run: literal sentences, or -# "file:{project-root}/..." paths whose contents are loaded as facts. -persistent_facts = [] diff --git a/skills/mc-ograf/references/engine-rationale.md b/skills/mc-ograf/references/engine-rationale.md deleted file mode 100644 index 3b78043..0000000 --- a/skills/mc-ograf/references/engine-rationale.md +++ /dev/null @@ -1,27 +0,0 @@ -# Engine: OGraf (dual-target graphics) - -Ships with the module: `assets/` (graphic templates), `references/` (spec notes, Resolve workflow notes), and `scripts/` (scaffold and Playwright-verify). - -## Why this engine exists in the suite - -OGraf (EBU spec) is the only graphics format that is simultaneously: - -- a native DaVinci Resolve 21 media-pool citizen (graphics stay EDITABLE inside Resolve during the creator's polish pass, unlike baked alpha MOVs), and -- an OBS/CasparCG live graphic via the free SPX-GC controller (click-to-trigger during streams). - -No commercial stream-package vendor sells assets that work in both places. That makes OGraf the engine for lower thirds and topic cards that need to live in both worlds, and the backbone of the livestream pack. - -## Jobs - -- Lower thirds and topic cards for mc-stream-pack (SPX-GC compatible, standalone-capable HTML). -- Resolve-editable titles/lower thirds for edited videos when post-import tweaking matters more than motion complexity (baked HyperFrames alpha is the default; OGraf is the "keep it editable" escape hatch). - -## Ported knowledge - -- `references/ograf-spec.md`: the loading rules and spec constraints that were hard-won getting OGraf graphics loading reliably in Resolve. -- `references/resolve-workflow.md`: how OGraf graphics behave in Resolve 21. -- `scripts/scaffold_ograf.py` + `scripts/verify_ograf.py` (+ tests): scaffold a compliant graphic; Playwright-verify rendering. - -## Note - -No commercial vendor sells brand-locked dual-target packs like this; it is a genuine differentiator for creators who both edit and stream. diff --git a/skills/mc-ograf/references/ograf-spec.md b/skills/mc-ograf/references/ograf-spec.md deleted file mode 100644 index cc33a5c..0000000 --- a/skills/mc-ograf/references/ograf-spec.md +++ /dev/null @@ -1,71 +0,0 @@ -# OGraf v1 — manifest + Web Component contract - -The authoring contract for an [OGraf v1](https://ograf.ebu.io/v1/specification/docs/Specification.html) graphic. The scaffold script emits a correct skeleton; consult this when editing the manifest or the `.mjs` by hand. - -## Package = folder - -A graphic is a folder whose entrypoint is the manifest; everything else is referenced **relative to the manifest's location**: - -``` -my-graphic/ - my-graphic.ograf.json # manifest (filename MUST end .ograf.json) - my-graphic.mjs # the manifest's "main" - preview.html # local harness (not part of the spec) - assets/ # optional: fonts, images (or inline them) -``` - -There is no `.ograf` zip container in v1 — it's a plain folder. Multiple `*.ograf.json` files in one folder = multiple independent graphics. - -## Manifest fields - -Required: `$schema`, `id` (unique, **no forward slashes**), `name`, `main` (path to the JS), `supportsRealTime`, `supportsNonRealTime`. - -Common: `version`, `description`, `author` ({`name` required, `email`/`url` optional}), `schema`, `actionDurations`, `stepCount` (default 1). - -- **`schema`** — JSON Schema describing the data model for `load()` and `updateAction()`. Each property is one operator-editable field; give it a `title` and `default`. This is the config surface a renderer exposes as Inspector controls. Keep it tight. -- **`actionDurations`** — entries key on **`type`** (`playAction` | `updateAction` | `stopAction` | `customAction`), each with a `duration` in ms. (Keying on `id` is wrong and silently breaks timing.) A `customAction` entry also carries `customActionId`. -- Vendor-specific fields use a `v_` prefix. - -## Web Component class - -The `main` file's default export is a class that `extends HTMLElement`. **Do not call `customElements.define()`** — the renderer registers the class, and a class binds to only one tag name; self-registering makes the renderer's `define()` throw and the graphic fails to load. - -```js -class Graphic extends HTMLElement { /* methods below */ } -export default Graphic; // export only -``` - -All action methods are `async` and return `Promise`. - -### Required (all graphics) - -| Method | Params | Purpose | -| --- | --- | --- | -| `load` | `{ data, renderType, renderCharacteristics }` | Init; resolve when ready for actions. `data` conforms to manifest `schema`. `renderType` is `"realtime"` or non-real-time. | -| `dispose` | `{ ... }` | Cleanup. | -| `playAction` | `{ goto, delta, skipAnimation }` | Advance/play; returns `{ currentStep }`. | -| `stopAction` | `{ skipAnimation }` | End display (play out). | -| `updateAction` | `{ data, skipAnimation }` | Merge new `data`, re-render. | -| `customAction` | `{ id, payload, skipAnimation }` | Invoke a manifest-declared custom action. | - -### Required additionally for non-real-time - -Declare `supportsNonRealTime: true` and implement: - -| Method | Params | Purpose | -| --- | --- | --- | -| `goToTime` | `{ timestamp }` | Render the exact frame at `timestamp` (ms). This is how offline renderers (Resolve) draw every frame. | -| `setActionsSchedule` | `{ schedule }` | Queue timed actions (may be a no-op for a self-contained baked timeline). | - -## The determinism rule - -Real-time playout drives in/out via `playAction`/`stopAction`; offline render draws each frame via `goToTime`. To satisfy both, derive **all** visual state from a single pure `render(tMs)` and call it from `goToTime`. Use `requestAnimationFrame` only to animate the real-time `playAction`/`stopAction` — never for the core animation, because an offline renderer never runs that loop. - -## Transparency & sizing - -- No background on the host element or page — paint only the graphic. Alpha is preserved by the renderer. -- The renderer sizes the host to the output canvas. Fill it (`:host{position:absolute;inset:0}`). Never read `clientHeight` and scale by it — it can be `0` offline and collapse the graphic to nothing. - -## Asset loading - -Inline small assets (logo SVG as markup; font as a bundled file or base64) for a self-contained package. Wrap every external load (`FontFace`, `new URL(..., import.meta.url)`, image fetch) in `try/catch` with a fallback — an unguarded throw during `load()` blanks the graphic. diff --git a/skills/mc-ograf/references/resolve-workflow.md b/skills/mc-ograf/references/resolve-workflow.md deleted file mode 100644 index 7bc194b..0000000 --- a/skills/mc-ograf/references/resolve-workflow.md +++ /dev/null @@ -1,49 +0,0 @@ -# Using an OGraf graphic in DaVinci Resolve 21 / Fusion - -OGraf HTML graphics and Lottie are native in **Resolve 21+** (not 20.x). Resolve drives OGraf in **non-real-time** mode — it renders frame-by-frame via `goToTime`, so the graphic must declare `supportsNonRealTime: true` and implement `goToTime` (the scaffold does). The message *"Resolve and Fusion currently only support non-real time OGraf files"* is informational, not an error. - -## Import (Edit/Cut page) - -1. Keep the package **folder** intact — `*.ograf.json` + the `.mjs` + any assets together. The manifest loads its `main` and assets relative to itself. -2. Drag the **`*.ograf.json`** into the Media Pool **from inside that folder** (so Resolve resolves the siblings). It appears as a clip with alpha. -3. Drop the clip on a track above your footage. Transparency composites automatically — no matte. -4. Select the clip and edit the schema fields in the **Inspector** (Title, Subtitle, etc. — whatever the manifest `schema` exposes). - -## Fusion - -Use the **OGrafLoader** node (added in 21) and point it at the manifest to bring the graphic into the node graph. - -## Black-screen troubleshooting - -A black clip has no error dialog, so check these in order: - -1. **Imported only the `.json`?** That orphans it from its `.mjs`/assets. Re-import the `.json` from inside the intact folder. -2. **The `.mjs` calls `customElements.define()`?** Remove it. Resolve registers the class; self-registration throws *"this constructor has already been used with this registry"* → load fails, **no Inspector fields appear** (this is the tell: no fields = didn't load). -3. **No Inspector fields at all?** The graphic didn't load (see #1/#2) or the manifest `schema` is empty/malformed. -4. **Loads but renders blank?** An asset/font load threw (wrap in `try/catch`), or the graphic sized itself to the host's `clientHeight` which was `0` (fill the host instead). -5. **Still the old broken render after a fix?** Resolve **caches** the clip — delete it from the Media Pool (and timeline) and re-import. -6. Confirm Resolve is **21.0+**. - -Run `uv run scripts/verify_ograf.py ` first — it reproduces #2 and #4 headlessly before you ever open Resolve. - -## Previewing locally before Resolve - -Serve the folder and open `preview.html` over HTTP: `uv run python -m http.server 8771` in the package folder, then open `localhost:/preview.html` (verify_ograf.py prints per-OS steps and, from a human terminal, serves and opens the preview itself). **Never double-click `preview.html` (`file://`)**: the browser blocks its ES-module import and inlined data-URL assets, so the graphic silently fails while the controls/checkerboard still show ("Some content has been disabled"). This is a browser preview limit only; Resolve's renderer is unaffected. - -## Scripted import on free Resolve (Fusion Scripts menu) - -Resolve's external scripting API is Studio-only through Resolve 21; the free edition only executes scripts launched from inside the app (Console or Workspace > Scripts). To run the pipeline's scripted timeline import (resolve_import.py, once implemented) on free Resolve, copy the script into the Fusion Scripts folder and launch it from Workspace > Scripts: - -- macOS: `~/Library/Application Support/Blackmagic Design/DaVinci Resolve/Fusion/Scripts/Utility/` -- Windows: `%APPDATA%\Blackmagic Design\DaVinci Resolve\Support\Fusion\Scripts\Utility\` -- Linux: `~/.local/share/DaVinciResolve/Fusion/Scripts/` - -This upgrades the free lane from manual FCPXML import to native scripted import with zero dependencies. - -## Linux free-edition codec caveat - -The Linux free edition cannot decode or encode H.264 or H.265 and has no AAC at all. The FCPXML timeline imports fine, but mp4/AAC media is undecodable there; transcode sources to ProRes or DNxHR first, or use Resolve Studio. - -## Studio power lane: Resolve MCP (opt-in) - -Studio users who want conversational post-import work (timeline surgery, Text+ titles, markers, render-queue automation) can opt into the community [samuelgursky/davinci-resolve-mcp](https://github.com/samuelgursky/davinci-resolve-mcp) server (MIT, macOS/Windows/Linux; Studio-only, because it uses the external scripting API). Manticore keeps doing media import and timeline construction itself through the FCPXML lane and never requires an MCP server; record an already-running server via mc-setup's `[mcp]` step if you want skills to use it. diff --git a/skills/mc-ograf/scripts/scaffold_ograf.py b/skills/mc-ograf/scripts/scaffold_ograf.py deleted file mode 100644 index 2725b43..0000000 --- a/skills/mc-ograf/scripts/scaffold_ograf.py +++ /dev/null @@ -1,187 +0,0 @@ -#!/usr/bin/env python3 -"""Scaffold a spec-compliant OGraf graphic package from the skill's asset templates. - -Writes // with: - .ograf.json manifest (correct fields, type-keyed actionDurations, schema) - .mjs Web Component (deterministic render, NO self-registration) - preview.html local harness that registers + scrubs the graphic - -The STRUCTURAL standards that produce a silent black clip in a renderer (no -self-registration, type-keyed actionDurations, full non-real-time API, -transparent host) are baked in here, so callers should generate rather than -hand-write. Standards that live in the caller's own edits (deterministic -render(tMs), hardened asset/font loads) still apply afterward. Stdlib only. - -Palette/font args should come from {brand-path}/tokens.json; any left at the -placeholder defaults are reported in the JSON output as placeholder_palette -so a placeholder look never ships unnoticed. - -Fields define the operator-editable config surface (the manifest schema). Pass as -JSON: --fields '[{"key":"title","title":"Title","default":"Guest Name"}]' -or simply: --field title="Guest Name" --field subtitle="Channel Name" -""" -# /// script -# requires-python = ">=3.11" -# dependencies = [] -# /// -import argparse -import json -import re -import sys -from pathlib import Path - -ASSETS = Path(__file__).resolve().parent.parent / "assets" - - -def jstr(s): - """JSON-string-safe inner text (escapes quotes/backslashes/newlines).""" - return json.dumps(str(s))[1:-1] - - -def humanize(key): - """Field key -> Inspector title. Splits camelCase + snake/kebab, Title-Cases. - guestName -> 'Guest Name', accent_color -> 'Accent Color'.""" - s = re.sub(r"(?<=[a-z0-9])(?=[A-Z])", " ", key).replace("_", " ").replace("-", " ") - return " ".join(s.split()).title() - - -def parse_fields(args): - fields = [] - if args.fields: - try: - raw = json.loads(args.fields) - except json.JSONDecodeError as e: - sys.exit(f"--fields is not valid JSON: {e}") - for f in raw: - key = f.get("key") - if not key: - sys.exit("each --fields entry needs a 'key'") - fields.append({"key": key, "title": f.get("title", humanize(key)), - "default": f.get("default", "")}) - for kv in args.field or []: - if "=" not in kv: - sys.exit(f"--field must be key=default, got: {kv}") - key, default = kv.split("=", 1) - key = key.strip() - fields.append({"key": key, "title": humanize(key), "default": default}) - if not fields: - fields = [{"key": "title", "title": "Title", "default": args.name}, - {"key": "subtitle", "title": "Subtitle", "default": ""}] - seen = set() - for f in fields: - if not re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", f["key"]): - sys.exit(f"field key '{f['key']}' must be a valid identifier (no spaces/dashes)") - if f["key"] in seen: - sys.exit(f"duplicate field key: {f['key']}") - seen.add(f["key"]) - return fields - - -def main(): - p = argparse.ArgumentParser(description="Scaffold an OGraf graphic package") - p.add_argument("--id", required=True, help="graphic id (kebab-case, no slashes)") - p.add_argument("--name", required=True, help="human-readable name") - p.add_argument("--dest", required=True, help="parent directory; package goes in //") - p.add_argument("--description", default="") - p.add_argument("--author", default="", help="creator/channel name; pass from the studio config [owner] table") - p.add_argument("--width", type=int, default=1920) - p.add_argument("--height", type=int, default=1080) - p.add_argument("--duration", type=int, default=10000, help="timeline length in ms") - p.add_argument("--fields", help="JSON array of {key,title,default}") - p.add_argument("--field", action="append", help="key=default (repeatable)") - p.add_argument("--accent", default="#4F8CFF", help="pass from {brand-path}/tokens.json") - p.add_argument("--surface", default="#1D1D1D", help="pass from {brand-path}/tokens.json") - p.add_argument("--text", default="#FFFFFF") - p.add_argument("--muted", default="#A8A8A8", help="pass from {brand-path}/tokens.json") - p.add_argument("--font", default="system-ui, sans-serif", help="pass from {brand-path}/tokens.json") - p.add_argument("--force", action="store_true", help="overwrite an existing package") - args = p.parse_args() - - gid = args.id.strip().lower() - if not re.fullmatch(r"[a-z0-9][a-z0-9-]*", gid): - sys.exit("--id must be kebab-case with no slashes (spec: id has no forward slashes)") - - fields = parse_fields(args) - main_file = f"{gid}.mjs" - - schema_props = { - f["key"]: {"type": "string", "title": f["title"], "default": f["default"]} - for f in fields - } - field_keys = [f["key"] for f in fields] - defaults = {f["key"]: f["default"] for f in fields} - - subs = { - "{{ID}}": gid, - "{{NAME}}": jstr(args.name), - "{{DESCRIPTION}}": jstr(args.description), - "{{AUTHOR}}": jstr(args.author), - "{{MAIN}}": main_file, - "{{WIDTH}}": str(args.width), - "{{HEIGHT}}": str(args.height), - "{{DURATION}}": str(args.duration), - "{{SCHEMA_PROPERTIES}}": json.dumps(schema_props, indent=6).replace("\n", "\n "), - "{{FIELD_KEYS_JS}}": json.dumps(field_keys), - "{{DEFAULTS_JS}}": json.dumps(defaults), - "{{ACCENT}}": jstr(args.accent), - "{{SURFACE}}": jstr(args.surface), - "{{TEXT}}": jstr(args.text), - "{{MUTED}}": jstr(args.muted), - "{{FONT_STACK}}": jstr(args.font), - } - - def fill(template_name): - text = (ASSETS / template_name).read_text(encoding="utf-8") - for k, v in subs.items(): - text = text.replace(k, v) - leftover = re.findall(r"\{\{[A-Z_]+\}\}", text) - if leftover: - sys.exit(f"unfilled placeholders in {template_name}: {sorted(set(leftover))}") - return text - - pkg = Path(args.dest) / gid - if pkg.exists() and not args.force: - sys.exit(f"{pkg} already exists (use --force to overwrite)") - pkg.mkdir(parents=True, exist_ok=True) - - written = {} - for tmpl, out in [ - ("manifest.template.json", f"{gid}.ograf.json"), - ("graphic.template.mjs", main_file), - ("preview.template.html", "preview.html"), - ]: - dest = pkg / out - dest.write_text(fill(tmpl), encoding="utf-8") - written[out] = str(dest) - - # validate the manifest we just wrote - try: - json.loads((pkg / f"{gid}.ograf.json").read_text(encoding="utf-8")) - except json.JSONDecodeError as e: - sys.exit(f"generated manifest is invalid JSON: {e}") - - placeholder_defaults = {"accent": "#4F8CFF", "surface": "#1D1D1D", - "muted": "#A8A8A8", "font": "system-ui, sans-serif"} - placeholders = sorted(k for k, v in placeholder_defaults.items() - if getattr(args, k) == v) - - out = { - "ok": True, - "package": str(pkg), - "manifest": written[f"{gid}.ograf.json"], - "main": written[main_file], - "preview": written["preview.html"], - "fields": field_keys, - "next": f"Edit {main_file} render(tMs) for your design, then: uv run scripts/verify_ograf.py {pkg} " - f"--width {args.width} --height {args.height} --duration {args.duration}", - } - if placeholders: - out["placeholder_palette"] = placeholders - out["warning"] = ("placeholder palette: --" + ", --".join(placeholders) + - " left at module defaults; pass values from {brand-path}/tokens.json " - "before this graphic ships") - print(json.dumps(out, indent=2)) - - -if __name__ == "__main__": - main() diff --git a/skills/mc-ograf/scripts/tests/test-scaffold_ograf.py b/skills/mc-ograf/scripts/tests/test-scaffold_ograf.py deleted file mode 100644 index f6c4f52..0000000 --- a/skills/mc-ograf/scripts/tests/test-scaffold_ograf.py +++ /dev/null @@ -1,114 +0,0 @@ -#!/usr/bin/env python3 -# /// script -# requires-python = ">=3.11" -# /// -"""Tests for scaffold_ograf.py — the package generator must always emit a -spec-compliant, standards-clean package.""" -import json -import re -import subprocess -import sys -import tempfile -import unittest -from pathlib import Path - -SCRIPT = Path(__file__).resolve().parent.parent / "scaffold_ograf.py" - - -def run(args, expect_ok=True): - r = subprocess.run([sys.executable, str(SCRIPT), *args], capture_output=True, text=True) - if expect_ok: - assert r.returncode == 0, f"scaffold failed: {r.stderr}\n{r.stdout}" - return r - - -class TestScaffold(unittest.TestCase): - def _build(self, tmp, extra=None): - args = ["--id", "demo-l3", "--name", "Demo L3", "--dest", tmp, - "--fields", json.dumps([{"key": "title", "title": "Title", "default": "Hi"}, - {"key": "subtitle", "title": "Sub", "default": "there"}])] - return run(args + (extra or [])) - - def test_emits_three_files(self): - with tempfile.TemporaryDirectory() as tmp: - self._build(tmp) - pkg = Path(tmp) / "demo-l3" - self.assertTrue((pkg / "demo-l3.ograf.json").exists()) - self.assertTrue((pkg / "demo-l3.mjs").exists()) - self.assertTrue((pkg / "preview.html").exists()) - - def test_manifest_is_spec_compliant(self): - with tempfile.TemporaryDirectory() as tmp: - self._build(tmp) - m = json.loads((Path(tmp) / "demo-l3" / "demo-l3.ograf.json").read_text()) - self.assertEqual(m["main"], "demo-l3.mjs") - self.assertTrue(m["supportsNonRealTime"]) - self.assertIn("title", m["schema"]["properties"]) - self.assertEqual(m["schema"]["properties"]["title"]["default"], "Hi") - # actionDurations key on 'type', never 'id' - for ad in m["actionDurations"]: - self.assertIn("type", ad) - self.assertNotIn("id", ad) - - def test_mjs_never_self_registers(self): - with tempfile.TemporaryDirectory() as tmp: - self._build(tmp) - mjs = (Path(tmp) / "demo-l3" / "demo-l3.mjs").read_text() - self.assertIn("export default", mjs) - # the only mention of customElements.define must be a comment, never a call - for line in mjs.splitlines(): - if "customElements.define" in line: - self.assertTrue(line.lstrip().startswith("//"), - f"non-comment customElements.define: {line!r}") - - def test_preview_registers_under_id(self): - with tempfile.TemporaryDirectory() as tmp: - self._build(tmp) - html = (Path(tmp) / "demo-l3" / "preview.html").read_text() - self.assertIn("demo-l3.mjs", html) - self.assertIn('customElements.define("demo-l3"', html) - - def test_default_fields_when_none_given(self): - with tempfile.TemporaryDirectory() as tmp: - run(["--id", "bare", "--name", "Bare", "--dest", tmp]) - m = json.loads((Path(tmp) / "bare" / "bare.ograf.json").read_text()) - self.assertIn("title", m["schema"]["properties"]) - - def test_rejects_id_with_slash(self): - with tempfile.TemporaryDirectory() as tmp: - r = run(["--id", "bad/id", "--name", "X", "--dest", tmp], expect_ok=False) - self.assertNotEqual(r.returncode, 0) - - def test_rejects_field_key_with_dash(self): - with tempfile.TemporaryDirectory() as tmp: - r = run(["--id", "ok", "--name", "X", "--dest", tmp, - "--field", "bad-key=v"], expect_ok=False) - self.assertNotEqual(r.returncode, 0) - - def test_warns_on_placeholder_palette(self): - with tempfile.TemporaryDirectory() as tmp: - r = self._build(tmp) - out = json.loads(r.stdout) - self.assertIn("placeholder_palette", out) - self.assertIn("accent", out["placeholder_palette"]) - self.assertIn("tokens.json", out["warning"]) - - def test_no_warning_when_palette_passed(self): - with tempfile.TemporaryDirectory() as tmp: - r = self._build(tmp, extra=["--accent", "#AA2200", "--surface", "#101418", - "--muted", "#8899AA", "--font", "Inter, sans-serif"]) - out = json.loads(r.stdout) - self.assertNotIn("placeholder_palette", out) - self.assertNotIn("warning", out) - - def test_no_unfilled_placeholders(self): - with tempfile.TemporaryDirectory() as tmp: - self._build(tmp) - pkg = Path(tmp) / "demo-l3" - for f in pkg.iterdir(): - self.assertFalse(re.search(r"\{\{[A-Z_]+\}\}", f.read_text()), - f"unfilled placeholder in {f.name}") - - -if __name__ == "__main__": - unittest.main() diff --git a/skills/mc-ograf/scripts/tests/test-verify_ograf.py b/skills/mc-ograf/scripts/tests/test-verify_ograf.py deleted file mode 100644 index aafa15e..0000000 --- a/skills/mc-ograf/scripts/tests/test-verify_ograf.py +++ /dev/null @@ -1,141 +0,0 @@ -#!/usr/bin/env python3 -# /// script -# requires-python = ">=3.11" -# /// -"""Tests for verify_ograf.py — the deterministic, browser-free parts: manifest -discovery, usage/structure errors, the per-OS manual-step strings, and the -graceful no-headless fallback. - -The full headless render check requires Playwright + a browser and is exercised -by running the script directly on a package; it is not unit-tested here. The -interactive serve-and-open branch is tty-gated and preview.html-gated, so it -never fires under the test harness; only its gating is asserted.""" -import contextlib -import importlib.util -import io -import json -import subprocess -import sys -import tempfile -import unittest -from pathlib import Path - -SCRIPT = Path(__file__).resolve().parent.parent / "verify_ograf.py" - -spec = importlib.util.spec_from_file_location("verify_ograf", SCRIPT) -verify = importlib.util.module_from_spec(spec) -spec.loader.exec_module(verify) - - -def make_pkg(tmp, *, with_main=True): - pkg = Path(tmp) / "pkg" - pkg.mkdir() - (pkg / "g.ograf.json").write_text(json.dumps({ - "id": "g", "name": "G", "main": "g.mjs", - "supportsNonRealTime": True, "schema": {"type": "object", "properties": {}}, - })) - if with_main: - (pkg / "g.mjs").write_text( - "class G extends HTMLElement{\n" - " constructor(){super();this.attachShadow({mode:'open'});}\n" - " async load(){this.shadowRoot.innerHTML=" - "\"
    x
    \";return undefined;}\n" - " async goToTime(){return undefined;}\n" - " async dispose(){return undefined;}\n" - " async updateAction(){return undefined;}\n" - " async playAction(){return {currentStep:1};}\n" - " async stopAction(){return undefined;}\n" - " async customAction(){return undefined;}\n" - " async setActionsSchedule(){return undefined;}\n" - "}\nexport default G;\n") - return pkg - - -def run(args): - return subprocess.run([sys.executable, str(SCRIPT), *args], capture_output=True, text=True) - - -class TestVerifyHelpers(unittest.TestCase): - def test_find_manifest_returns_the_manifest(self): - with tempfile.TemporaryDirectory() as tmp: - pkg = make_pkg(tmp) - self.assertEqual(verify.find_manifest(pkg).name, "g.ograf.json") - - def test_find_manifest_exits_when_absent(self): - with tempfile.TemporaryDirectory() as tmp: - empty = Path(tmp) / "empty" - empty.mkdir() - with self.assertRaises(SystemExit): - verify.find_manifest(empty) - - -class TestManualSteps(unittest.TestCase): - def test_steps_per_os(self): - pkg = Path("/x/pkg") - expected_open = {"Darwin": "open http://", - "Windows": "start http://", - "Linux": "xdg-open http://"} - for system, needle in expected_open.items(): - steps = verify.manual_verify_steps(pkg, system=system) - joined = "\n".join(steps) - self.assertIn(needle, joined, system) - # No shell chaining and no bare python3: both are POSIX-shaped. - self.assertNotIn("&&", joined, system) - self.assertNotIn("python3", joined, system) - self.assertIn("uv run python -m http.server 8771", joined, system) - self.assertIn(str(pkg), joined, system) - - def test_windows_cd_crosses_drives(self): - steps = verify.manual_verify_steps(Path("/x/pkg"), system="Windows") - self.assertIn("cd /d", "\n".join(steps)) - - def test_default_system_is_this_machine(self): - a = verify.manual_verify_steps(Path("/x/pkg")) - b = verify.manual_verify_steps(Path("/x/pkg"), - system=verify.platform.system()) - self.assertEqual(a, b) - - def test_manual_steps_prints_json_and_never_blocks_without_tty(self): - with tempfile.TemporaryDirectory() as tmp: - pkg = make_pkg(tmp) # no preview.html and no tty: both gates hold - buf = io.StringIO() - with contextlib.redirect_stdout(buf): - verify.manual_steps(pkg, "Playwright not installed") - payload = json.loads(buf.getvalue()) - self.assertEqual(payload["status"], "skipped-no-headless") - self.assertEqual(payload["reason"], "Playwright not installed") - self.assertEqual(payload["manual_verify"], - verify.manual_verify_steps(pkg)) - - def test_open_preview_noop_without_preview_html(self): - with tempfile.TemporaryDirectory() as tmp: - pkg = make_pkg(tmp) - # Must return immediately (no server, no browser, no input()). - self.assertIsNone(verify.open_preview_if_interactive(pkg)) - - -class TestVerifyCli(unittest.TestCase): - def test_missing_main_fails_usage(self): - with tempfile.TemporaryDirectory() as tmp: - pkg = make_pkg(tmp, with_main=False) - r = run([str(pkg)]) - self.assertEqual(r.returncode, 2) - self.assertIn("folder", r.stderr.lower() + r.stdout.lower()) - - def test_not_a_directory(self): - r = run(["/nonexistent/path/xyz"]) - self.assertEqual(r.returncode, 2) - - def test_full_run_is_ok_or_graceful(self): - # With a valid package: either headless verifies (exit 0) or Playwright is - # absent and it degrades with manual steps (exit 3). Never crashes (1/2). - with tempfile.TemporaryDirectory() as tmp: - pkg = make_pkg(tmp) - r = run([str(pkg)]) - self.assertIn(r.returncode, (0, 3), f"unexpected: {r.returncode}\n{r.stdout}\n{r.stderr}") - payload = json.loads(r.stdout) - self.assertIn("status", payload) - - -if __name__ == "__main__": - unittest.main() diff --git a/skills/mc-ograf/scripts/verify_ograf.py b/skills/mc-ograf/scripts/verify_ograf.py deleted file mode 100644 index 580c346..0000000 --- a/skills/mc-ograf/scripts/verify_ograf.py +++ /dev/null @@ -1,243 +0,0 @@ -#!/usr/bin/env python3 -"""Verify an OGraf package against the exact code path a renderer (DaVinci Resolve) -uses, so the black-screen failure modes are caught before handoff. - -It serves the package folder and, in a headless browser, does what Resolve does: - register the class under a FRESH tag (throws if the .mjs self-registered) - -> instantiate -> load({renderType:"nonrealtime"}) -> goToTime() across the timeline -then checks for console/page errors and that something actually renders. - -Exit codes: - 0 verified OK - 1 verification FAILED (error thrown, or nothing rendered) -> details printed - 2 bad usage / package not found - 3 could not run headless (Playwright missing or no browser) -> manual steps printed - -Playwright is an OPTIONAL dependency. Without it, this prints how to verify by -hand: per-OS instructions (open/start/xdg-open, no shell chaining), and when a -human terminal is attached AND the package has a preview.html, it also serves -the package in-process and opens the preview in the default browser -(webbrowser.open) so the manual check starts already running. Agent/CI runs -(no tty) never block. -Stdlib only except for the optional Playwright import. -""" -# /// script -# requires-python = ">=3.11" -# dependencies = ["playwright"] -# /// -# playwright is declared so `uv run` provisions it, but it is imported lazily and -# its absence is handled gracefully (plain `python3` prints manual steps instead). -# Browser one-time setup: playwright install chromium -import argparse -import http.server -import json -import platform -import socket -import socketserver -import sys -import tempfile -import threading -import webbrowser -from pathlib import Path - -VERIFY_HTML = "_ograf_verify.html" - - -def die(msg: str, code: int = 2): - """Usage/structure error → exit 2 (distinct from 1 = verification failed).""" - print(msg, file=sys.stderr) - sys.exit(code) - - -def find_manifest(pkg: Path) -> Path: - hits = sorted(pkg.glob("*.ograf.json")) - if not hits: - die(f"no *.ograf.json found in {pkg}") - return hits[0] - - -def serve(directory: Path): - class Quiet(http.server.SimpleHTTPRequestHandler): - def __init__(self, *a, **k): - super().__init__(*a, directory=str(directory), **k) - - def log_message(self, *a): - pass - - handler = Quiet - with socket.socket() as s: - s.bind(("127.0.0.1", 0)) - port = s.getsockname()[1] - httpd = socketserver.TCPServer(("127.0.0.1", port), handler) - t = threading.Thread(target=httpd.serve_forever, daemon=True) - t.start() - return httpd, port - - -def manual_verify_steps(pkg: Path, system: str | None = None) -> list[str]: - """Per-OS manual verification steps (no shell chaining, no bare python3). - - system defaults to this machine (platform.system()); tests pass it - explicitly to assert every OS variant from any OS.""" - system = system or platform.system() - url = "http://localhost:8771/preview.html" - if system == "Windows": - cd_cmd = f'cd /d "{pkg}"' - open_cmd = f"start {url}" - elif system == "Darwin": - cd_cmd = f'cd "{pkg}"' - open_cmd = f"open {url}" - else: - cd_cmd = f'cd "{pkg}"' - open_cmd = f"xdg-open {url}" - return [ - f"In a terminal, change into the package: {cd_cmd}", - "Serve it: uv run python -m http.server 8771", - f"Open the preview in a browser: {open_cmd}", - "Scrub the slider end-to-end; the graphic must animate in, hold, and out.", - "Open the browser console — there must be ZERO errors.", - "The checkerboard must show through (transparency).", - ] - - -def open_preview_if_interactive(pkg: Path) -> None: - """When a human terminal is attached, serve the package in-process and - open preview.html in the default browser (webbrowser.open picks the - right opener on every OS), then hold the server until Enter. - - A no-op when there is no tty (agent/CI runs must never block), when the - package has no preview.html, or on any failure.""" - if not (pkg / "preview.html").is_file(): - return - try: - if not (sys.stdin.isatty() and sys.stderr.isatty()): - return - httpd, port = serve(pkg) - url = f"http://127.0.0.1:{port}/preview.html" - try: - if not webbrowser.open(url): - return - print(f"preview served at {url}; press Enter to stop the server " - "when done...", file=sys.stderr) - try: - input() - except EOFError: - pass - finally: - httpd.shutdown() - except Exception: - return - - -def manual_steps(pkg: Path, msg: str): - print(json.dumps({ - "ok": None, - "status": "skipped-no-headless", - "reason": msg, - "manual_verify": manual_verify_steps(pkg), - "enable_headless": "install the 'playwright' package, then run: playwright install chromium", - }, indent=2)) - open_preview_if_interactive(pkg) - - -def main(): - ap = argparse.ArgumentParser( - description="Verify an OGraf package against the renderer code path (catches the " - "black-screen failure modes). Exits 0 ok, 1 failed, 2 usage, 3 no-headless.") - ap.add_argument("package", help="path to the OGraf package directory (contains *.ograf.json)") - ap.add_argument("--width", type=int, default=1920, - help="viewport width in px; pass the value used at scaffold time") - ap.add_argument("--height", type=int, default=1080, - help="viewport height in px; pass the value used at scaffold time") - ap.add_argument("--duration", type=int, default=10000, - help="timeline length in ms; pass the value used at scaffold time") - args = ap.parse_args() - pkg = Path(args.package).resolve() - if not pkg.is_dir(): - die(f"not a directory: {pkg}") - - manifest_path = find_manifest(pkg) - manifest = json.loads(manifest_path.read_text(encoding="utf-8")) - main_file = manifest.get("main") - if not main_file or not (pkg / main_file).exists(): - die(f"manifest 'main' ({main_file!r}) is missing next to the manifest — " - "an OGraf graphic is a folder; the .mjs must sit beside the .json") - duration = args.duration - settle = min(2000, duration // 2) - - # The renderer simulation: register under a fresh tag (catches self-registration), - # then load non-real-time and scrub. - test_html = f"""
    - """ - - try: - from playwright.sync_api import sync_playwright - except Exception: - manual_steps(pkg, "Playwright not installed") - return 3 - - (pkg / VERIFY_HTML).write_text(test_html, encoding="utf-8") - httpd, port = serve(pkg) - shot = Path(tempfile.gettempdir()) / f"ograf-verify-{manifest.get('id','graphic')}.png" - try: - with sync_playwright() as pw: - try: - browser = pw.chromium.launch() - except Exception as e: - manual_steps(pkg, f"no headless browser: {e}") - return 3 - page = browser.new_page(viewport={"width": args.width, "height": args.height}) - console_errors = [] - page.on("console", lambda m: console_errors.append(m.text) if m.type == "error" else None) - page.goto(f"http://127.0.0.1:{port}/{VERIFY_HTML}", wait_until="networkidle") - page.wait_for_function("window.__done === true", timeout=8000) - result = page.evaluate("window.__result") - page.screenshot(path=str(shot), omit_background=True) - browser.close() - finally: - httpd.shutdown() - (pkg / VERIFY_HTML).unlink(missing_ok=True) - - errors = (result.get("errors") or []) + console_errors - ok = result.get("ok") and not errors - print(json.dumps({ - "ok": bool(ok), - "status": "verified" if ok else "failed", - "package": str(pkg), - "visible": result.get("visible"), - "errors": errors, - "screenshot": str(shot), - "hint": None if ok else ( - "Self-registration error ('already been used with this registry') means the .mjs " - "calls customElements.define() — remove it. A blank render usually means an asset " - "load threw (wrap it in try/catch) or the host was sized to 0."), - }, indent=2)) - return 0 if ok else 1 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/skills/mc-outline/SKILL.md b/skills/mc-outline/SKILL.md index c56593f..e11cf26 100644 --- a/skills/mc-outline/SKILL.md +++ b/skills/mc-outline/SKILL.md @@ -1,27 +1,47 @@ --- name: mc-outline -description: Produce hook candidates, a tight outline, and the title/thumbnail promise for a Manticore project, then STOP for gate 1 approval. Use at the outline stage. Never writes the script. +description: Draft hooks, the outline, and the packaging promise. Use at the outline stage, or when the user says "outline this", "write the outline", or "what is the hook". --- # mc-outline -Gate 1. The deliverable is a decision artifact for the creator, not a script. +Gate 1. The outcome is `outline.md`: three hooks, one outline, and the packaging promise, in a form the creator can approve, edit, or kill before a line of script exists. It is a decision artifact, not a draft script. Everything in it traces to `braindump.md`, because the script stage may only use words already spoken there. -## Steps +## Resolution rules -1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there) and this skill's own surface (`uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}`; run `{workflow.activation_steps_prepend}` now, `{workflow.activation_steps_append}` after this step, and hold `{workflow.persistent_facts}` as standing context). Resolve `paths` values against `{project-root}`. Read `project.json`, `braindump.md`, `{brand-path}/voice-bible.md` (hook section, if built), and the format profile. Confirm stage is `outline`. -2. Write 3 hook candidates. Each is built Target-Transformation-Stakes (who it is for, what they will be able to do, why it matters now) and uses the creator's braindump phrasing wherever a phrase fits. Note which braindump line each hook leans on. -3. Write ONE outline (not options): tight beat list from hook to payoff. Every beat cites the braindump passage that fills it. Order for retention: strongest material early, open loops closed late, no throat-clearing beat at the top. Where the braindump names something the viewer should see (a demo, a screen, a drawing, motion, a moment the creator pictures as a graphic), attach an optional visual-moment note to the beat, citing the braindump passage and marked non-binding: mc-beats reads these as candidate compositions, not commitments. -4. Write the packaging promise: the working title and thumbnail concept this video must pay off. Add a CTA plan line drawn from `[cta]` in the studio config: which configured CTA(s) this video will make and roughly where, sized to the configured appetite; if no CTAs are configured, the line says so. If the video cannot pay off a clickable promise, say so now; that is a project problem, not a packaging problem. -5. Write it all to `outline.md`, set `approvals.outline = "pending"`, update `artifacts`, present to the creator, and STOP. +- Bare paths resolve against `{video-path}`, the current video project at `{projects-path}//`. +- `{skill-root}` → this skill's installed directory; files in it always carry it (`{skill-root}/scripts/lint_script.py`). +- `{project-root}` → the project working directory. -## Gate behavior +## On Activation -Do not start the script. Do not describe what the script will say. Wait for the creator to approve, edit, or kill. On approval, record the ISO date in `approvals.outline`, append `outline` to `stages_done`, and set `stage` to the next entry in project.json's `stages` array. +1. Resolve the studio config: `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`. Empty means mc-setup has not run: stop and route the creator there. Its `paths` values resolve against `{project-root}`. +2. Read `project.json` and confirm stage is `outline`, then `braindump.md` and the format profile. +3. Read `{brand-path}/voice-bible.md`. If it does not exist, tell the creator it is missing and that hooks in their voice cannot happen without it, then route to mc-setup and stop. Its hook section governs the candidates below. +4. Read `{brand-path}/blacklist.md`. If it does not exist, tell the creator it is missing and that the blacklist lint before gate 1 cannot happen without it, then route to mc-setup and stop. -## Checklist +## Three hooks -- Exactly 3 hooks, one outline, one packaging promise with a CTA plan line. -- Every outline beat has a braindump citation; beats without material are marked GAP with a question to ask the creator. -- Visual-moment notes, where present, cite braindump material and are marked non-binding. -- Run `uv run {skill-root}/scripts/lint_script.py {projects-path}//outline.md --blacklist {brand-path}/blacklist.md`; nothing in the artifact may violate the blacklist. +Three candidates, each built Target-Transformation-Stakes: who it is for, what they will be able to do, why it matters now. Use the creator's braindump phrasing wherever a phrase fits, and note which braindump line each hook leans on. + +## One outline + +One outline, not options: a tight beat list from hook to payoff. Every beat cites the braindump passage that fills it, and a beat with no material behind it is marked GAP with the question to ask the creator. Order for retention, with no throat-clearing beat at the top and open loops closed late. + +Where the braindump names something the viewer should see, attach a visual-moment note to that beat, citing the passage and marked non-binding. mc-beats reads these as candidate compositions, not commitments. + +## The packaging promise + +The working title and thumbnail concept this video must pay off, plus a CTA plan line drawn from `[cta]` in the studio config: which configured CTA(s) this video will make and roughly where, sized to the configured appetite. If no CTAs are configured, the line says so. + +If the video cannot pay off a clickable promise, say so now; that is a project problem, not a packaging problem. + +## Before presenting + +`uv run {skill-root}/scripts/lint_script.py {projects-path}//outline.md --blacklist {brand-path}/blacklist.md`. Exit 1 lists blacklist violations; fix every one. + +## Gate 1 + +Write it all to `outline.md`, set `approvals.outline = "pending"`, update `artifacts`, present to the creator, and STOP. Do not start the script. Do not describe what the script will say. + +Only the creator's explicit say-so moves this gate. On approval, record the ISO date in `approvals.outline`, append `outline` to `stages_done`, and set `stage` to the next entry in project.json's `stages` array. diff --git a/skills/mc-outline/customize.toml b/skills/mc-outline/customize.toml deleted file mode 100644 index eeea6d2..0000000 --- a/skills/mc-outline/customize.toml +++ /dev/null @@ -1,20 +0,0 @@ -# DO NOT EDIT -- overwritten on every update. -# -# Customization surface for mc-outline. -# Override files (not edited here): -# {project-root}/_bmad/custom/mc-outline.toml (team) -# {project-root}/_bmad/custom/mc-outline.user.toml (personal) -# -# Studio-wide config (owner, paths, editor, transcription, assets, tools) -# lives in [modules.manticore] in {project-root}/_bmad/custom/config.toml, -# maintained by mc-setup. This file holds only this skill's own surface. - -[workflow] - -# Steps to run before / after the standard activation. -activation_steps_prepend = [] -activation_steps_append = [] - -# Persistent facts held for the whole run: literal sentences, or -# "file:{project-root}/..." paths whose contents are loaded as facts. -persistent_facts = [] diff --git a/skills/mc-package/SKILL.md b/skills/mc-package/SKILL.md index 387b606..2e67838 100644 --- a/skills/mc-package/SKILL.md +++ b/skills/mc-package/SKILL.md @@ -1,70 +1,163 @@ --- name: mc-package -description: Produce title+thumbnail packages, description, CTA metadata, and chapters for a Manticore project; series A/B pairs, dual-timeline chapters, and a live-event mode for scheduled broadcasts. Use at the package stage (may start any time after gate 1, since the packaging promise exists from the outline). +description: Produce titles, thumbnails, description, chapters, and metadata. Use at the package stage or any time after gate 1, or when the user says "title ideas", "thumbnail", or "package this". --- # mc-package -Packaging pays off the promise approved at gate 1; it is not invented fresh here. Read `references/cta-placement.md` in full before writing the description, the pinned comment, or the end-screen guidance (the file is duplicated from mc-beats; keep both copies identical). Two flows: the VOD flow (steps 1 to 10) and the live-event flow for scheduled broadcasts (see Live-event mode below). - -## Steps - -1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there) and this skill's own surface (`uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}`; run `{workflow.activation_steps_prepend}` now, `{workflow.activation_steps_append}` after this step, and hold `{workflow.persistent_facts}` as standing context; the `[packaging]` keys below come from the same resolution). Resolve `paths` values against `{project-root}`. Read `project.json`, then branch on the promise source: - - When the project's `stages` includes `outline`: confirm `approvals.outline` is an ISO date; if null or `"pending"`, stop and route the creator back to the outline gate. Read `outline.md` (the packaging promise) and `script.md`. - - When it does not (footage-first and livestream projects have no outline stage): the promise comes from the footage itself; derive it from the final transcript, packaging only what the video actually delivers. - - Also read: the final transcript if the cut exists, the format profile at `{formats-path}/.md`, `{brand-path}/tokens.json`, `{brand-path}/production-bible.md` (thumbnail style, the series templates it records, and the CTA section), `{brand-path}/headshots/` with its `index.md`, and from the studio config `[cta]` (inventory and appetite) and `[owner]` `links`. Mode check: when the format is `livestream-pack` (or `stages` contains `stream-pack`), this is a scheduled broadcast; run step 2, then jump to Live-event mode below. -2. Series template. When `series` in `project.json` is set, read `{brand-path}/templates/.md` (see The series template contract below): its locked anchors bind every candidate in step 3 and 4; its per-episode variables are what this episode fills in. When `series` is set but the template file is missing, flag it, draft one from the Production Bible's series notes plus this episode's choices, save it to `{brand-path}/templates/`, and tell the creator the next episode inherits it. -3. Titles: `{packaging.candidates}` candidates (default 3) that pay off the approved promise. Under `{packaging.title-max-chars}` characters, front-loaded, no clickbait the video cannot cash. In a series, every candidate conforms to the template's locked title pattern. -4. Thumbnails, the locked flow. Face-plus-hook is the default treatment: an approved headshot plus a 2 to 4 word hook (`{packaging.hook-words-max}` is the cap). - - Headshots: use ONLY approved headshots from `{brand-path}/headshots/`, picking the expression from `index.md` that matches the hook's emotion. When `headshots/` is missing or empty, flag it loudly: face-plus-hook is blocked, and mc-setup's headshot collection step is the fix. Proceed with a non-face treatment only on the creator's explicit say-so. - - Draft, built programmatically: author each draft as a self-contained SVG or HTML composition themed from `tokens.json` (background, layout anchors, the hook text set in real type, the chosen headshot placed in the layout) and render it. The draft exists so the text is pixel-accurate: never ask a generative model to render the hook text. Drafts land in `packaging/work/`. - - Improvement pass, ALWAYS: run every draft through the creator's configured image lane (`[assets]` `image-provider`, resolved to a `[[tools]]` entry and driven EXACTLY per its `headless` string and `notes`, or the API provider), passing the draft plus the ORIGINAL reference images (the chosen headshot photo, past blessed thumbnails when they exist) and an explicit mandate: use the person in the headshot image, optimize this thumbnail for clicks, keep the hook text verbatim. The pass may recompose, relight, and exaggerate. When it mangles the text, composite the text back programmatically over the improved image; never regenerate just to fix text. - - Revisions start clean: when a candidate needs a change, re-send the SAME original inputs (draft, original headshot, references) with one improved prompt. Never pass a previous improved output back in as the base; a revision of a revision degrades like a photocopy of a photocopy. - - 120px verification, mandatory: `uv run {skill-root}/scripts/verify_thumb.py --out-dir /packaging/work --width {packaging.verify-width}` on every candidate, then LOOK at the proof image it writes before presenting anything. No thumbnail ships unseen at 120px. A hook that is not instantly readable in the proof sends the candidate back to the draft or improvement step. - - The presented candidates go to `packaging/thumbs/`; every draft, retry, and proof stays in `packaging/work/`. -5. Pairing and A/B: present the candidates as title+thumbnail PAIRS (title A with thumbnail A). Within a pair the two complement and never repeat each other: they share attention, not words. Series projects present exactly 3 pairs built on the template's locked anchors, ready for YouTube's Test & Compare to run as pairs; recommend which pair to lead with and why. -6. Description: the first 2 lines carry the hook and the search terms (they show before the fold), and when the video has a conversion CTA its link goes there too (description-top is half of its click surface, per the reference). Then the CTA lines drawn from `[cta]` items in priority order (imperative plus benefit, 7 words or fewer of ask copy per item); then the creator's `[owner]` `links`, in order; then the chapters block. Copy matches the lane: never live framing on a VOD ("enjoying the stream?", "link in chat" are wrong on a replay; use "comment below", "link in the description", schedule-tied subscribe framing), and livestream-vod projects get replay framing throughout. -7. Pinned comment and end screen, to `packaging/cta.md`: a paste-ready pinned-comment suggestion pointing at the same next step as the description-top link (identical URL; end screen, cards, pinned comment, and description-top all point at one next step), plus end-screen guidance for upload: the final 10 to 20 seconds are the outro runway, a 2-element layout (one watch-next plus one subscribe) beats cluttered screens, the watch-next target must be topically continuous, and the narration must verbally bridge to it. Check the script or transcript for that verbal bridge and flag loudly when it is missing. -8. Chapters: from the edited transcript's beat boundaries; first chapter 0:00, honest labels, no keyword stuffing. Dual-timeline rule: whenever `cut/edl.json` exists, chapters are a dual-timeline deliverable. `packaging/chapters.md` opens with the paste-ready block in edited (published) timecodes, followed by a clearly labeled table adding the original-source timecode per chapter (for finding the moment in the raw footage or VOD). The original column comes from this skill's own remap utility (a duplicate of the cut stage's, per the script-duplication convention), run against `cut/edl.json` (a project file): write the edited-timecode chapter list to `packaging/work/chapters-edited.md`, run `uv run {skill-root}/scripts/remap_timecode.py cut/edl.json --direction clean-to-orig --chapters packaging/work/chapters-edited.md -o packaging/work/chapters-orig.md`, and pair the two files line by line into the table. On a multi-source EDL, add a source column: use the script's `--events` mode instead (it records `source` on each remapped entry). If the cut does not exist yet (early run), chapters are pending: skip this step and the description's chapter block, and tell the creator to re-run mc-package after the cut to finish them. -9. Captions and transcript (only when the cut exists, the same gate as chapters: `cut/edl.json` and `transcript/words.json` present). Run `uv run {skill-root}/scripts/captions.py cut/edl.json --words transcript/words.json --out-dir packaging/captions/`. Multi-source projects pass one `--words` per source words file (`transcript/.words.json`), binding explicitly when media fields differ: `--words raw/=transcript/.words.json`. The script emits `packaging/captions/final.srt`, `final.vtt`, and `transcript.md` for the EDITED timeline; a light filler/stutter cleanup runs by default on this derived rendition only (`transcript/words.json` is never modified). Offer the creator `--no-clean` if they want verbatim captions. Present `transcript.md` for a skim before calling the deliverable done. If the cut does not exist yet, skip like chapters and finish captions on the re-run. -10. Write `packaging/titles.md`, `packaging/description.md`, `packaging/cta.md`, `packaging/chapters.md`, and `packaging/captions/` (the last two only when the cut existed to produce them); update `artifacts` in `project.json`. If the project's stage is `package` and chapters are done, append `package` to `stages_done` and set `stage` to the next stage in the project's `stages`; on an early run, leave `stage` and `stages_done` untouched. +Act as the creator's packaging partner. The outcome is the click surface for the video, under +`packaging/`: titles, thumbnails, description, CTA metadata, chapters, and captions. + +Two consumers set the bar. A browse feed decides in a fraction of a second at thumbnail size, so +every candidate has to survive being small. The creator uploading has to be able to paste each +file straight into YouTube. Packaging pays off the promise approved at gate 1; it is never +invented fresh here. + +## Resolution rules + +- Bare paths resolve against `{video-path}`, the current video project at `{projects-path}//`. +- `{skill-root}` → this skill's installed directory; files in it always carry it (`{skill-root}/references/cta-placement.md`). +- `{project-root}` → the project working directory. + +## On Activation + +1. Load the studio config: `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`. Empty means mc-setup has not run; stop and route the creator there. Resolve `paths` values against `{project-root}`. +2. Read `project.json` and find the promise. When `stages` includes `outline`, `approvals.outline` must be an ISO date; null or `"pending"` stops the run and routes the creator back to the outline gate. Then read `outline.md` (the packaging promise) and `script.md`. When `stages` has no outline stage (footage-first and livestream projects), the promise comes from the footage: derive it from the final transcript and package only what the video actually delivers. +3. Read the final transcript if the cut exists, the format profile at `{formats-path}/.md`, and from the studio config `[packaging]` (the caps and counts below), `[cta]` (inventory and appetite) and `[owner]` `links`. +4. Read `{brand-path}/tokens.json`. If it does not exist, tell the creator it is missing and that theming the thumbnail drafts cannot happen without it, then route to mc-setup and stop. +5. Read `{brand-path}/production-bible.md`. If it does not exist, tell the creator it is missing and that thumbnail style, the series templates it records, and the CTA section cannot happen without it, then route to mc-setup and stop. +6. Read `{brand-path}/headshots/`. If it does not exist, tell the creator it is missing and that face-plus-hook thumbnails cannot happen without it, then route to mc-setup and stop. Its `index.md` catalogs which expression is which. +7. Read `{brand-path}/blacklist.md`. If it does not exist, tell the creator it is missing and that the blacklist lint on the written copy cannot happen without it, then route to mc-setup and stop. +8. Route the branches. Format `livestream-pack` (or `stages` containing `stream-pack`) is a scheduled broadcast: work from `{skill-root}/references/live-event.md` instead of the sections below. A non-null `series` binds every candidate to that series' locked anchors: load `{skill-root}/references/series-template.md`. Read `{skill-root}/references/cta-placement.md` in full before writing the description, the pinned comment, or the end-screen guidance. + +## Titles + +`[packaging] candidates` candidates that pay off the approved promise, under +`[packaging] title-max-chars` characters, front-loaded, no clickbait the video cannot cash. In a +series, every candidate conforms to the template's locked title pattern. + +## Thumbnails + +Face-plus-hook is the default treatment: an approved headshot plus a 2 to 4 word hook, capped at +`[packaging] hook-words-max` words. The flow is locked. + +- Headshots come ONLY from `{brand-path}/headshots/`, picking the expression from `index.md` that + matches the hook's emotion. When `{brand-path}/headshots/` is missing or empty, flag it loudly: face-plus-hook + is blocked and mc-setup's headshot collection step is the fix. A non-face treatment proceeds only + on the creator's explicit say-so. +- Draft programmatically: author each draft as a self-contained SVG or HTML composition themed from + `{brand-path}/tokens.json`, with the chosen headshot placed in the layout, render it, and land it in + `packaging/work/`. The draft exists so the text is pixel-accurate; never ask a generative model to + render the hook text. +- Improvement pass, ALWAYS: run every draft through the creator's configured image lane (`[assets]` + `image-provider`, resolved to a `[[tools]]` entry and driven EXACTLY per its `headless` string and + `notes`, or the API provider). Pass the draft plus the ORIGINAL reference images (the chosen + headshot photo, past blessed thumbnails when they exist) and an explicit mandate: use the person + in the headshot image, optimize this thumbnail for clicks, keep the hook text verbatim. When it + mangles the text, composite the text back programmatically over the improved image; never + regenerate just to fix text. +- Revisions start clean: re-send the SAME original inputs (draft, original headshot, references) + with one improved prompt. Never pass a previous improved output back in as the base; a revision of + a revision degrades like a photocopy of a photocopy. +- 120px verification, mandatory, on every candidate: + `uv run {skill-root}/scripts/verify_thumb.py --out-dir /packaging/work --width <[packaging] verify-width>`. + Then LOOK at the proof image it writes before presenting anything. A hook that is not instantly + readable in the proof goes back to the draft or the improvement step. + +Presented candidates go to `packaging/thumbs/`; every draft, retry, and proof stays in +`packaging/work/`. + +## Pairing and A/B + +Present the candidates as title+thumbnail PAIRS (title A with thumbnail A). Within a pair the two +complement and never repeat each other: they share attention, not words. Series projects present +exactly `[packaging] candidates` pairs built on the template's locked anchors, ready for YouTube's +Test & Compare to run as pairs; recommend which pair to lead with and why. + +## Description + +The first 2 lines carry the hook and the search terms (they show before the fold), and when the +video has a conversion CTA its link goes there too, because description-top is half of its click +surface. Then the CTA lines drawn from `[cta]` items in priority order (imperative plus benefit, 7 +words or fewer of ask copy per item); then the creator's `[owner]` `links`, in order; then the +chapters block. + +Copy matches the lane. Never live framing on a VOD ("enjoying the stream?", "link in chat" are +wrong on a replay; use "comment below", "link in the description", schedule-tied subscribe +framing), and livestream-vod projects get replay framing throughout. + +## Pinned comment and end screen + +To `packaging/cta.md`: a paste-ready pinned-comment suggestion pointing at the same next step as +the description-top link (identical URL; end screen, cards, pinned comment, and description-top all +point at one next step), plus end-screen guidance for upload. The final 10 to 20 seconds are the +outro runway, a 2-element layout (one watch-next plus one subscribe) beats cluttered screens, the +watch-next target must be topically continuous, and the narration must verbally bridge to it. Check +the script or transcript for that verbal bridge and flag loudly when it is missing. + +## Chapters + +From the edited transcript's beat boundaries: first chapter 0:00, honest labels, no keyword +stuffing. + +Whenever `cut/edl.json` exists, chapters are a dual-timeline deliverable. `packaging/chapters.md` +opens with the paste-ready block in edited (published) timecodes, followed by a clearly labeled +table adding the original-source timecode per chapter, for finding the moment in the raw footage or +VOD. The original column comes from this skill's own remap utility, run against `cut/edl.json`: +write the edited-timecode chapter list to `packaging/work/chapters-edited.md`, then + +``` +uv run {skill-root}/scripts/remap_timecode.py cut/edl.json --direction clean-to-orig --chapters packaging/work/chapters-edited.md -o packaging/work/chapters-orig.md +``` + +and pair the two files line by line into the table. On a multi-source EDL, use the script's +`--events` mode instead and add a source column (it records `source` on each remapped entry). + +If the cut does not exist yet (an early run), chapters are pending: skip this section and the +description's chapter block, and tell the creator to re-run mc-package after the cut to finish them. + +## Captions and transcript + +Needs the same gate as chapters: `cut/edl.json` and `transcript/words.json` present. Skip like +chapters and finish on the re-run when they are not. + +``` +uv run {skill-root}/scripts/captions.py cut/edl.json --words transcript/words.json --out-dir packaging/captions/ +``` + +Multi-source projects pass one `--words` per source words file +(`transcript/.words.json`), binding explicitly when media fields differ: +`--words raw/=transcript/.words.json`. + +The script emits `packaging/captions/final.srt`, `final.vtt`, and `transcript.md` for the EDITED +timeline. A light filler/stutter cleanup runs by default on this derived rendition only; +`transcript/words.json` is never modified. Offer the creator `--no-clean` if they want verbatim +captions. Present `transcript.md` for a skim before calling the deliverable done. + +## Finish + +Lint the written copy: `uv run {skill-root}/scripts/lint_script.py --blacklist {brand-path}/blacklist.md` +on `titles.md`, `description.md`, and `cta.md`. + +Write `packaging/titles.md`, `packaging/description.md`, `packaging/cta.md`, +`packaging/chapters.md`, and `packaging/captions/` (the last two only when the cut existed to +produce them); update `artifacts` in `project.json`. If the project's stage is `package` and +chapters are done, append `package` to `stages_done` and set `stage` to the next stage in the +project's `stages`; on an early run, leave `stage` and `stages_done` untouched. ## The pick and blessed slots -Candidates accumulate in `packaging/thumbs/` and `titles.md`; the pick can happen immediately or after Test & Compare results come back. Whenever the creator declares the winners, write exactly one blessed asset per slot to `packaging/final/` (`packaging/final/title.txt`, `packaging/final/thumbnail.png`), record them in `project.json` `artifacts` (`"title"`, `"thumbnail"`), and set the project's `title` field to the blessed title. Alternates, drafts, and retries stay in `packaging/thumbs/` and `packaging/work/`; nothing downstream ever has to guess which asset shipped, because the deliverable path is `packaging/final/` and nothing else. +Candidates accumulate in `packaging/thumbs/` and `titles.md`; the pick can happen immediately or +after Test & Compare results come back. Whenever the creator declares the winners, write exactly +one blessed asset per slot to `packaging/final/` (`packaging/final/title.txt`, +`packaging/final/thumbnail.png`), record them in `project.json` `artifacts` (`"title"`, +`"thumbnail"`), and set the project's `title` field to the blessed title. Alternates, drafts, and +retries stay in `packaging/thumbs/` and `packaging/work/`, so nothing downstream ever has to guess +which asset shipped. -## Live-event mode (scheduled broadcasts) +## What no script can check -For livestream-pack projects, packaging serves the scheduled broadcast, not a finished video: - -1. Apply the series template (step 2) when the show belongs to a series; recurring shows usually do. -2. Produce one title (locked anchors apply) and one description. This is the live lane, so live framing is correct here (chat asks, the schedule, membership mentions), with the CTA lines and `[owner]` `links` per step 6. -3. Produce ONE scheduled-broadcast thumbnail through the full step 4 flow: face plus a 2 to 4 word hook, programmatic draft, mandatory improvement pass, mandatory 120px verification. -4. The two-asset rule, explicit: the scheduled-broadcast thumbnail competes in browse and search exactly like a VOD thumbnail and gets the full face-plus-hook treatment; it is NEVER a plain brand card. The plain branded card with the countdown safe zone is a different asset with a different job: the in-stream Starting Soon SCENE, produced by the stream-pack stage, not here. Never present one asset for both jobs, and never let the scene card become the broadcast thumbnail. -5. No chapters (nothing is cut). Write `packaging/titles.md`, `packaging/description.md`, and the thumbnail per the folder rules above; update `artifacts`. Touch `stage` and `stages_done` ONLY when `package` is the project's current stage AND appears in its `stages` array, mirroring the VOD flow's step 10 guard; livestream-pack has no `package` stage (its stages are `new`, `stream-pack`, `final`, `retro`), so leave `stage` and `stages_done` untouched and let mc-stream-pack advance the lane on the creator's gate-4 approval. Blessed slots apply once the creator approves the assets. - -## The series template contract - -One file per series at `{brand-path}/templates/.md`, filename matching the `series` value in `project.json`. mc-setup's brand scaffold creates the `templates/` folder and one file per recurring series the creator names; mc-package is the consumer. The shape both sides honor: - -- Locked anchors: everything each episode repeats so the series reads as a set in a feed. Thumbnail layout constants (face position and scale, wordmark or episode badge placement, background treatment, palette accents drawn from `tokens.json`) and the title pattern (fixed prefix, suffix, or numbering scheme). -- Per-episode variables: the slots each episode fills. Hook words, episode-specific imagery, guest name, episode number. - -Locked anchors are non-negotiable within an episode; changing them is a series-level decision that routes through mc-retro into the template, ISO dated, so packaging wins compound across the series. - -## Checklist - -- Every title pays off something the video actually delivers. -- Within every pair, thumbnail text and title complement and do not repeat each other (they share attention, not words). -- Every presented thumbnail has a verify_thumb.py proof that was actually viewed; no thumbnail ships unseen at 120px. -- Every thumbnail hook is 2 to 4 words and survived the improvement pass verbatim. -- Every thumbnail revision was regenerated from the original inputs (draft, original headshot) with a revised prompt; no improved output was ever fed back in as a base. -- Face-plus-hook thumbnails use only approved headshots from `{brand-path}/headshots/`; the missing-headshots case was flagged loudly, not worked around silently. -- The description's first 2 lines carry the hook, search terms, and the conversion link when one exists; CTA copy matches the lane (no live framing on a VOD). -- Pinned comment, description-top, and end-screen guidance all point at the same next step. -- Chapters are dual-timeline whenever an EDL exists; the original column came from this skill's remap_timecode.py run against `cut/edl.json`. -- Captions and transcript (`packaging/captions/final.srt`, `final.vtt`, `transcript.md`) were emitted whenever the cut exists; the cleanup pass touched only the caption rendition, never `transcript/words.json`. -- A scheduled-broadcast thumbnail is never a plain brand card (the two-asset rule). -- After a pick: exactly one blessed asset per slot in `packaging/final/`, recorded in `project.json` `artifacts`. -- Run `uv run {skill-root}/scripts/lint_script.py --blacklist {brand-path}/blacklist.md` on titles.md, description.md, and cta.md. +- The 120px proof was actually looked at. `verify_thumb.py` writes it; only you can judge whether + the hook reads at that size. +- The narration verbally bridges to the watch-next target, checked in the script or transcript. +- `transcript.md` was skimmed before the captions deliverable was called done. diff --git a/skills/mc-package/customize.toml b/skills/mc-package/customize.toml deleted file mode 100644 index cf50eb7..0000000 --- a/skills/mc-package/customize.toml +++ /dev/null @@ -1,37 +0,0 @@ -# DO NOT EDIT -- overwritten on every update. -# -# Customization surface for mc-package. -# Override files (not edited here): -# {project-root}/_bmad/custom/mc-package.toml (team) -# {project-root}/_bmad/custom/mc-package.user.toml (personal) -# -# Studio-wide config (owner, paths, editor, transcription, assets, tools) -# lives in [modules.manticore] in {project-root}/_bmad/custom/config.toml, -# maintained by mc-setup. This file holds only this skill's own surface. - -[workflow] - -# Steps to run before / after the standard activation. -activation_steps_prepend = [] -activation_steps_append = [] - -# Persistent facts held for the whole run: literal sentences, or -# "file:{project-root}/..." paths whose contents are loaded as facts. -persistent_facts = [] - -[packaging] - -# Title+thumbnail candidates to present (series projects present them as -# exactly this many A/B pairs). -candidates = 3 - -# Hard cap on on-image thumbnail hook words (face-plus-hook convention). -hook-words-max = 4 - -# Title length cap in characters (front-load what matters; longer gets cut -# off in feeds). -title-max-chars = 60 - -# Proof width in pixels for the mandatory downscale verification -# (verify_thumb.py). 120 approximates the smallest feed rendering. -verify-width = 120 diff --git a/skills/mc-package/references/cta-placement.md b/skills/mc-package/references/cta-placement.md index a6d847a..b94494f 100644 --- a/skills/mc-package/references/cta-placement.md +++ b/skills/mc-package/references/cta-placement.md @@ -1,15 +1,13 @@ # CTA Placement Reference -Research-backed rules (2024-2026 era) for deciding when, where, and how to place CTA beats in a video, using the transcript and position-based retention logic. mc-beats reads this during its CTA placement pass; mc-package reads it for description lines, the pinned-comment suggestion, and end-screen guidance. Brand-agnostic; the creator's actual inventory comes from `[cta]` in the studio config and the Production Bible's CTA section. +Rules for deciding when, where, and how to place CTA beats in a video, using the transcript and position-based retention logic. mc-beats reads this during its CTA placement pass; mc-package reads it for description lines, the pinned-comment suggestion, and end-screen guidance. Brand-agnostic; the creator's actual inventory comes from `[cta]` in the studio config and the Production Bible's CTA section. ## Core principles 1. Earn before asking. CTAs convert best immediately after a moment of delivered value: a payoff, insight, demo result, or completed segment. Asking before value is delivered depresses both conversion and retention. 2. One primary CTA per video. Multiple CTAs are fine only when spaced apart, serving different purposes, with a clear hierarchy. Competing asks in the same window create choice paralysis and read as noise. -3. Verbal plus on-screen beats either alone. A spoken ask reinforced by a synchronized graphic outperforms voice-only or graphic-only CTAs. When the transcript contains a verbal CTA, always pair it with a graphic, synced to start within about half a second of the spoken words. A silent graphic is acceptable only for low-friction asks (subscribe bug, link-in-description lower third). +3. Verbal plus on-screen beats either alone, and beats no ask at all: a clear, non-pushy verbal ask lifts subscribe conversion by a few percent up to 30-40% relative. When the transcript contains a verbal CTA, always pair it with a graphic synced to start within about half a second of the spoken words. A silent graphic is acceptable only for low-friction asks (subscribe bug, link-in-description lower third). 4. Continue the journey, do not interrupt it. Retention graphs commonly dip at CTA moments; the dips come from jarring, disconnected asks, not from CTAs per se. A CTA framed as the natural next step holds retention; a hard sales pivot does not. -5. Never interrupt tension. No CTAs mid-explanation, mid-demo, during a build-up, or in high-information-density passages. Place them at natural seams: topic transitions, post-payoff moments, chapter boundaries. -6. Asking works. Controlled creator tests consistently show meaningful lift (a few percent up to 30-40% relative increase in subscribe conversion, in some tests a doubling) when a clear, non-pushy verbal ask is present versus absent. The graphic supports the spoken ask; it does not replace it. ## Placement zones by video position @@ -25,7 +23,7 @@ Avoid engagement CTAs; retention is still settling. Permitted: a brief content-r ### Zone C: mid-video peak zone (~25% to ~60%), the primary CTA zone -The primary engagement CTA belongs here, at a post-payoff seam near the 40-50% mark. Retention is typically at its healthiest and viewers have received tangible value; end-of-video placement wastes the ask because a large share of viewers never reaches it (retention collapses in the final 30 seconds). Trigger on payoff moments, not the clock: find the strongest completed value moment nearest 40-50% (a problem just solved, a demo that just worked, a section just wrapped) and anchor the CTA there. Reason-based asks outperform bare asks: prefer copy that states a benefit ("Subscribe for weekly deep dives") over a bare "Subscribe". Platform cards (the info teaser) also belong here, at 50-75% of runtime, tied to the moment a related topic is mentioned; use 1-2 at most. +The primary engagement CTA belongs here, at a post-payoff seam near the 40-50% mark. Trigger on payoff moments, not the clock: find the strongest completed value moment nearest 40-50% (a problem just solved, a demo that just worked, a section just wrapped) and anchor the CTA there. Reason-based asks outperform bare asks, so prefer copy that states a benefit ("Subscribe for weekly deep dives") over a bare "Subscribe". Platform cards (the info teaser) also belong here, at 50-75% of runtime, tied to the moment a related topic is mentioned; use 1-2 at most. ### Zone D: valleys and interior dips (anywhere in the body) @@ -46,7 +44,7 @@ Reserve the last 10-20 seconds as a deliberate outro runway: a talking-head or h - Minimum spacing: no two CTA graphics within 2 minutes or 20% of runtime of each other, whichever is larger. - Never stack: no two different asks within the same 30-second window. Rapid-fire "like AND subscribe AND join" is the canonical failure; one well-placed prompt outperforms three. - Duplicate suppression: if the speaker verbally asks for the same action more than once, graphic-support only the strongest instance (best zone per the rules above); leave the others voice-only, or flag them for trimming when editing is in scope. -- Forbidden zones: the first 30 seconds; mid-sentence; mid-demo or mid-tension; retention valleys; under the end screen. +- Forbidden zones: the first 30 seconds; mid-sentence; mid-demo, mid-explanation, information-dense passages, or any other build-up of tension; retention valleys; under the end screen. ## On-screen treatment @@ -62,7 +60,7 @@ Reserve the last 10-20 seconds as a deliberate outro runway: a talking-head or h | Goal | Best position | Why | |---|---|---| -| Subscribe / like (channel growth) | Mid-video, at the strongest payoff near 40-50% | Retention is highest; end-of-video asks miss most viewers | +| Subscribe / like (channel growth) | Mid-video, at the strongest payoff near 40-50% | Retention is highest; end-of-video asks miss most viewers, since retention collapses in the final 30 seconds | | Comment prompt | Mid-video, phrased as a specific question tied to the content | Specific questions drive meaningful comments; generic "comment below" does not | | Watch next / session continuation | End screen (final 10-20 s), optional card at 50-75% | Natural next step at content end | | Conversion (community, site, newsletter, product) | Pre-outro (~85-95%) as the spoken pitch; description and pinned comment as the click surface | Remaining viewers are most invested; the full video earns the ask | @@ -80,12 +78,11 @@ When the source is an edited livestream VOD: 4. Always add an end-screen runway. Raw stream endings have none; hold or extend the final shot to create the 10-20 second zone and point at a related VOD or highlight video. 5. Clip-to-full-video CTAs: any short or clip cut from the VOD ends with an on-screen CTA pointing to the full video. This is the highest-leverage CTA in a clipping pipeline. -## Confidence notes +## Overrides and trade-offs -- High confidence (multiple independent sources or platform mechanics): end screens limited to the final 5-20 s; simple 2-element end screens beat cluttered ones; verbal plus visual pairing beats either alone; no CTAs in the opening seconds; retention collapses in the final 30 seconds, making end-only subscribe asks weak. -- Medium confidence (creator experiments, tool-vendor studies): the size of the ask-vs-no-ask lift; animated CTAs outperforming static; benefit-framed asks strongly outperforming bare commands. -- Directional (practitioner consensus, no controlled data): exact zone percentages, the 2-minute spacing rule, the 3-CTA cap. Treat these as sane defaults; per-channel retention data overrides them when available. -- Known trade-off: even well-executed CTAs produce small retention dips. A small dip at a well-placed ask is an acceptable cost for conversion; a large dip means the ask was jarring, mistimed, or too long. +The zone percentages, the 2-minute spacing rule, and the 3-CTA cap are defaults; per-channel retention data overrides them wherever the creator has it. + +Even well-executed CTAs produce small retention dips. A small dip at a well-placed ask is an acceptable cost for conversion; a large dip means the ask was jarring, mistimed, or too long. ## Config schema @@ -112,14 +109,3 @@ priority = 1 4. Fill CTA beats from the configured inventory by priority: sync graphics to kept verbal CTAs first, then place silent-eligible low-friction CTAs at the best remaining seams; enforce spacing, caps, and one ask per window. 5. Reserve and validate the end-screen runway: the final shot must tolerate overlays, and the narration should verbally bridge to the watch-next target (flag it when it does not). 6. Emit CTA rows in the beat table with timestamps, anchors, transcript evidence, and rationale, for gate-3 approval like any other beat. - -## Sources - -- vidIQ, What Is a YouTube CTA? Definition, Examples, and How to Write One. https://vidiq.com/blog/post/youtube-cta/ -- Ventress, YouTube CTA Strategy 2025: Convert Viewers to Subscribers. https://ventress.app/blog/youtube-call-to-action-strategy-convert-viewers-subscribers/ -- Mark Brinker, The Real Reason YouTubers Obsess Over Likes and Subscribes (TubeBuddy ask-vs-no-ask experiment). https://www.markbrinker.com/youtube-engagement -- TubeAnalytics, YouTube Cards and End Screens Checklist. https://www.tubeanalytics.net/blog/youtube-cards-end-screens-checklist-for-retention -- Humble & Brag, YouTube End Screens: How to Set Them Up and Optimise Them. https://humbleandbrag.com/blog/youtube-end-screens -- OverseerOS, YouTube Retention Curve Audit. https://www.overseeros.com/blog/youtube-retention-curve-audit -- Viral Idea Marketing, YouTube Video Editing for Livestream Replays. https://www.viralideamarketing.com/post/youtube-video-editing-for-livestream-replays-how-to-cut-and-repurpose-content -- Restream, 9 Ways to Repurpose Your Live Video Content. https://restream.io/blog/repurpose-live-videos/ diff --git a/skills/mc-package/references/live-event.md b/skills/mc-package/references/live-event.md new file mode 100644 index 0000000..85e0b30 --- /dev/null +++ b/skills/mc-package/references/live-event.md @@ -0,0 +1,41 @@ +# Live-event mode (scheduled broadcasts) + +For `livestream-pack` projects, packaging serves a scheduled broadcast rather than a finished +video. The deliverables are one title, one description, and one broadcast thumbnail, all aimed at +a stream that has not happened yet. + +Load this instead of the VOD flow when the format is `livestream-pack` or `stages` contains +`stream-pack`. The Thumbnails, Description, and blessed-slot sections of `{skill-root}/SKILL.md` still govern +how each asset is made; this file says what changes. + +## What to produce + +1. Apply the series template first when the show belongs to a series; recurring shows usually do. + See `{skill-root}/references/series-template.md`. +2. One title (locked anchors apply) and one description. This is the live lane, so live framing is + correct here: chat asks, the schedule, membership mentions, alongside the CTA lines and + `[owner]` `links` the Description section specifies. +3. ONE scheduled-broadcast thumbnail through the full Thumbnails flow: face plus a 2 to 4 word + hook, programmatic draft, mandatory improvement pass, mandatory 120px verification. +4. No chapters and no captions; nothing is cut. + +## The two-asset rule + +The scheduled-broadcast thumbnail competes in browse and search exactly like a VOD thumbnail and +gets the full face-plus-hook treatment. It is NEVER a plain brand card. + +The plain branded card with the countdown safe zone is a different asset with a different job: the +in-stream Starting Soon SCENE, produced by the stream-pack stage, not here. Never present one +asset for both jobs, and never let the scene card become the broadcast thumbnail. + +## Finish + +Write `packaging/titles.md`, `packaging/description.md`, and the thumbnail per the folder rules in +`{skill-root}/SKILL.md`; update `artifacts` in `project.json`. + +Touch `stage` and `stages_done` ONLY when `package` is the project's current stage AND appears in +its `stages` array. livestream-pack has no `package` stage (its stages are `new`, `stream-pack`, +`final`, `retro`), so leave both untouched and let mc-stream-pack advance the lane on the +creator's gate-4 approval. + +Blessed slots apply once the creator approves the assets. diff --git a/skills/mc-package/references/series-template.md b/skills/mc-package/references/series-template.md new file mode 100644 index 0000000..733d84a --- /dev/null +++ b/skills/mc-package/references/series-template.md @@ -0,0 +1,27 @@ +# The series template contract + +One file per series at `{brand-path}/templates/.md`, filename matching the `series` value +in `project.json`. mc-setup's brand scaffold creates the `{brand-path}/templates/` folder and one file per +recurring series the creator names; mc-package is the consumer. + +Load this when `series` in `project.json` is set. Its locked anchors bind every title and +thumbnail candidate produced for the episode. + +## The shape both sides honor + +- Locked anchors: everything each episode repeats so the series reads as a set in a feed. + Thumbnail layout constants (face position and scale, wordmark or episode badge placement, + background treatment, palette accents drawn from `{brand-path}/tokens.json`) and the title pattern (fixed + prefix, suffix, or numbering scheme). +- Per-episode variables: the slots each episode fills. Hook words, episode-specific imagery, guest + name, episode number. + +Locked anchors are non-negotiable within an episode. Changing them is a series-level decision that +routes through mc-retro into the template, ISO dated, so packaging wins compound across the +series. + +## When the file is missing + +When `series` is set but the template file does not exist, flag it, draft one from the Production +Bible's series notes plus this episode's choices, save it to `{brand-path}/templates/`, and tell +the creator the next episode inherits it. diff --git a/skills/mc-pipeline/PIPELINE.md b/skills/mc-pipeline/PIPELINE.md index 4cdddab..acd90bf 100644 --- a/skills/mc-pipeline/PIPELINE.md +++ b/skills/mc-pipeline/PIPELINE.md @@ -1,13 +1,14 @@ # PIPELINE.md: The State Machine -The master spec for the Manticore pipeline, owned by mc-pipeline (the router). It defines the stages, the artifacts each stage produces, the approval gates, and the `project.json` contract. Each stage skill is self-contained and carries its own steps, but everything below is the contract they conform to. +The master spec for the Manticore pipeline, owned by mc-pipeline (the router). It defines the stages, what each hands to the next, the approval gates, and the `project.json` contract. Each stage skill is self-contained and carries its own steps; this is the contract they conform to. -Conventions used below: +This file is a contract, not a summary of the stages. It carries what crosses a stage boundary. Anything a stage produces and consumes entirely within itself belongs to that skill, and adding it here is how this file goes stale. -- The studio config is the `[modules.manticore]` table in `{project-root}/_bmad/custom/config.toml` (personal overrides in `config.user.toml`), created by mc-setup and resolved with `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`. Table names like `[owner]`, `[paths]`, `[video]`, `[render]`, `[style]`, `[cta]`, `[live]`, `[editor]`, `[transcription]`, `[assets]`, `[mcp]` refer to its sub-tables. (`[defaults.*]` names appear only inside mc-setup's `customize.toml`, the seed that mc-setup copies from; a resolved studio config has no `[defaults]` table.) +Three naming notes, because each has caused real confusion: + +- Bracketed table names (`[owner]`, `[paths]`, `[render]`, `[editor]`, and the rest) are sub-tables of `[modules.manticore]`, the studio config. `[defaults.*]` names appear only in `mc-setup/assets/studio-defaults.toml`, the seed it copies from; a resolved studio config has no `[defaults]` table. - `{projects-path}`, `{brand-path}`, `{formats-path}`, `{engines-path}` are the `[paths]` values resolved against `{project-root}`. If `[modules.manticore]` is empty, run mc-setup first; no stage skill proceeds without it. -- Per-skill defaults and overrides live in each skill's `customize.toml`, resolved with `uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}`. Skills read only their own folder and project files, never another skill's folder. -- "the creator" is the human owner configured in `[owner]`; skills address them by their configured name. +- A bare path in any skill file is the current video project: `{video-path}` = `{projects-path}//`. A file inside a skill's own folder always carries `{skill-root}`, so bare `assets/` is the project's farmed-asset folder and `{skill-root}/assets/` is the skill's own. A path led by a skill name (`mc-cut/scripts/preflight.py`) names a file in that skill's folder, and is the form any skill file uses to record which skill owns a script or a document; a skill still reads only its own folder. ## Stage sequence (master list) @@ -20,12 +21,12 @@ Format profiles select a subset of these stages (see the `stages:` frontmatter o | 3 | outline | mc-outline | gate 1: outline | `outline.md` (hooks + outline + packaging promise) | | 4 | script | mc-script | | `script.md` (lint passed, craft QA passed) | | 5 | record | the creator | | `raw/*` recordings, constant frame rate | -| 6 | cut | mc-cut | gate 2: cutplan | `transcript/words.json` (suffixed `.words.json` when a project has multiple sources), `cut/candidates.json`, `cut/cutplan.md`, `cut/edl.json`, `cut/rough.fcpxml` (per `[editor] timeline-format`; `none` skips), `renders/preview.mp4` (fast low-res preview, re-rendered each iteration; once stage 9 has rendered overlays, the router sends the project back through mc-cut to re-render it with graphics composited) | +| 6 | cut | mc-cut | gate 2: cutplan | `transcript/words.json`, `cut/edl.json`, `cut/cutplan.md`, `cut/editorial-review.md`, `cut/rough.fcpxml` (per `[editor] timeline-format`; `none` skips), `renders/preview.mp4` | | 7 | beats | mc-beats | gate 3: beats | `beats/beats.md` (the beat table), `beats/STORYBOARD.md` | | 8 | assets | mc-assets | | `assets/` + `assets/manifest.json` | -| 9 | graphics | mc-graphics | | `graphics/` alpha MOVs + `graphics/HANDOFF.md`; on completion the router routes through mc-cut to re-render `renders/preview.mp4` with the overlays composited | -| 10 | package | mc-package | | `packaging/titles.md`, `packaging/thumbs/`, `packaging/description.md`, `packaging/chapters.md`, `packaging/captions/` (final.srt, final.vtt, transcript.md, when the cut exists) | -| 11 | final | the creator, with an offered pipeline render | gate 4: final | `renders/final.mp4` (the offered final-quality render: same EDL, graphics composited from the beat table, delivery resolution per `[video]` delivery-resolution and codec per `[render]`, loudness-normalized to the `[render]` loudness-target unless loudnorm is off), or the creator's own editor render into `renders/` | +| 9 | graphics | mc-graphics | | `graphics/` alpha MOVs + `graphics/HANDOFF.md` | +| 10 | package | mc-package | | `packaging/titles.md`, `packaging/thumbs/`, `packaging/description.md`, `packaging/chapters.md`, `packaging/captions/` | +| 11 | final | the creator, with an offered pipeline render | gate 4: final | `renders/final.mp4`, or the creator's own editor render into `renders/` | | 12 | retro | mc-retro | | edits to `{formats-path}/.md` learnings + offending skill files | Stage 8 (assets) runs before stage 9 (graphics) so the farmed stills and clips exist before graphics composes with them; both unlock at gate 3. Stage 10 may start any time after gate 1 (the packaging promise exists from the outline). @@ -62,16 +63,16 @@ Field rules: - `approvals` values are `null` (not reached), `"pending"` (artifact presented, waiting on the creator), or an ISO date string (approved that day). Only the creator's explicit say-so in conversation moves pending to a date. - `artifacts` maps artifact names to paths as they are produced, e.g. `"edl": "cut/edl.json"`. - `parent` links a short to its long-form parent project slug. -- `sources` (optional) registers media inputs as they arrive: `{"id": "camera-a", "file": "raw/camera-a.mp4", "role": "primary", "cfr": true}`. Roles: `primary` (talking-head take), `interview` (a recorded braindump session; the creator reads each question aloud prefixed with the marker cue so the cut stage can segment it mechanically; the default cue is "question from the interviewer", configurable via cutplan.py `--marker-cues` and the setup interview; the older "question from claude" phrasing remains a documented alternative for studios that recorded with it), `screen` (screen share). Stages that ingest media append here. +- `sources` (optional) registers media inputs as they arrive: `{"id": "camera-a", "file": "raw/camera-a.mp4", "role": "primary", "cfr": true}`. Roles are `primary` (a talking-head take), `interview` (a recorded braindump, segmented mechanically by its spoken marker cue), and `screen` (screen share). Stages that ingest media append here, and a source corrected by mc-cut's `normalize_source.py` is registered as the project's source of truth. ## The stage skill algorithm Every mc-* stage skill follows the same shape. No exceptions, no creativity in the mechanics: -1. Resolve the studio config (`resolve_config.py --key modules.manticore`) and the skill's own surface (`resolve_customization.py --skill {skill-root}`); if the studio config is empty, stop and run mc-setup. -2. Read `project.json`. If the project's `stage` does not match this skill's stage, stop and say so (mc-pipeline routes; stage skills do not self-route). -3. Read the format profile at `{formats-path}/.md` and any taste files it names (all under `{brand-path}`). -4. Do the stage work, calling the scripts in the skill's own `scripts/` folder for anything mechanical. +1. Resolve the studio config (`resolve_config.py --key modules.manticore`); if it is empty, stop and run mc-setup. +2. Read `project.json`. If the project's `stage` does not match this skill's stage, stop and say so (mc-pipeline routes; stage skills do not self-route). The one exception is a declared ROUTED ENTRY POINT: a section a skill declares for the router to re-enter after its own stage has closed. Those touch no gates, no approvals and no stage fields, so they cannot advance or rewind the project, and a skill that declares none has no exception. mc-cut is the only skill with any today. +3. Read the format profile at `{formats-path}/.md` and any taste files it names (all under `{brand-path}`). A taste file that does not exist stops the stage: name the file, say what cannot happen without it, route to mc-setup. +4. Do the stage work, calling the stage skill's own scripts for anything mechanical. 5. Run the stage's checklist (in the skill file). Fix failures before presenting. 6. Write the artifacts to the paths in the table above. Update `artifacts` in project.json. 7. If the stage is a gate: set the approval to `"pending"`, present the artifact to the creator, and STOP. Do not proceed, do not start the next stage, do not summarize what the next stage will do. @@ -81,19 +82,25 @@ If the config exists but a key this stage needs is missing or empty, ask for jus ## Gate behavior -- Gate 1 (outline): the creator approves hook + outline + the title/thumbnail promise before any script is written. -- Gate 2 (cutplan): the creator approves the cut plan summary (the taste calls, e.g. "trailing 'so' at 42:20, keep or cut?") before the preview render and the exported timeline are treated as the rough cut. -- Gate 3 (beats): the creator approves the beat table before any graphics code is written. -- Gate 4 (final): Manticore offers the final-quality render (`renders/final.mp4`) and the creator approves the deliverable. Finishing in their own editor from the always-exported timeline is an equally supported path; approval of either closes the gate. Either way the timeline export, edl.json, cutplan, and overlay assets already exist, so switching paths never loses work. +What each gate blocks is the part other stages depend on: + +| Gate | Approves | Nothing may happen until it clears | +|---|---|---| +| 1: outline | hook, outline, the title/thumbnail promise | any script is written | +| 2: cutplan | the taste calls | the preview and timeline count as the rough cut | +| 3: beats | the beat table | any graphics code is written | +| 4: final | the deliverable | publishing | + +Gate 4 has two equally supported paths: the offered `renders/final.mp4`, or the creator's own editor render from the always-exported timeline. Approval of either closes it, and because the timeline, edl.json, cutplan and overlays all exist regardless, switching paths never loses work. ## Engine policy - HyperFrames: the graphics engine. Per-video overlay beats, stingers and transitions (dual render: VP9 alpha WebM for OBS + ProRes 4444 for the editor timeline lane), and shorts karaoke captions. Registry blocks before authoring (`npx hyperframes add`). Export overlay-only ProRes 4444 MOV with alpha. Apache 2.0, local, no commercial-use threshold. -- OGraf (the mc-ograf skill): ONLY when the target supports it. Editor lane requires `[editor] ograf-editable = true` (DaVinci Resolve 21+); the live lane (OBS/SPX-GC via mc-stream-pack) is editor-independent. Everyone else gets baked alpha MOVs, which work in every editor. +- Baked alpha MOVs are the deliverable everywhere, unconditionally. There is no editable-graphics lane and no editor-dependent branch: what one editor gets, every editor gets. - Everything is themed through `{brand-path}/tokens.json`. Component sourcing rule: registries and open libraries first, author from scratch only when nothing fits. - Engine workspaces (the pinned HyperFrames project) live at `{engines-path}`; mc-setup or the first graphics run initializes them. -- Remotion was a second engine through 0.x and was removed on 2026-07-22: its license is free only up to 3 people, and its React authoring model bought nothing in a frame-deterministic renderer. Rationale in `mc-graphics/engines/hyperframes.md`. -- Compatibility alias (unconditional, any vintage): `remotion` is a permanent alias for `hyperframes` wherever an engine is named — a beat-table `engine` value OR a format profile's `engine_overlays`/`engine_stingers` frontmatter. A studio configured before 2.0.0 keeps its own copied profiles that may still say `remotion`; every skill reads that as `hyperframes` and no creator file is rewritten. There is no Remotion engine doc or workspace to route to. +- Compatibility aliases (unconditional, any vintage): `remotion` and `ograf` are permanent aliases for `hyperframes` wherever an engine is named, whether a beat-table `engine` value or a format profile's `engine_overlays`/`engine_stingers` frontmatter. A studio configured before a given engine was dropped keeps its own copied profiles and beat tables naming it; every skill reads those as `hyperframes` and no creator file is ever rewritten. Neither has an engine doc or workspace to route to. +- `[editor] ograf-editable` is a retired key. A studio config written before 3.0.0 may still carry it; ignore it rather than acting on it, and never write it. ## The beat table (engine-neutral graphics contract) @@ -105,7 +112,7 @@ One row per graphic beat, produced by mc-beats, consumed by mc-graphics and mc-a Column rules: - `type` is a beat type from the format profile's `beat-types` frontmatter list (e.g. `lower-third`, `diagram`, `stat-card`, `cta`); the profile is the single type vocabulary for its format. The reserved placeholder `overlay` is legal only when reading legacy tables (tolerance rule below) and is never written. -- `engine` names the engine that renders the beat, per the Engine policy below (e.g. `hyperframes`, `ograf`, `html`). +- `engine` names the engine that renders the beat, per the Engine policy below (e.g. `hyperframes`, `html`). - `asset` is `null` or a farmed-asset id from `assets/manifest.json`; mc-assets farms the listed assets, mc-graphics composes with them. - Tolerance rule: consumers MUST accept rows missing `type`, `engine`, or `asset` (beat tables written by 0.x projects). Treat a missing `type` as the reserved placeholder `overlay` (informational only; rendering keys off `engine` and `composition`), a missing `engine` as the Engine policy default, an `engine` of `remotion` (from any vintage of table, per the Engine policy's compatibility alias) as `hyperframes`, and a missing `asset` as `null`. A stage that rewrites the table (mc-beats) replaces every `overlay` placeholder with a type from the profile's `beat-types`. An in-flight 0.x project never breaks on the extended contract. @@ -117,4 +124,24 @@ Deliverable folders hold exactly one blessed asset per slot; alternates, drafts, ## Cutting rules -The non-negotiable cutting rules (never cut inside a word, padding, fades, quote + reason per EDL segment, frame-verified boundaries, constant frame rate sources) live in the mc-cut skill, which is the only stage that applies them. +mc-cut is the only stage that cuts, so its rules and the gates enforcing them live there. One binds every stage, which is why it is here: the TRANSCRIPT is the authority on CONTENT, the AUDIO is the authority on TIMING. No stage derives a cut time, a beat time, or a silence from transcript timestamps. + +## Verification contract + +A check the pipeline claims to perform must be a script that exits non-zero; AGENTS.md carries the rule and the four shipped defects that produced it. What belongs here is the cross-stage view, because no single skill can see it: which gates block what, and where a blocked run can legitimately proceed. + +Every blocking gate carries an acknowledged override (`preflight.py --allow-qc-defects`, `verify_transcript.py --accept-region ... --reason ...`), so a false positive becomes a recorded human decision rather than a reason to work around the gate silently. + +| Check | Script | Blocks | +|---|---|---| +| Source edge defects | `mc-cut/scripts/preflight.py` (exit 3) | transcription and everything after | +| Transcript completeness | `mc-cut/scripts/verify_transcript.py` | candidate detection | +| Cut integrity | `mc-cut/scripts/verify_edl.py` | rendering, timeline export, gate 2 | +| Beat anchor placement | `mc-beats/scripts/verify_anchors.py` | gate 3 and graphics | +| Render output integrity | `mc-cut/scripts/render_preview.py` / `mc-cut/scripts/render_final.py` | publishing the output file | + +## Spatial normalize + +`cut/edl.json` is purely TEMPORAL: segments are `{source, start, end}` and the renderers expose no crop, scale, or position transform. A defect baked into the pixels (a recorded-in border, letterboxing, off-centre framing) therefore cannot be corrected anywhere downstream, and every later stage inherits whatever canvas the cut stage leaves it. + +That is why correction is the cut stage's job and mc-cut owns the mechanics. Two consequences bind other stages: a normalize moves nothing in time, so an existing transcript, EDL, cutplan and beat table all stay valid across one; and corrective normalize is global and defect-driven, where creative reframing (punch-ins, motion zooms) is per-moment emphasis belonging to the beats stage, on the already-clean canvas. diff --git a/skills/mc-pipeline/SKILL.md b/skills/mc-pipeline/SKILL.md index b816fd9..5608072 100644 --- a/skills/mc-pipeline/SKILL.md +++ b/skills/mc-pipeline/SKILL.md @@ -1,21 +1,27 @@ --- name: mc-pipeline -description: Manticore pipeline manager. Reports where a video project stands and routes to the next stage skill. Zero creative instruction. Use for "where is my project", "what's next", "run the pipeline", or when unsure which stage skill applies. +description: Report project state and route the next stage. Use when the user says "where is my project", "what is next", or "run the pipeline". --- # mc-pipeline The manager. Contains no creativity and makes no taste calls. +## Resolution rules + +- Bare paths resolve against `{video-path}`, the current video project at `{projects-path}//`. +- `{skill-root}` → this skill's installed directory; files in it always carry it (`{skill-root}/PIPELINE.md`). +- `{project-root}` → the project working directory. + ## Steps -1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there) and this skill's own surface (`uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}`; run `{workflow.activation_steps_prepend}` now, `{workflow.activation_steps_append}` after this step, and hold `{workflow.persistent_facts}` as standing context). Resolve `paths` values against `{project-root}`. Read `{skill-root}/PIPELINE.md` (the stage table and project.json contract) if not already in context. +1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there). Resolve `paths` values against `{project-root}`. Read `{skill-root}/PIPELINE.md` (the stage table and project.json contract) if not already in context. 2. If no project was named: list `{projects-path}/*/project.json`, show a one-line status per project (slug, format, stage, pending approvals), and stop. 3. For the named project, read `project.json` and report: - current `stage` and what artifact it produces, - any approval sitting at `"pending"` (the creator owes a decision; nothing moves until they give it), - artifacts produced so far. -4. Route: name the mc-* skill for the current stage (see the stage table in PIPELINE.md) and invoke it if the creator asked to proceed. Route by the project's own `stages` array, never the master list: footage-first projects (created by mc-new's ingest mode) carry the ingest-first list (`new`, `cut`, `beats`, `graphics`, `assets`, `package`, `final`, `retro`) and never visit the ideation stages. One standing extra hop: when the graphics stage completes (mc-graphics hands back with `graphics/` holding rendered overlays), route through mc-cut's composited preview re-render (its "Composited preview (after graphics)" section) before invoking the next stage skill, and again whenever an overlay is later re-rendered. +4. Route: name the mc-* skill for the current stage (see the stage table in `{skill-root}/PIPELINE.md`) and invoke it if the creator asked to proceed. Route by the project's own `stages` array, never the master list: footage-first projects carry the ingest-first list (`new`, `cut`, `beats`, `graphics`, `assets`, `package`, `final`, `retro`) and never visit the ideation stages. One standing extra hop: when the graphics stage completes (mc-graphics hands back with `graphics/` holding rendered overlays), route through mc-cut's composited preview re-render (its `Routed re-entries` section) before invoking the next stage skill, and again whenever an overlay is later re-rendered. 5. If the stage owner is the creator: - record: say exactly what they need to do (record takes at constant frame rate into `raw/`) and stop. - final: branch on `[render]` `self-render` in the studio config. When true (the default), route to mc-cut's offered renders: the fast low-res preview for iteration and the final-quality render offer at gate 4. When false, the finish is editor-only: point the creator at the always-exported timeline and assets (edl.json, cutplan, overlays, the timeline file per `[editor]` timeline-format) and stop. Either way gate 4 closes only on the creator's recorded approval of a deliverable. diff --git a/skills/mc-pipeline/customize.toml b/skills/mc-pipeline/customize.toml deleted file mode 100644 index 90d9c5a..0000000 --- a/skills/mc-pipeline/customize.toml +++ /dev/null @@ -1,20 +0,0 @@ -# DO NOT EDIT -- overwritten on every update. -# -# Customization surface for mc-pipeline. -# Override files (not edited here): -# {project-root}/_bmad/custom/mc-pipeline.toml (team) -# {project-root}/_bmad/custom/mc-pipeline.user.toml (personal) -# -# Studio-wide config (owner, paths, editor, transcription, assets, tools) -# lives in [modules.manticore] in {project-root}/_bmad/custom/config.toml, -# maintained by mc-setup. This file holds only this skill's own surface. - -[workflow] - -# Steps to run before / after the standard activation. -activation_steps_prepend = [] -activation_steps_append = [] - -# Persistent facts held for the whole run: literal sentences, or -# "file:{project-root}/..." paths whose contents are loaded as facts. -persistent_facts = [] diff --git a/skills/mc-retro/SKILL.md b/skills/mc-retro/SKILL.md index b1826a9..e4d4eba 100644 --- a/skills/mc-retro/SKILL.md +++ b/skills/mc-retro/SKILL.md @@ -1,57 +1,73 @@ --- name: mc-retro -description: Post-publish feedback ratchet and wrap-up. Takes the creator's notes on a finished video and edits the format profile, blacklist, voice bible, production bible, and offending skill files so the pipeline compounds, then offers the post-publish wrap lane (archive hygiene, asset promotion). Runs with or without project.json. Use after a video ships, whenever the creator gives pipeline feedback, or to wrap up a published project. +description: Turn post-publish notes into pipeline improvements. Use after a video ships, or when the user says "retro", "here is my feedback", or "wrap up this project". --- # mc-retro -The compounding mechanism: feedback edits FILES, not just memory. Every note improves the taste files the next run obeys. Two lanes: retro (route notes into taste files) and wrap (post-publish cleanup and asset promotion). Retro runs first and offers wrap after; the creator can also request wrap on its own for an already-retroed project. +The compounding mechanism: feedback edits FILES, not just memory. Every note lands in a taste file that the next run obeys, which is why a note left in the conversation is lost work. The consumer of your edits is a later stage reading those files cold, with none of this session in the room. -## Steps +Two lanes. Retro routes the creator's notes into the files that would have prevented them. Wrap does post-publish cleanup and asset promotion. Retro runs first and offers wrap after; the creator can also ask for wrap alone on an already-retroed project. -1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there) and this skill's own surface (`uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}`; run `{workflow.activation_steps_prepend}` now, `{workflow.activation_steps_append}` after this step, and hold `{workflow.persistent_facts}` as standing context). Resolve `paths` values against `{project-root}`. Then establish the target: - - With a project: read `project.json` (stage `retro`) and take the format from it. - - Without `project.json` (ad-hoc run): ask the creator which format profile in `{formats-path}` the notes concern, or which brand file directly (`voice-bible.md`, `blacklist.md`, `production-bible.md`), and route notes to that target. Skip the `project.json` bookkeeping in step 7; everything else applies, so ad-hoc feedback still compounds. +## Resolution rules - Read the taste files that may receive edits: `{brand-path}/voice-bible.md`, `{brand-path}/blacklist.md`, `{brand-path}/production-bible.md`, and the target format profile. Collect the creator's notes: what felt wrong, what they re-edited in their editor, what they rewrote in the script, what looked off on screen, packaging performance (Test & Compare results if run). -2. Route every note to the file that would have prevented it: - - voice/wording miss: `{brand-path}/voice-bible.md` (new rule with the verbatim example) and/or a new pattern in `{brand-path}/blacklist.md`, - - visual style miss (graphics density, overlay aesthetic, image-type choice, CTA placement): `{brand-path}/production-bible.md`, in the global section or the matching per-format override section; ISO-dated, one-way ratchet (entries only accumulate; a change of taste gets a new dated entry that supersedes by date, never a deletion), - - structural/retention miss: the Learnings section of `{formats-path}/.md` (ISO-dated, newest first), - - a stage doing the wrong thing: route the note FIRST to that skill's durable per-skill surface, a `workflow.persistent_facts` entry (or the matching `workflow` key, e.g. one of the `*_flags`) in `{project-root}/_bmad/custom/.toml`, the team-override layer resolve_customization.py loads on every activation and module updates never touch (the same file where mc-setup records mc-cut's `cutplan_flags`); edit it surgically, preserving existing keys. Editing that skill's installed SKILL.md is the last resort, only when the note fits nowhere else: skill edits apply to the installed module and may be overwritten by module updates. If the harness blocks access to another skill's folder, record the note in the format profile's Learnings instead, - - a tool being driven wrong: the `notes` field of that tool's `[[tools]]` entry in the studio config (`[modules.manticore]` in `{project-root}/_bmad/custom/config.toml`), - - a mechanical failure: an issue note in the relevant engine README or script docstring, - - a pipeline gap (a stage could not do its job because the module itself is missing a feature, has a wrong contract, or a broken mechanic): append an entry to the studio improvements log (see Improvements log below), in addition to any local fix above. -3. Make the edits. Small and surgical: one note, one edit, at the point of failure. Show the creator the diff summary. -4. If the cut stage's judgment was overridden repeatedly in the editor, mine the pattern (e.g. "always keep pre-demo breaths") into the format profile's Learnings. -5. Offer the wrap lane (below). Run it if the creator accepts, or if they invoked mc-retro for wrap directly. -6. In `project.json` (skip on ad-hoc runs), append `retro` to `stages_done` and set `stage` to `done`, append retro notes (and whether wrap ran) to its `notes` field, and report. +- Bare paths resolve against `{video-path}`, the current video project at `{projects-path}//`. +- `{skill-root}` → this skill's installed directory; files in it always carry it. +- `{project-root}` → the project working directory. -## Wrap lane +## On Activation + +1. Studio config: `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`. Empty means mc-setup has not run: stop and route the creator there. Resolve `paths` values against `{project-root}`. +2. Establish the target. With a project, read `project.json` (stage `retro`) and take the format from it. On an ad-hoc run with no `project.json`, ask which format profile in `{formats-path}` the notes concern, or which brand file directly (`{brand-path}/voice-bible.md`, `{brand-path}/blacklist.md`, `{brand-path}/production-bible.md`), and route to that. Ad-hoc runs skip only the `project.json` bookkeeping, so ad-hoc feedback still compounds. +3. Read `{brand-path}/voice-bible.md`. If it does not exist, tell the creator it is missing and that routing a voice or wording note cannot happen without it, then route to mc-setup and stop. +4. Read `{brand-path}/blacklist.md`. If it does not exist, tell the creator it is missing and that routing a new blocked pattern cannot happen without it, then route to mc-setup and stop. +5. Read `{brand-path}/production-bible.md`. If it does not exist, tell the creator it is missing and that routing a visual-style note cannot happen without it, then route to mc-setup and stop. + +## Collecting the notes + +Read the target format profile too before asking for anything, so every file that may receive an edit is loaded. Then take one round of notes: what felt wrong, what they re-edited in their editor, what they rewrote in the script, what looked off on screen, packaging performance (Test & Compare results if run). One round per session; do not fish for endless feedback. + +## Routing + +Every note goes to the file that would have prevented it. + +- Voice or wording miss: a rule in `{brand-path}/voice-bible.md` carrying the verbatim example, and/or a new pattern in `{brand-path}/blacklist.md`. +- Visual style miss (overlay aesthetic, image-type choice, CTA placement): `{brand-path}/production-bible.md`, global section or the matching per-format override section. +- One beat type dominating, too many plain text cards, too few beats for the runtime: the same file's visual density and variety section, where the beats-per-minute floor, the variety quota, and the static-card cap are numbers. Change the number, globally or in the per-format override; nothing else in the module holds those three, so this edit is what actually moves the next plan. +- Every video running too dense or too sparse: that is the graphics-frequency tier, and the bible only mirrors it. Change `graphics-frequency` in `[style]` of the studio config for the whole studio, or the tier in a format profile's frontmatter for one format, then update the mirror in the bible's same section. +- Motion feeling wrong (how things enter, move, and exit): the same file's animation and motion look-and-feel section, which outranks the recipes any skill ships. A motion pattern that worked and should recur is the same edit, written as the convention rather than the complaint. +- Structural or retention miss: the Learnings section of `{formats-path}/.md`. +- A stage doing the wrong thing: split the note. A mechanical value (a flag, a threshold) goes to that stage's sub-table of the studio config (`[cut]`, `[packaging]`, `[retro]` under `[modules.manticore]` in `{project-root}/_bmad/custom/config.toml`, the same file where mc-setup records mc-cut's `cutplan-flags`); edit it surgically, preserving existing keys, and module updates never touch it. Taste routes as above: the production bible when it is studio-wide, the format profile's Learnings when it is per-format. Editing that skill's installed SKILL.md is the last resort, only when the note fits neither, because module updates may overwrite it. If the harness blocks access to another skill's folder, record the note in the format profile's Learnings instead. +- A tool being driven wrong: the `notes` field of that tool's `[[tools]]` entry in the studio config (`[modules.manticore]` in `{project-root}/_bmad/custom/config.toml`). +- A mechanical failure: an issue note in the relevant engine README or script docstring. +- A pipeline gap, meaning a stage could not do its job because the module itself is missing a feature, has a wrong contract, or has a broken mechanic: an entry in the improvements log below, in addition to any local fix above. -Post-publish cleanup for a shipped project. Requires a project folder; run in order, no skipping ahead. +Entries are ISO-dated, newest first, and one-way: the blacklist, Learnings, the production bible, and the improvements log only accumulate, so a change of taste is a new dated entry that supersedes by date rather than a deletion. Deletions are the creator's call. -1. Confirm the published master is safe. The creator confirms the final master exists at its published or archived location before anything is reclaimed. This confirmation is a hard stop: nothing below runs on assumption. -2. Reclaim reproducible render scratch: preview renders, intermediate proxies, render caches, anything regenerable from `edl.json` plus the sources. List every candidate with its size, subtract anything matching `{wrap.preserve}`, and delete only after the creator approves the list. Never candidates: source footage, transcripts, `edl.json`, the cutplan, overlays, `project.json`, and the master itself. -3. Enforce one blessed asset per slot: for each asset slot (thumbnail, title card, any overlay with multiple candidates), keep only the shipped version in the slot; rejected candidates move to the reclaim list or an archive folder, the creator's choice. -4. Promote evergreen assets: anything useful beyond this video (reusable overlays, diagrams, series templates) moves to the series `common/` folder beside its project folders under `{projects-path}`; anything brand-wide moves to `{brand-path}`. Write or update a `README.md` in each destination listing each promoted asset, the project it came from, and the ISO date. -5. Keep transcripts. Transcripts are never reclaimed; they are the cheapest permanent record of what was said. -6. Write any durable rules discovered during wrap into the format profile's Learnings and `{brand-path}/production-bible.md`, same ratchet as step 2. +Make the edits small and surgical, one note one edit at the point of failure, and show the creator the diff summary. A repeated override of the cut stage's judgment in the editor is itself one note: mine the pattern ("always keep pre-demo breaths") into the format profile's Learnings. + +Never weaken a gate or remove a hard rule in response to convenience feedback; flag those for the creator explicitly. ## Improvements log -The structured upstream feedback channel. mc-retro, and any stage that hits a pipeline gap mid-run, appends to a studio-level log so module maintainers get comparable reports from every studio. +The structured upstream feedback channel, so module maintainers get comparable reports from every studio. mc-retro, and any stage that hits a pipeline gap mid-run, appends to `improvements-log.md` in the studio root, the parent folder of `{brand-path}` (with default paths that is `manticore/improvements-log.md`). Create it with a `# Improvements log` heading if missing. One line per entry, append-only, newest first under the heading: + +`- YYYY-MM-DD [stage/skill]: what happened; why it is a gap; suggested fix; severity: low|medium|high` + +Entries describe module gaps, not creator taste. Taste goes to the brand and format files per the routing above. + +## Wrap lane + +Post-publish cleanup for a shipped project; requires a project folder. -- Location: `improvements-log.md` in the studio root, the parent folder of `{brand-path}` (with default paths that is `manticore/improvements-log.md`). Create it with a `# Improvements log` heading if missing. -- Entry format, one line per entry, append-only, newest first under the heading: +Nothing is reclaimed until the creator confirms the final master exists at its published or archived location. That confirmation is a hard stop: no part of this lane runs on assumption. - `- YYYY-MM-DD [stage/skill]: what happened; why it is a gap; suggested fix; severity: low|medium|high` +- Reclaim reproducible render scratch: preview renders, intermediate proxies, render caches, anything regenerable from `edl.json` plus the sources. List every candidate with its size, subtract anything matching `[retro] preserve` in the studio config, and delete only after the creator approves the list. Never candidates: source footage, transcripts, `edl.json`, the cutplan, overlays, `project.json`, and the master itself. +- Enforce one blessed asset per slot: for each asset slot (thumbnail, title card, any overlay with multiple candidates), keep only the shipped version; rejected candidates move to the reclaim list or an archive folder, the creator's choice. +- Promote evergreen assets: anything useful beyond this video (reusable overlays, diagrams, series templates) moves to the series `common/` folder beside its project folders under `{projects-path}`; anything brand-wide moves to `{brand-path}`. Write or update a `README.md` in each destination listing each promoted asset, the project it came from, and the ISO date. -- Entries describe module gaps, not creator taste. Taste goes to the brand and format files per step 2. +Durable rules discovered during wrap route like any other note. -## Rules +## Closing out -- One round of notes per session; do not fish for endless feedback. -- Never weaken a gate or remove a hard rule in response to convenience feedback; flag those for the creator explicitly. -- Blacklist, Learnings, the production bible, and the improvements log only grow; deletions are the creator's call. -- Wrap deletes nothing the creator has not seen listed, and never runs past step 1 without the published-master confirmation. +In `project.json` (skip on ad-hoc runs), append `retro` to `stages_done` and set `stage` to `done`, append the retro notes and whether wrap ran to its `notes` field, and report. diff --git a/skills/mc-retro/customize.toml b/skills/mc-retro/customize.toml deleted file mode 100644 index d02c63c..0000000 --- a/skills/mc-retro/customize.toml +++ /dev/null @@ -1,28 +0,0 @@ -# DO NOT EDIT -- overwritten on every update. -# -# Customization surface for mc-retro. -# Override files (not edited here): -# {project-root}/_bmad/custom/mc-retro.toml (team) -# {project-root}/_bmad/custom/mc-retro.user.toml (personal) -# -# Studio-wide config (owner, paths, editor, transcription, assets, tools) -# lives in [modules.manticore] in {project-root}/_bmad/custom/config.toml, -# maintained by mc-setup. This file holds only this skill's own surface. - -[workflow] - -# Steps to run before / after the standard activation. -activation_steps_prepend = [] -activation_steps_append = [] - -# Persistent facts held for the whole run: literal sentences, or -# "file:{project-root}/..." paths whose contents are loaded as facts. -persistent_facts = [] - -[wrap] - -# Path globs, relative to the project folder, that the wrap lane must -# never list for reclaim (in addition to the built-in never-reclaim set: -# source footage, transcripts, edl.json, cutplan, overlays, project.json, -# and the published master). -preserve = [] diff --git a/skills/mc-script/SKILL.md b/skills/mc-script/SKILL.md index 054da55..dbf915f 100644 --- a/skills/mc-script/SKILL.md +++ b/skills/mc-script/SKILL.md @@ -1,28 +1,50 @@ --- name: mc-script -description: Weave the full script from the approved outline using the creator's braindump words under the quote-or-cut contract, lint it, and run the craft QA. Use at the script stage, only after gate 1 (outline) is approved. +description: Weave the script from outline and braindump. Use at the script stage after gate 1, or when the user says "write the script" or "draft the script". --- # mc-script -The anti-LLM-slop stage. The script is woven, not written. +The anti-LLM-slop stage. The script is woven, not written: everything worth saying is already in `braindump.md`, and your craft is the weave. The outcome is `script.md`, performed to camera by the creator who dumped it. Read every line aloud in your head at their pace; anything you stumble on, they will too. -## Steps +## Resolution rules -1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there) and this skill's own surface (`uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}`; run `{workflow.activation_steps_prepend}` now, `{workflow.activation_steps_append}` after this step, and hold `{workflow.persistent_facts}` as standing context). Resolve `paths` values against `{project-root}`. Read `project.json` (confirm `approvals.outline` is a date, stage is `script`), `outline.md`, `braindump.md`, `{brand-path}/voice-bible.md`, 1 to 2 files from `{brand-path}/exemplars/`, and the format profile. -2. Weave beat by beat under the quote-or-cut contract: - - Default move: lift the creator's braindump phrasing directly, smoothing only for spoken flow. - - Every sentence either traces to a braindump passage or carries an inline `[INVENTED]` flag. Flags are for connective tissue only; if a content claim needs inventing, the braindump has a gap: ask the creator instead. - - Write the hook LAST, from the approved candidate, once the body proves what the hook can promise. - - Weave the configured CTA line(s) from `[cta]` in the studio config, following the outline's CTA plan line and craft rule 15: raise stakes before the ask, lower the barrier right before it, end on ease. CTA copy is the one sanctioned exception to quote-or-cut: it comes from the configured items (kind, label, url), not the braindump, and needs no `[INVENTED]` flag. If `[cta]` is empty, weave no ask rather than invent one. - - If project.json `sources` has an `interview` recording, transcribe it if not yet done (mc-cut's transcribe.py) and mark every script line whose phrasing was already spoken well on camera with an inline `[TAKE s-s]` from the word timestamps. Those lines may not need re-recording, only reorganizing at the cut stage. -3. Lint: `uv run {skill-root}/scripts/lint_script.py {projects-path}//script.md --blacklist {brand-path}/blacklist.md`. Fix every violation before presenting. -4. Craft QA: run the checklist at `{workflow.craft_checklist}` (relative paths resolve against `{skill-root}`; default is the packaged 16-rule list), plus the manual QA list at the bottom of the creator's blacklist. Fix, do not annotate around, failures. -5. Compute runtime from the real word count at the creator's measured wpm (`[owner] wpm` in the config) and state it. Flag if it misses the format's target length. -6. Write `script.md` (with the `[INVENTED]` flags still visible), update project.json (append `script` to `stages_done`, set `stage` to the next entry in its `stages` array), and present. Tell the creator the ball is theirs: record, drop takes in `raw/` at constant frame rate. If `[TAKE ...]` markers exist, list the delta explicitly: which lines are already captured on the interview recording and which still need recording. +- Bare paths resolve against `{video-path}`, the current video project at `{projects-path}//`. +- `{skill-root}` → this skill's installed directory; files in it always carry it (`{skill-root}/scripts/lint_script.py`). +- `{project-root}` is the project working directory; the studio config's `paths` values resolve against it. -## Rules +## On Activation +1. Resolve the studio config: `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`. Empty means mc-setup has not run: stop and route the creator there. +2. Read `project.json` (confirm gate 1 passed: `approvals.outline` is a date, stage is `script`), `outline.md`, `braindump.md`, and the format profile. +3. Read `{brand-path}/voice-bible.md`. If it does not exist, tell the creator it is missing and that writing in their voice cannot happen without it, then route to mc-setup and stop. +4. Read `{brand-path}/exemplars/`. If it does not exist, tell the creator it is missing and that weaving against their own proven scripts cannot happen without it, then route to mc-setup and stop. Take 1 to 2 files from it. +5. Read `{brand-path}/blacklist.md`. If it does not exist, tell the creator it is missing and that the blacklist lint before handoff cannot happen without it, then route to mc-setup and stop. +6. Read `{brand-path}/craft-checklist.md`. If it does not exist, tell the creator it is missing and that the craft pass before presenting cannot happen without it, then route to mc-setup and stop. + +## The weave + +Work beat by beat under the quote-or-cut contract: + +- Lift the creator's braindump phrasing directly, smoothing only for spoken flow. +- Every sentence either traces to a braindump passage or carries an inline `[INVENTED]` flag. Flags are for connective tissue only; a content claim that needs inventing is a gap in the braindump, so ask the creator. If more than roughly a quarter of the sentences would need a flag, stop and send the project back to mc-braindump: the raw material is not there yet. +- Write the hook LAST, from the approved candidate, once the body proves what the hook can promise. - No stage directions, no camera notes, no "(pause)" theater unless the creator asked for them. -- Read the script aloud in your head at their pace; anything you stumble on, they will too. -- If more than roughly a quarter of sentences would need `[INVENTED]`, stop and send it back to mc-braindump: the raw material is not there yet. + +## The CTA + +Weave the configured CTA line(s) from `[cta]` in the studio config, following the outline's CTA plan line and the craft checklist's lower-the-barrier rule. CTA copy is the one sanctioned exception to quote-or-cut: it comes from the configured items (kind, label, url), not the braindump, and needs no `[INVENTED]` flag. If `[cta]` is empty, weave no ask rather than invent one. + +## Lines already on camera + +If project.json `sources` has an `interview` recording, transcribe it if not yet done (mc-cut's transcribe.py) and mark every script line whose phrasing was already spoken well on camera with an inline `[TAKE s-s]` from the word timestamps. Those lines may not need re-recording, only reorganizing at the cut stage. + +## QA before presenting + +- `uv run {skill-root}/scripts/lint_script.py {projects-path}//script.md --blacklist {brand-path}/blacklist.md`. Exit 1 lists violations; fix every one. +- Run the creator's craft checklist at `{brand-path}/craft-checklist.md`, plus the manual QA list at the bottom of their blacklist. +- Runtime from the real word count at the creator's measured wpm (`[owner] wpm`), stated and flagged if it misses the format's target length. + +## Handoff + +Write `script.md` with the `[INVENTED]` flags still visible, update project.json (append `script` to `stages_done`, set `stage` to the next entry in its `stages` array), and present. The ball is then the creator's: record, drop takes in `raw/` at constant frame rate. Where `[TAKE ...]` markers exist, list the delta explicitly: which lines are already captured on the interview recording, which still need recording. diff --git a/skills/mc-script/customize.toml b/skills/mc-script/customize.toml deleted file mode 100644 index ad7ebb4..0000000 --- a/skills/mc-script/customize.toml +++ /dev/null @@ -1,26 +0,0 @@ -# DO NOT EDIT -- overwritten on every update. -# -# Customization surface for mc-script. -# Override files (not edited here): -# {project-root}/_bmad/custom/mc-script.toml (team) -# {project-root}/_bmad/custom/mc-script.user.toml (personal) -# -# Studio-wide config (owner, paths, editor, transcription, assets, tools) -# lives in [modules.manticore] in {project-root}/_bmad/custom/config.toml, -# maintained by mc-setup. This file holds only this skill's own surface. - -[workflow] - -# Steps to run before / after the standard activation. -activation_steps_prepend = [] -activation_steps_append = [] - -# Persistent facts held for the whole run: literal sentences, or -# "file:{project-root}/..." paths whose contents are loaded as facts. -persistent_facts = [] - -# Craft QA checklist used by step 4. Relative paths resolve against the -# skill folder. Override with your own file (absolute or {project-root}/... -# path) if you have developed your own retention rules; the packaged -# default is the 16-rule list in assets/craft-checklist.md. -craft_checklist = "assets/craft-checklist.md" diff --git a/skills/mc-setup/SKILL.md b/skills/mc-setup/SKILL.md index 9c2e5a6..d1f1c5f 100644 --- a/skills/mc-setup/SKILL.md +++ b/skills/mc-setup/SKILL.md @@ -1,175 +1,120 @@ --- name: mc-setup -description: First-run (or any-time) configuration for the Manticore video pipeline. Creates and updates the studio config ([modules.manticore] in {project-root}/_bmad/custom/config.toml), verifies dependencies and platform support, installs the HyperFrames graphics skills, runs the onboarding interview (basics, render consent, video style, CTAs, audio lanes), builds the brand for real (tokens, Production Bible, headshots, guided voice bible), registers the creator's generation CLIs with end-to-end verification, scaffolds .env.example, and migrates 0.x studios. Run when any mc-* skill reports missing config, or to change tools later. +description: Configure the studio, brand, and generation tools. Use on a missing-config report, or when the user says "set up manticore", "change my tools", or "update my studio". --- # mc-setup -Idempotent: on re-run, existing values are shown as defaults and only what the creator wants changed is changed. Never clobber, never silently overwrite. +Act as the studio's configurator. The outcome is a studio config every mc-* skill can resolve, a brand folder with real content in it rather than placeholders, and an honest report of what is still missing. -The studio config is one table, `[modules.manticore]`, in `{project-root}/_bmad/custom/config.toml` (personal overrides in `config.user.toml` next to it). Every mc-* skill resolves it with the installed `{project-root}/_bmad/scripts/resolve_config.py`. The full default values ship in this skill's `customize.toml` under `[defaults]`. +Two consumers set the bar. Every other mc-* skill reads `[modules.manticore]` and fails closed without it, so a half-written config is worse than none. The creator needs to know what will actually happen on their first project before they start one, which is why this stage ends in a runnability report rather than a success message. -## Steps +Idempotent throughout: on re-run, existing values are the defaults offered, only what the creator wants changed is changed, and a re-run with no changes writes nothing. Never clobber, never silently overwrite. -### 0. Bootstrap BMad core +## Resolution rules -Check four paths: `{project-root}/_bmad/config.toml`, `{project-root}/_bmad/scripts/resolve_config.py`, `{project-root}/_bmad/scripts/resolve_customization.py`, `{project-root}/_bmad/custom/`. All present: go to step 1. Any missing: the project is not BMad-initialized; say so and confirm before running the installer (it writes `{project-root}/_bmad/` plus IDE integration files for the chosen tool). Resolve the tool id first: `claude-code` under Claude Code; otherwise run `npx -y bmad-method install --list-tools` and let the creator pick. +- Bare paths resolve against `{video-path}`, the current video project at `{projects-path}//`. +- `{skill-root}` → this skill's installed directory; files in it always carry it (`{skill-root}/references/bootstrap.md`). +- `{project-root}` → the project working directory. +- `{brand-path}`, `{formats-path}`, `{projects-path}`, `{engines-path}` → the `[paths]` values from the studio config, resolved against `{project-root}`. -- No `{project-root}/_bmad/` at all: `npx -y bmad-method@latest install --directory {project-root} --modules core --tools -y`. Never omit `--modules core` (a bare `-y` installs the default module set, not just core) or `--tools` (fresh `-y` installs fail without it). -- `{project-root}/_bmad/` exists but incomplete: `npx -y bmad-method@latest install --directory {project-root} -y` (quick-update re-syncs `{project-root}/_bmad/scripts/` and keeps configured tools). If that fails, retry with `--action update --modules core --tools ` added. +## On Activation -Verify: both resolver scripts exist under `{project-root}/_bmad/scripts/`, `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore` exits 0 (empty output just means the interview has not run), and `uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}` returns this skill's `[defaults]`. If uv itself is missing, do step 2's uv bootstrap before verifying. On verification failure, stop and surface the installer output; never hand-copy scripts or vendor a resolver. If npx, node, or the network is unavailable, or the creator declines, have them run `npx bmad-method install` interactively from the project root (core alone is enough for Manticore), then re-run mc-setup. A stale npx cache can serve an old CLI missing current flags; on unknown-option errors, run `npm cache clean --force` and retry. +1. Check three paths: `{project-root}/_bmad/config.toml`, `{project-root}/_bmad/scripts/resolve_config.py`, `{project-root}/_bmad/custom/`. Any missing means the project is not BMad-initialized; load `{skill-root}/references/bootstrap.md` and finish it before continuing. +2. Resolve the current state: `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`. +3. Read `{skill-root}/assets/studio-defaults.toml`. Its `[defaults]` tables are the seed values for the whole interview and the authority on every default. -### 1. Locate and load +Then route on the resolved config: -Resolve the current state: `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`. Load the defaults: `uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}` (run `{workflow.activation_steps_prepend}` now, `{workflow.activation_steps_append}` after this step, and hold `{workflow.persistent_facts}` as standing context). If `[modules.manticore]` already has values, tell the creator this is an update pass; offer the sections below as a menu instead of walking all of them. But first check for a 0.x studio. +| State | What to do | +|---|---| +| Empty | First run. Work through everything below. | +| Present, missing any of `[render]`, `[style]`, `[cta]`, `[live]`, `[audio]` | A 0.x studio. Load `{skill-root}/references/migration-0x.md`. | +| Present and complete | An update pass. Say so and offer the sections below as a menu; do not walk them all. Copy in any shipped template `{brand-path}` or `{formats-path}` is missing first, whatever the creator picks: a studio configured before a template shipped is a studio whose stage skills stop on it. | -#### 1a. 0.x migration +## Dependencies and platform -An existing `[modules.manticore]` that is missing any of the 1.0 tables (`[render]`, `[style]`, `[cta]`, `[live]`, `[audio]`) is a 0.x studio. Say so, then migrate instead of re-interviewing everything: +Bootstrap uv first: check `uv --version`, and if it is missing offer the official installer from docs.astral.sh/uv and wait for confirmation. Every pipeline script runs through uv, so nothing works without it. -- Backfill each missing table from this skill's `[defaults]` (render, style, cta, live, audio), writing them into the existing config surgically. -- If `[transcription] api-key-env` names a key the configured local provider never uses, blank it (the 1.0 default; metered keys are set only when a metered provider is chosen). -- If the studio recorded interview footage against the pre-1.0 marker cue ("question from claude"), offer the step 3 marker-cue question and record the `--marker-cues` override in `{project-root}/_bmad/custom/mc-cut.toml` so cutplan keeps segmenting that footage. -- If the `[assets]` lanes still carry pre-1.0 defaults pointing at a metered API the creator never opted into or verified, flag that in the summary and offer step 5 to repoint them at a registered CLI tool (or leave them empty so mc-assets asks at farming time). -- Refresh the creator's format profiles surgically: for every profile in `{formats-path}` that also ships in `{skill-root}/assets/formats/`, run `uv run {skill-root}/scripts/merge_profile_frontmatter.py --shipped {skill-root}/assets/formats/.md --studio {formats-path}/.md`. It adds only the frontmatter keys new in 1.0 (`beat-types`, `density`, and any future ones) that stages like mc-beats require, never overwriting an existing key, the creator's prose, or the Learnings. Then copy any newly shipped profiles that do not exist in `{formats-path}` (the step 4 rule). -- A pre-1.0 series or thumbnail template at the brand root (for example `thumbnail-template.md`) predates the `{brand-path}/templates/.md` contract: offer to move it there, named for the series it describes, so mc-package finds it. -- Run the delta interview: step 3b (render consent), then step 3c (the video style interview), then step 3d (audio lanes). -- Scaffold `{brand-path}/production-bible.md` per step 4, seeded from the brand assets that already exist (tokens.json, shipped overlays, exemplars, format-profile learnings) plus the step 3c answers, not from a blank slate. -- Offer, without forcing, the other new builds: the HyperFrames graphics skills (step 2b), headshot collection (step 4), the guided voice bible (step 4b), `.env.example` (step 7). -- Leave every other existing value untouched; those are already the creator's answers. +``` +uv run {skill-root}/scripts/check_deps.py +``` -Finish with step 8 as usual so the migrated config is verified and the pending gaps are reported. +Report what is missing with the exact install command for the platform, and install nothing without the creator confirming each item. The report ends with a platform verdict naming a stack file (`{skill-root}/references/stack-macos.md`, `{skill-root}/references/stack-windows.md`, or `{skill-root}/references/stack-linux.md`). Read the named file now and hold it for the rest of setup: it carries the transcription lane, torch index, encoder ladder, SVG rasterizer, and fonts approach this machine actually needs. -### 2. Dependencies +If this machine is not Apple Silicon, say so plainly rather than letting the creator discover it at cut time: the parakeet-mlx reference lane will not run here, and the recommended lane is onnx-asr on the same weights, with fillers and word timestamps carrying over. Carry that honesty into the transcription question. -Bootstrap first: check `uv --version`. If uv is missing, offer to install it; otherwise the official installer from docs.astral.sh/uv, and wait for the creator's confirmation; every pipeline script runs through uv, so nothing works without it. +## Install the HyperFrames graphics skills -Then run `uv run {skill-root}/scripts/check_deps.py`. Report what is missing with the exact install command (brew/apt/winget as fits the platform). Install nothing without the creator confirming each item. The report ends with a platform verdict: detected OS, CPU architecture, and GPU vendor, plus the recommended stack file (`{skill-root}/references/stack-macos.md`, `stack-windows.md`, or `stack-linux.md`) and the platform-specific defaults it implies (transcription lane, torch index, encoder ladder, SVG rasterizer, fonts approach). Read the named stack file now and hold it as context for the rest of setup. If this machine is not Apple Silicon, relay the script's pointer honestly: the parakeet-mlx reference lane will not run here, and the recommended lane is onnx-asr with the same parakeet-tdt-0.6b-v3 weights (implemented; verbatim fillers and word timestamps carry over, CPU or CUDA per the verdict). Carry that honesty into step 3's transcription question. +HyperFrames is the graphics engine, and its Agent Skills carry the agent's current, self-refreshing knowledge of what it can do. Installing them at setup rather than at first graphics run is deliberate: the capability surface is then known from the beats stage onward, where it changes what gets planned. -### 2b. HyperFrames graphics skills +``` +npx skills add heygen-com/hyperframes --all --full-depth +``` -HyperFrames is the graphics engine, and its Agent Skills carry the agent's current, self-refreshing knowledge of everything it can do (the block catalog, WebGL transitions, color grading, background removal, HTML-in-Canvas, the authoring patterns). Install them now, provided node and npx are present (step 2), so that capability knowledge is live from the beats and graphics stages onward instead of only after the first graphics run. Confirm before installing (it writes skill files into the project); the install is idempotent, a re-run refreshes rather than duplicates: +Idempotent, so a re-run refreshes rather than duplicates. `npx hyperframes skills update` takes only the maintained core set for a lighter footprint. This runs on the local CLI with no account and no credits, and it is not the engine workspace, which is a multi-GB npm install that still builds lazily at first graphics run. If node or npx is missing, say HyperFrames graphics need them and defer rather than blocking setup. -- Install and favor the full catalog: `npx skills add heygen-com/hyperframes --all --full-depth`. A creator who wants a lighter footprint can take just the maintained core set instead: `npx hyperframes skills update`. -- Everything here runs on the local CLI; no HeyGen account or credits. The engine WORKSPACE itself (`{engines-path}/hyperframes/`, a multi-GB npm install) still builds lazily on the first graphics run; this step installs only the lightweight skill knowledge so the whole capability surface is known from the start. -- If node or npx is missing, say HyperFrames graphics need them and defer the skills to the first graphics run rather than blocking setup. +## Interview the studio -### 3. The basics interview +Walk the `[defaults]` tables from `{skill-root}/assets/studio-defaults.toml`, offering current values as defaults. That file is the authority on the schema and every value in it; do not re-derive the question list here. What follows is only what reading the schema will not tell you. -Ask, offering current values (or the `[defaults]` from customize.toml) as defaults: +**Traps in the basics.** -- Name and channel/brand name (`[owner]`). -- Description links, in order (`[owner] links`). -- Speaking rate: if they have a published transcript, offer to measure it (word count / duration); otherwise leave 145 with a note that mc-script will flag it as unmeasured. Step 4b measures it for real from their own transcript if the voice bible gets built. -- Paths: accept the defaults (`manticore/brand`, `manticore/formats`, `manticore/projects`, `manticore/engines`) unless they have a place they want things, e.g. an existing brand folder via `brand-path`. -- Video defaults (`[video]`): confirm record resolution, delivery resolution, and fps. Offer to ffprobe a recent recording and fill the values from reality instead of guessing. -- Live tool (`[live] tool`): obs, ecamm, or other. Drives the stream-pack lane's deliverable format. -- Recurring shows or series they produce (names, cadence). Note them for format-profile choices and future series folders. -- Editor (`[editor]`): which NLE they finish in. Set `timeline-format` accordingly: fcpxml for DaVinci Resolve or Final Cut Pro; xmeml/edl for Premiere; none for Descript or manual workflows (they get the cut plan, edl.json, and preview instead of a timeline file). Set `ograf-editable = true` ONLY for DaVinci Resolve 21+. -- Transcription (`[transcription]`): default `auto` (free, local, verbatim fillers preserved, no API key): parakeet-mlx on Apple Silicon, onnx-asr with the same parakeet-tdt-0.6b-v3 weights everywhere else. If this machine is not Apple Silicon, present the stack file's recommended lane (onnx-asr, cpu or gpu extra per the step 2 verdict) as what this machine will use, relaying implemented/planned status from customize.toml; whisper-based fallbacks are no longer recommended (they normalize fillers away). Metered API lanes exist behind the same switch as explicit opt-in choices; if, and only if, the creator picks one, set `provider` and `api-key-env` now and handle key sourcing in step 7. -- Interview marker cue: the spoken phrase that marks each question read aloud during interview-recording capture (mc-braindump's camera-rolling mode), so the cut stage can segment the recording mechanically. Default "question from the interviewer"; keep it unless the creator wants their own phrasing or has footage recorded against the older "question from claude" convention. Record a non-default cue as `cutplan_flags = '--marker-cues ""'` in `{project-root}/_bmad/custom/mc-cut.toml` (mc-cut's team override file, resolved by resolve_customization.py; edit it surgically, preserving any existing keys); mc-cut passes those flags straight to cutplan.py. The default needs no entry. +- Video defaults: offer to ffprobe a recent recording and fill them from reality instead of asking the creator to recall numbers. +- Speaking rate: leave the default and say mc-script will flag it as unmeasured. The voice bible measures it for real. +- A non-default interview marker cue is not a key of its own. Record it as `cutplan-flags = '--marker-cues ""'` in the `[cut]` sub-table, edited surgically. +- `[cut]`, `[packaging]` and `[retro]` are mechanical knobs rather than taste: write them from the defaults instead of interviewing them. The one worth raising is `[cut] silence-floor-db`, a property of the creator's room and mic; offer it when they mention a noisy room. -Be honest about lane status: the comments in this skill's `customize.toml` under `[defaults.editor]` and `[defaults.transcription]` mark each timeline and transcription lane as implemented or planned; relay that, and never promise a planned lane as working. +**Render consent is performed, not assumed.** Before writing `[render]`, present the render-first default and get an explicit answer: Manticore previews every cut and beats iteration and offers a final at gate 4, while the timeline export and all assets are ALWAYS produced alongside, so the creator can move into their own editor at any step. Declining makes the renders offers instead of automatic outputs and changes nothing else. Offer the quality knobs only if asked. -### 3b. Render consent +On a non-Mac machine, confirm the stack file's expectations here too: the encoder ladder the final render will probe, and on Windows with NVIDIA that the torch cu126 index adds roughly 2.5 to 3 GB. -Present and confirm the render-first default before writing `[render]`: Manticore renders a fast low-res preview for every cut and beats iteration, and offers a final-quality render at gate 4; the editor timeline export and all assets (edl.json, cutplan.md, overlays) are ALWAYS created alongside, so the creator can jump into Resolve, Premiere, or any editor at any step. +**Video style.** `{skill-root}/assets/production-bible-spec.md` is the build spec; follow it rather than inventing a question order. Two things it does not carry: ask about creators to emulate first and, with permission, study the links they give rather than asking anyone to describe a style in the abstract; and echo the distilled takeaways back in your own words for confirmation before they land, since they then seed every remaining question as a proposed default. Style answers go in BOTH places, the config keys for mechanical consumption and the bible for taste, and neither is a copy of the other. Section 5 is the exception: only its tier has a config key, so the floor, the variety quota, and the card cap live in the bible alone. -- Accepted (the default): `self-render = true`. Offer the quality knobs (preview-height, preview-crf, final-codec, final-crf, loudnorm, loudness-target) only if they ask; the defaults are sane. Mention that the final render is loudness-normalized to -14 LUFS by default (the YouTube reference; the preview never is), alongside the final-codec and final-crf defaults. On a non-Mac machine, also confirm the stack file's platform expectations here: the encoder ladder the final render will probe, and on Windows with NVIDIA the torch cu126 index consent note (CUDA wheels add roughly 2.5 to 3 GB) for step 3d's audio workspace. -- Declined: `self-render = false`. Previews and finals become offers the pipeline makes instead of automatic outputs; the timeline export and assets remain always-on. +**Audio lanes.** Confirm the local-first `[audio]` defaults; the ladder itself is mc-audio's knowledge, loaded when that skill runs. Be straight about three things the defaults do not admit: local TTS is stock voices with no cloning, so narration in the creator's own voice still means recording it; `song-provider` ships empty because no local lane is validated, so never promise the planned ACE-Step lane; and the engine workspace costs a multi-GB venv plus roughly 5 GB of model cache on first use. Offer to build it now with mc-audio's `ensure_workspace.py` or defer. An existing lab is reused, never rebuilt. -Record the answer explicitly; this consent is required, not assumed. +## Build the brand -### 3c. The video style interview +Create the four path folders if missing, then fill `{brand-path}` from `{skill-root}/assets/`: `tokens.json` from the template, `blacklist.md` from the starter, `craft-checklist.md` as shipped, and `production-bible.md` and `voice-bible.md` per their specs, which are the build instructions whether or not either gets built today. Copy into `{formats-path}` every profile from `{skill-root}/assets/formats/` that is not already there. Everything that lands in either folder is the creator's from that moment: copy only what is missing, never overwrite on a re-run, because their copies carry edits and accumulated learnings. -This step seeds the Production Bible; the build spec is `{skill-root}/assets/production-bible-spec.md`. Ask, and hold every answer for step 4: +Four things govern that work: -- Creators to emulate, first: ask whether there is a creator or channel whose video style they want to lean toward, and take video links. With permission, study what the links offer (titles, thumbnails, pacing, a transcript via yt-dlp) and distill what the creator is actually after: fast funny meme cuts, polished charts and dataviz, kinetic captions, calm long-form explainers, a particular edit rhythm. Echo the takeaways back in your own words ("here is what I take from these: ...") and confirm before writing anything; the confirmed takeaways go in the bible and seed every question below as proposed defaults. This is video style, distinct from step 4b's reference creators for spoken voice, though the same links can feed both. -- Visual density (`[style] graphics-frequency`): high (a graphic beat roughly every 10 to 20 seconds), medium (20 to 45), or low (45 to 90), on a front-loaded pacing curve. Default medium; nudge toward high for tutorial and explainer formats. Per-format overrides go in the bible, not the config. -- Preferred image types, with per-purpose splits: SVG or diagrammatic builds where text must be accurate, generative imagery for what does not exist, real verified imagery first for anything that does, or a stated mix. The sourcing hierarchy is real, then generative, then hand-built text card. -- Overlay and popup aesthetic: a described look, reference screenshots or creators to emulate, or overlays they have already shipped. Capture surface treatment (solid, glass, gradient, neon, flat, native-platform), blur, border, corner radius, shadow or glow, and placement taste. Store any supplied reference images beside the bible. -- Animation feel: snappy, smooth, or dramatic, plus entrance and exit conventions (for example fly-in and fly-out with optional whoosh). Mapped onto tokens.json motion values in step 4. -- CTA inventory and appetite (`[cta]` and `[[cta.items]]`): which CTAs they run (subscribe, community, support, product, next-video, playlist, site), each with label, URL, optional brand asset path, and priority; appetite aggressive, moderate, or minimal. Mirror the inventory into the bible's CTA section along with the native-platform styling rule (a subscribe element reads YouTube-red, a community element reads that platform's own colors). -- Asset libraries they already own (icon sets, b-roll folders, screenshot archives, photo libraries) and their locations, for the bible's image-type policy. +- Ask before interviewing: "point me at anything that already defines your brand or voice, a website, CSS, design tokens, style guides, past videos." Mine those, then interview only what mining could not answer. +- The exit state is filled, never placeholders. A placeholder survives only when the creator genuinely has nothing to give, and every survivor goes on the pending list loudly. +- Offer to build the voice bible now rather than leaving the spec sitting there. It needs the creator's own corpus, and it yields a measured wpm that replaces the estimate in `[owner] wpm`. +- `{brand-path}/headshots/` takes 3 to 6 approved photos across varied expressions (neutral, surprised, thinking, excited), renamed to expression slugs with an `index.md` catalog. Say the rule while collecting, because it governs what the creator hands over: approved photos only, never arbitrary frames from footage. A thumbnail sends the original photo to the image model, and every revision re-sends that same original rather than a prior generation. No headshots blocks thumbnails; flag it loudly. -Answers land in BOTH places: the config keys (`[style]`, `[cta]`) for mechanical consumption, the bible for taste. +## Register the creator's tools -### 3d. Audio lanes +CLI-first: a registered CLI backed by a subscription the creator already pays for is the preferred lane for every `[assets]` slot. -Present `[audio]` and confirm the local-first defaults (the full ladder and its honesty rules live in mc-audio's `references/audio-lanes.md`): TTS narration and two-host dialogue via kokoro-local, instrumental music beds via musicgen-local, SFX via audioldm2-local. All free and local; the mc-audio service skill farms them for graphics, stream packs, and voiceover narration. +Ask what they use for image generation, video generation, and offloaded research. Record each as a `[[modules.manticore.tools]]` block with name, capabilities, the exact headless invocation, and preferred models. Two fields earn their own attention: -- Be honest about what local TTS is: stock voices only, no cloning, so "narration in your own voice" still means recording it yourself; a paid cloning lane is opt-in and planned. -- Full songs with vocals: `song-provider` ships empty because no local lane is validated yet; say so if asked and never promise the planned ACE-Step lane. -- Disk and download honesty before any bootstrap: the engine workspace at `{engines-path}/audio-lab` needs a venv of several GB, ~340 MB of Kokoro models, and ~5 GB of Hugging Face cache on the first music/sfx run. Offer to build it now (`uv run` mc-audio's `ensure_workspace.py`) or defer; mc-audio asks again at first farming. An existing workspace (a lab the creator already built) is detected and reused, never rebuilt. -- Paid audio lanes (Gemini TTS, ElevenLabs) exist behind the same keys as explicit opt-in choices; if, and only if, the creator picks one, set the provider and `api-key-env` now and handle key sourcing in step 7. +- The `notes` field is the persistent memory: quirks, output behavior, what the tool is bad at. Write it now, because this is what stops every future session rediscovering the same tool. +- Verify end to end with permission: the version command first, then one small real invocation per registered capability, confirming the output file exists. Record the result in `notes` as verified end-to-end with the ISO date, or as unverified. -### 4. Brand build +Then set the `[assets]` lanes to registered, verified tool names. A lane with no good answer stays empty, so mc-assets stops and asks at farming time rather than billing anyone by default. -Create the four path folders if missing. The exit state is filled, never placeholders: a placeholder survives only when the creator genuinely has nothing to give, and every survivor goes on the step 8 pending list, loudly. +## Editor integration -Ask first: "point me at anything that already defines your brand or voice: a website, CSS, design tokens, style guides, writing skills, past videos." Mine those sources before interviewing; interview only what mining could not answer. +Native scripting is the default DaVinci Resolve path and no MCP server is required for any shipped lane: the cut stage exports an fcpxml timeline and Resolve-side automation drives Resolve's own API. Only if the creator ALREADY runs a Resolve MCP server and wants skills to use it, confirm with `claude mcp list` and record `davinci-resolve = true` under `[mcp]`. Otherwise record false and move on; do not suggest installing one. Other editors need nothing here. -Into `{brand-path}`: +## Keys and .env.example -- `tokens.json` from `{skill-root}/assets/tokens.template.json`, filled from the mined brand sources (site CSS, style guides, brand system docs) when they exist; otherwise walk the creator through canvas/accent/text colors and fonts. Map the step 3c animation feel onto the motion values. -- `production-bible.md`: scaffold from `{skill-root}/assets/production-bible-spec.md` and fill sections 1 through 6 from the step 3c answers and the mined sources. Any section with genuinely no answer stays a marked placeholder and is flagged in step 8. -- `blacklist.md` from `{skill-root}/assets/blacklist-starter.md`. -- `voice-bible.md`: built in step 4b. -- `headshots/`: collect 3 to 6 approved photos of the creator with varied expressions (neutral, surprised, thinking, excited). Auto-classify each expression, rename to expression-slug filenames, and write an `index.md` expression catalog (one line per photo: file, expression). Explain how they get used: when a thumbnail or asset needs the creator in it, the original photo goes to the configured image model with a "use the person in this image to ..." prompt, and any revision re-sends the same original photo, never a prior generation. State the rule inline: approved photos only; mc-package never uses arbitrary frames from footage. If no headshots exist yet, flag it loudly: thumbnails are blocked until they do. -- `exemplars/` folder (filled in step 4b). +For each metered lane the creator explicitly chose, confirm the env var name recorded in the config and check whether it is set. If nothing metered was chosen, skip key sourcing entirely. -Into `{formats-path}`: copy every profile from `{skill-root}/assets/formats/` that does not already exist there (never overwrite; the creator's copies accumulate learnings). +Then scaffold `{project-root}/.env.example` listing exactly the env vars the resolved config references: every non-empty `*-key-env` value a configured lane actually uses. That is possibly none, in which case write no file and say so. One line per var with a one-line source note, under a header comment saying real values never go in TOML, in chat, or in this file. Update an existing `.env.example` surgically and never touch a real `.env`. -### 4b. Guided voice-bible build +## Write and report -Offer to build `{brand-path}/voice-bible.md` now instead of leaving the spec (`{skill-root}/assets/voice-bible-spec.md`) as a placeholder. If accepted: +Write the results as `[modules.manticore]` and its sub-tables into `{project-root}/_bmad/custom/config.toml`, editing surgically: create the file if needed, preserve everything else in it because other modules configure themselves there too, and preserve any section the creator skipped. Mention `config.user.toml` for personal overrides in shared repos. Verify with `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore` and show the resolved summary. -- Ask for the creator's own corpus (YouTube URLs, published transcripts, writing) and, separately, any reference creators whose spoken style they want to lean toward. -- Fetch transcripts with yt-dlp (`--write-auto-subs`), with permission. Save cleaned exemplars into `{brand-path}/exemplars/`, keeping the creator's own voice separate from reference creators (subfolders `own/` and `reference/`; frontmatter with URL and capture date). -- Distill per the spec: every rule in the bible cites at least one verbatim example from an exemplar. Never write voice rules from memory or imagination. -- Measure the real wpm from the creator's OWN transcript (word count / duration) and write it into `[owner] wpm`, replacing the estimate. -- Encode the rule explicitly in the bible: written voice is not spoken voice; only spoken transcripts define the spoken register, and reference creators inform but never replace the creator's own patterns. +Close with the runnability report, which is the actual deliverable of this stage: -If declined, the spec stays in place as the build instructions and the unbuilt bible goes on the step 8 pending list. - -### 5. CLI tools and asset lanes - -CLI-tool-first: a registered CLI backed by a subscription the creator already pays for is the preferred lane for every `[assets]` slot; metered APIs are an explicit opt-in choice, never a silent default. - -Ask what they use for image generation, video generation, and offloaded research: any agentic or generation CLI they run. For each tool: - -- name, capabilities (image/video/research/...), the exact headless invocation, preferred models. The `headless` template is parsed with POSIX shell quoting on every OS (prefer forward slashes in template paths; backslash is an escape), and the tool name is resolved with a PATH lookup at run time, so npm-installed tools register by bare name on Windows too (.cmd/.exe shims are found via PATH+PATHEXT). -- The `notes` field: everything future sessions must remember about driving it (quirks, output behavior, what it is bad at). Write it down now; this is the memory that stops every session from rediscovering the tool. -- Verify end to end, with permission: first the version/help command (exists on PATH), then one small real invocation per registered capability (for example a tiny test image into a scratch folder), confirming the output file actually exists. Record the result in `notes` as verified end-to-end with the ISO date, or unverified. - -Write each as a `[[modules.manticore.tools]]` block. Then set the `[assets]` lanes (image-provider, video-provider, escalation-provider): default each lane to a registered, verified tool name. Only if the creator explicitly chooses a metered API lane instead, set that provider value and confirm its key env var, relaying implemented/planned status honestly. A lane with no good answer stays empty: mc-assets stops and asks at farming time rather than billing anyone by default. - -### 6. Editor integration - -Native scripting is the default DaVinci Resolve path: the cut stage always exports an fcpxml timeline, and Resolve-side automation drives Resolve's own scripting API; no MCP server is required for any shipped lane. Only if the creator ALREADY runs a Resolve MCP server and wants skills to use it: check `claude mcp list` (or the harness equivalent), and record `davinci-resolve = true` under `[mcp]` once confirmed. Otherwise record false and move on; do not suggest installing one. Other editors need nothing in this step. - -### 7. Keys and .env.example - -Key talk happens only inside opt-in branches. For each metered lane the creator explicitly chose in steps 3 and 5 (a metered transcription provider, a metered asset API): confirm the env var name recorded in the config, check whether that env var is set (presence only; NEVER read or echo values), and, inside that branch only, tell them where that vendor issues keys. If nothing metered was chosen (the local-first default), skip key sourcing entirely. - -Then scaffold `{project-root}/.env.example`: - -- List exactly the env vars the resolved config references: every non-empty `api-key-env` / `*-key-env` value that a configured lane actually uses. Possibly none; if none, write no file and say so. -- One line per var with a one-line source note (where the key comes from). -- A header comment: real values never go in TOML, in chat, or in this file; copy to `.env` or export in the shell. -- Idempotent like everything else: update an existing `.env.example` surgically and never touch a real `.env`. - -### 8. Write and confirm - -Write the interview results as the `[modules.manticore]` table (with its sub-tables: owner, paths, video, render, style, cta, live, editor, transcription, assets, audio, mcp, and `[[modules.manticore.tools]]` entries) into `{project-root}/_bmad/custom/config.toml`. Edit surgically: create the file if needed, preserve everything else in it (other modules configure themselves there too), and preserve any sections the creator skipped. Mention `config.user.toml` for personal overrides in shared repos. Verify by running `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore` and showing the resolved summary. - -Close with the honest runnability report: - -- Locked behavior: what will actually happen on the first project with these settings. Render-first preview and offered final per `[render]`, the graphics-frequency tier, the CTA inventory, the transcription lane and whether THIS machine can run it, the audio lanes and whether the engine workspace is built yet, the editor timeline format, and whether the HyperFrames graphics skills are installed (or deferred to the first graphics run). -- Lane status: implemented vs planned for every configured lane, straight from the customize.toml comments. Never claim a planned lane works. -- Pending gaps, flagged loudly: missing headshots (thumbnails are blocked), unbuilt voice bible, placeholder Production Bible sections, unverified tools, empty asset lanes (mc-assets will stop and ask). -- Capability note: check whether the harness has browser automation available; packaging research degrades without it, and the report says so when it is absent. +- Locked behavior: what will happen on the first project with these settings. Render-first preview and offered final, the graphics-frequency tier, the CTA inventory, the transcription lane and whether THIS machine can run it, the audio lanes and whether the workspace is built, the timeline format, and whether the HyperFrames skills are installed or deferred. +- Lane status: implemented or planned for every configured lane, straight from the `{skill-root}/assets/studio-defaults.toml` comments. +- Pending gaps, flagged loudly: missing headshots (thumbnails blocked), unbuilt voice bible, placeholder bible sections, unverified tools, empty asset lanes. +- Whether the harness has browser automation. Packaging research degrades without it, and the report says so when it is absent. Point at mc-new to start the first project, and at the pending list as the highest-value next builds. @@ -178,6 +123,5 @@ Point at mc-new to start the first project, and at the pending list as the highe - Confirm before every install, every MCP add, every command that changes the system. - Presence checks only for secrets; never read, echo, or store key values. Keys never go in the TOML, in chat, or in `.env.example`. - Paid and metered vendors are opt-in only: no vendor key name, dashboard, or pricing mention outside the branch where the creator explicitly chose that lane. -- Never claim a planned lane works; relay implemented/planned status honestly everywhere it comes up. -- Re-runs edit the existing config surgically; a re-run with no changes writes nothing. -- Never touch `{project-root}/_bmad/config.toml` (installer-owned); Manticore's home is the `custom/` layer. +- Never claim a planned lane works. `{skill-root}/assets/studio-defaults.toml` marks each transcription, editor and audio lane implemented or planned; relay that status honestly everywhere it comes up. +- Never touch `{project-root}/_bmad/config.toml`, which is installer-owned. Manticore's home is the `custom/` layer. diff --git a/skills/mc-script/assets/craft-checklist.md b/skills/mc-setup/assets/craft-checklist.md similarity index 65% rename from skills/mc-script/assets/craft-checklist.md rename to skills/mc-setup/assets/craft-checklist.md index 7498534..cd7bfd1 100644 --- a/skills/mc-script/assets/craft-checklist.md +++ b/skills/mc-setup/assets/craft-checklist.md @@ -1,22 +1,22 @@ # Script Craft Checklist -The QA gate mc-script runs after weaving and linting, before presenting. Derived from studying high-retention scripted educational YouTube. Every rule is checkable against the text; fix failures, do not annotate around them. +The craft pass mc-script runs after weaving and linting, before presenting. Every rule is checkable against the text; fix failures, do not annotate around them. 1. Sentence 1 is a present-tense stakes claim; sentence 2 contains a proper noun. No greeting, no channel intro. 2. Hook shape: claim, named case, identity match ("people like you"), credibility ladder (others / people I met / me), explicit double promise (why + how), then into the video. -3. Cash every abstraction within one sentence: a number, a name, or a concrete scene. No paragraph without a numeral or proper noun. +3. Cash every abstraction within one sentence, with a number, a name, or a concrete scene. No paragraph without a numeral or proper noun. 4. Open the video's core question early, answer it late, and close the loop explicitly ("back to the question from earlier"). At least one loop held open for 60+ seconds. 5. Announce counts before lists ("four things") and number items aloud. Never an unannounced list. 6. At least 3 callbacks per 10 minutes to the script's own earlier lines. 7. Define jargon inline in one plain sentence: "X basically just means Y." 8. Default to "you"; "I" only for credibility and personal process. No "we," no passive voice. -9. Transitions are So/And/But/Okay/Now. Banned transitions: Furthermore, Moreover, Additionally, In conclusion, However-comma (these are also in the blacklist lint). +9. Transitions are So/And/But/Okay/Now. The stiff connectives (Furthermore, Moreover, Additionally, In conclusion) are blacklist patterns `lint_script.py` already fails on; However-comma is not, so it is yours to catch. 10. Repeat sentence stems for emphasis (anaphora); rewrite any elegant variation of a repeated idea. -11. Hedge only numbers, never claims, and hedge like a person ("around 10%," "maybe like 60 or 70%"). Never "it's important to note," "arguably," or both-sides pivots. +11. Hedge only numbers, never claims, and hedge like a person ("around 10%," "maybe like 60 or 70%"). No "arguably," no both-sides pivots. 12. One contrast pair per major section ("minutes, not months"). 13. Analogies anchor in the viewer's daily life (their notes app, their subscriptions), never generic metaphor (journeys, landscapes). 14. Escalate specificity inside stories; save the biggest fact for last. 15. Raise stakes mid-video, then LOWER the barrier right before the CTA. End on ease, not pressure. 16. After any dense stretch, a 3 to 6 word punch sentence. No more than ~4 consecutive sentences over 20 words. -Rules 1, 2, and 15 are checked against the creator's format profile: some formats (course-lesson) deliberately open with orientation instead of stakes, and the profile wins over this list where they conflict. +The stakes-claim opening, the hook shape, and the lower-the-barrier rule are checked against the creator's format profile: some formats (course-lesson) deliberately open with orientation instead of stakes, and the profile wins where they conflict. diff --git a/skills/mc-setup/assets/formats/livestream-pack.md b/skills/mc-setup/assets/formats/livestream-pack.md index 51c1488..1f238c2 100644 --- a/skills/mc-setup/assets/formats/livestream-pack.md +++ b/skills/mc-setup/assets/formats/livestream-pack.md @@ -1,7 +1,7 @@ --- format: livestream-pack stages: [new, stream-pack, final, retro] -engine_overlays: ograf +engine_overlays: hyperframes engine_stingers: hyperframes generated_broll: banned beat-types: [starting-soon-scene, brb-scene, ending-scene, full-overlay, lower-third, topic-card, stinger] @@ -27,7 +27,7 @@ Not a video. One run of mc-stream-pack producing a complete OBS asset pack from - Scenes are reactive via the `window.obsstudio` JS API (countdown resets on scene activation, lower thirds re-trigger entrance on visibility) with a plain-browser fallback. - Stinger transition: one HyperFrames comp rendered twice (VP9 yuva420p WebM for OBS, ProRes 4444 MOV for the editor lane), 1 to 2 seconds. Baked alpha scene and lower-third deliverables list WebM VP9 alpha (libvpx-vp9 yuva420p) for OBS on any platform alongside the ProRes 4444 MOV; render_verify.py can transcode and verify the WebM from the ProRes master in one step. - vMix note: vMix rejects MP4 stingers and prefers PNG sequences; when the live tool is vMix, deliver a PNG sequence or the ProRes 4444 MOV instead of WebM. Wirecast takes the ProRes 4444 MOV directly. -- Lower thirds and topic cards as OGraf (via the mc-ograf skill), standalone-capable and SPX-GC-compatible for click-to-trigger later. +- Lower thirds and topic cards as self-contained local HTML, styled from `{brand-path}/tokens.json`, drivable by SPX-GC or an OBS browser source for click-to-trigger later. ## Verification, not vibes diff --git a/skills/mc-setup/assets/formats/livestream-vod.md b/skills/mc-setup/assets/formats/livestream-vod.md index 5f51866..a1678a3 100644 --- a/skills/mc-setup/assets/formats/livestream-vod.md +++ b/skills/mc-setup/assets/formats/livestream-vod.md @@ -33,32 +33,25 @@ Post-production of a recorded livestream (or any long single-take source) into a ## Templates -- None yet. A recurring show should promote its packaging spec (locked thumbnail anchors vs per-episode variables) into the brand `templates/` folder so mc-package can generate against it. +- None yet. A recurring show should promote its packaging spec (locked thumbnail anchors vs per-episode variables) into the `{brand-path}/templates/` folder so mc-package can generate against it. -## Learnings - -(mc-retro appends here; newest first, ISO dated. Seeded below from the module's first production runs, genericized.) - -### 2026-07-07 asset tiers, live loading, and post-publish hygiene - -- Two-tier assets: evergreen chrome (scene stills, the persistent frame/bar system, host and CTA lower thirds, the show mark) is built ONCE and reused every episode from `common/` and `{brand-path}/`. Only per-episode TOPIC graphics get made fresh, and they are mined from the episode OUTLINE or plan before the show, never rebuilt from the transcript after. -- Loaded live, not re-edited later: the graphics pack must be prepped and switchable in the streaming tool BEFORE going live. The live pack itself is the livestream-pack format via mc-stream-pack; this VOD format consumes its output. -- Post-publish hygiene, once the published master is confirmed safe: purge reproducible render scratch from `work/` (intermediate segments, the baked review render, the debug image trail), keep transcripts in a keeper location, hold `deliverables/` to one blessed asset per slot (alternates stay in `work/`), and promote newly reusable chrome to `common/` and `{brand-path}/`. - -### 2026-07-06 format pivot after the first full run: publish as-is, graphics go live - -- After completing one full post-hoc re-edit, the creator pivoted the format: the VOD publishes as-is with a light trim; no overlay baking in post. The graphics effort moves ahead of the stream as a live-triggered per-episode scene and asset pack. Post-production shrinks to packaging, shorts, and socials. The design rules below still govern the live pack's look. - -### 2026-07-06 first full run: design rules learned through review corrections +## Design rules - Never shrink the source video to make room for UI; overlay on the full frame. No solid bars or panels behind persistent UI; elements float directly on the video with their own treatment. - Popups are large, centered, and straight; they fly in and out fast (about 0.35s) with a sound cue. Never small, askew, or corner-placed where they cover faces. - Photos get snug native-aspect frames; never fixed-size cards with filler panels. - The formal brand palette wears out fast on in-video graphics. Casual treatments and the referenced platform's own colors often read better on overlays; save the formal palette for professional surfaces. Record the split in the Production Bible. -- Real imagery beats invented: use real screenshots, box art, and artwork wherever the real thing exists; invent only when it does not. Verify every claim-bearing graphic against the transcript; an analysis sweep once invented an offer the creator never made. +- Real imagery beats invented: use real screenshots, box art, and artwork wherever the real thing exists; invent only when it does not. Verify every claim-bearing graphic against the transcript, because an analysis lane can invent an offer or claim that was never made. - No live-tense wording on VOD graphics ("Enjoying the stream?"); persistent show branding stays. - Community member names appear exactly as the creator uses them on stream, never normalized. - Emoji inside rasterized SVG text render as black silhouettes; keep SVG text vector-only. - Render mechanics: crop source-edge defects before upscaling (crop then scale keeps aspect); infinite generator sources need `shortest=1` AND an explicit `-t` cap or the render runs away; splitting a long render into parallel segments and concatenating halves wall-clock time; always extract spot-check frames before delivering a render. -- Packaging: title and thumbnail must complement, never repeat; whichever carries the promise, the other carries the intrigue. A recurring show keeps a thumbnail template (locked anchors vs per-episode variables) in the brand folder. -- Keep deliverables lean: one blessed asset per slot in `deliverables/`; alternates stay in `work/`. +- Packaging: title and thumbnail must complement, never repeat; whichever carries the promise, the other carries the intrigue. + +## Post-publish hygiene + +Once the published master is confirmed safe: purge reproducible render scratch from `work/` (intermediate segments, the baked review render, the debug image trail), keep transcripts in a keeper location, hold `deliverables/` to one blessed asset per slot (alternates stay in `work/`), and promote newly reusable chrome to `common/` and `{brand-path}/`. + +## Learnings + +(mc-retro appends here; newest first, ISO dated.) diff --git a/skills/mc-setup/assets/formats/screen-tutorial.md b/skills/mc-setup/assets/formats/screen-tutorial.md index 1b74add..6a3750d 100644 --- a/skills/mc-setup/assets/formats/screen-tutorial.md +++ b/skills/mc-setup/assets/formats/screen-tutorial.md @@ -20,7 +20,7 @@ Talking head plus screen recording. The teaching happens on screen; the pipeline - Real UI only. Generated b-roll is BANNED in this format (UI accuracy rule); that is why the assets stage is absent from the stage list. - Beat types extend talking-head with: zoom/pan on the screen recording, UI callouts (boxes, arrows, key-press chips), and step counters. -- Callouts use the brand accent color on a subtle border stroke (per tokens.json); never obscure the UI element being discussed, point at it. +- Callouts use the brand accent color on a subtle border stroke (per `{brand-path}/tokens.json`); never obscure the UI element being discussed, point at it. - Screen recordings are captured at native resolution, constant frame rate, and cursor visible. - Creativity: restrained. The UI is the star; graphics clarify, never decorate. Vary callout placement and pacing, not treatment. (mc-retro tunes this line per format.) diff --git a/skills/mc-setup/assets/formats/voiceover-explainer.md b/skills/mc-setup/assets/formats/voiceover-explainer.md index 7a53ab7..1547ab6 100644 --- a/skills/mc-setup/assets/formats/voiceover-explainer.md +++ b/skills/mc-setup/assets/formats/voiceover-explainer.md @@ -28,7 +28,7 @@ Narration status: creator-recorded narration is the default and the honest recom ## Engine defaults - Diagrams and slides: HyperFrames blocks or plain HTML/SVG comps (the creator's call per video; both read `{brand-path}/tokens.json`). -- Farmed stills/clips: per the configured `[assets]` lanes and the `PIPELINE.md` engine policy. No vendor is assumed; if a lane is unset, the assets stage stops and asks. +- Farmed stills/clips: per the configured `[assets]` lanes and the pipeline's engine policy. No vendor is assumed; if a lane is unset, the assets stage stops and asks. ## Templates diff --git a/skills/mc-setup/assets/production-bible-spec.md b/skills/mc-setup/assets/production-bible-spec.md index 26b0346..8fc974f 100644 --- a/skills/mc-setup/assets/production-bible-spec.md +++ b/skills/mc-setup/assets/production-bible-spec.md @@ -1,20 +1,20 @@ # Production Bible: Build Spec -The Production Bible is the visual half of the taste system (the voice bible is the verbal half). It is the styling contract every visual stage reads before authoring anything, and the file mc-retro ratchets when the creator corrects visual output. It lives at `{brand-path}/production-bible.md`. Machine-readable constants stay in `tokens.json`; the bible is the taste contract in prose plus structured style tokens. mc-setup copies this spec there as the placeholder until it is built. +The Production Bible is the visual half of the taste system (the voice bible is the verbal half). It is the styling contract every visual stage reads before authoring anything, and the file mc-retro ratchets when the creator corrects visual output. It lives at `{brand-path}/production-bible.md`. Machine-readable constants stay in `{brand-path}/tokens.json`; the bible is the taste contract in prose plus structured style tokens. mc-setup copies this spec there as the placeholder until it is built. ## The seven sections 1. Brand usage scope. Global rules first, then per-project-type override sections (per-format, per-series). Records where the corporate theme applies and where it must NOT: not everything gets the corporate palette, and accent colors wear out when overused. -2. Animation and motion look-and-feel. The feel in words (snappy, smooth, dramatic), entrance and exit conventions (for example fly-in and fly-out with optional whoosh SFX), mapped onto the motion values in `tokens.json`. +2. Animation and motion look-and-feel. The feel in words (snappy, smooth, dramatic), entrance and exit conventions (for example fly-in and fly-out with optional whoosh SFX), mapped onto the motion values in `{brand-path}/tokens.json`. 3. Overlay and popup aesthetic. Surface treatment (solid, glass, gradient, neon, flat, native-platform), blur, border, corner radius, shadow or glow, texture. Placement rules: overlays are large, centered, straight, composited over full-frame video in detected safe zones around talking heads; never letterbox the source to make room; no solid bars behind persistent UI; photos get snug native-aspect frames, never uniform letterboxed panels. Reference screenshots the creator supplies or wants to emulate are stored beside the bible. 4. Image-type policy. Preferred lanes per purpose: SVG or diagrammatic builds for anything whose text must be accurate, generative imagery for what does not exist, real verified imagery first for anything that does. Sourcing hierarchy: real, then generative, then hand-built text card. Also lists the creator's own asset libraries and their locations. -5. Visual density. The graphics-frequency tier (high, medium, low), per-format overrides, and the variety quota (see the density-and-creativity reference shipped with mc-beats). +5. Visual density and variety. The numbers every beat plan is held to. Three are owned here, stated globally first and then overridden per project type, since a short and a course lesson do not carry the same load: the beats-per-minute floor for each tier, which sets the minimum beat count for a runtime; how many distinct beat types a video over a stated length must use, and the share of rows any one type may take; and the cap on static text-only cards, plus whether two may run back to back. The fourth number, the graphics-frequency tier (high, medium, low), is only mirrored here: it is resolved from `graphics-frequency` in `[style]` of the studio config, overridden for one format by that format profile's frontmatter, so a tier change is made there and the mirror follows. Written as numbers rather than adjectives, since nothing scripted checks them and a plan cannot be held against an adjective. 6. CTA configuration. The creator's CTA inventory and appetite (mirroring `[cta]` in the studio config), plus the native-platform styling rule: a subscribe element reads YouTube-red, a community element reads that platform's own colors. 7. Learnings log. ISO-dated, append-only, one-way ratchet, exactly like format-profile Learnings. ## How mc-setup builds it -The video style interview fills the bible interactively. The creator supplies any of: (a) screenshots of overlays they have shipped, (b) reference screenshots, video links, or creators to emulate, or (c) a described aesthetic; setup distills these into the structured tokens plus prose. For emulation links, setup echoes back the distilled takeaways (edit rhythm, humor and meme usage, chart and dataviz polish, caption and overlay style) and the creator confirms them before they land in the bible. Density, image-type, CTA, and animation-feel answers land in BOTH places: the config keys (`[style]`, `[cta]`) for mechanical consumption and the bible for taste. A section left genuinely unanswered stays a marked placeholder, and the setup summary flags it as a pending gap. +The video style interview fills the bible interactively. The creator supplies any of: (a) screenshots of overlays they have shipped, (b) reference screenshots, video links, or creators to emulate, or (c) a described aesthetic; setup distills these into the structured tokens plus prose. For emulation links, setup echoes back the distilled takeaways (edit rhythm, humor and meme usage, chart and dataviz polish, caption and overlay style) and the creator confirms them before they land in the bible. Density, image-type, CTA, and animation-feel answers land in BOTH places: the config keys (`[style]`, `[cta]`) for mechanical consumption and the bible for taste. Section 5 is the exception, because only its tier has a config key: the floor, the variety quota, and the card cap live in the bible alone. Ask for them as numbers with a starting point on the table (six distinct types in anything over five minutes, no type above 40% of rows, static cards under 25% of rows and never two in a row, per-minute floors of 3 at high, 1.5 at medium, 0.7 at low), and write down what the creator lands on rather than the suggestion. A section left genuinely unanswered stays a marked placeholder, and the setup summary flags it as a pending gap. ## How it evolves @@ -25,8 +25,8 @@ The video style interview fills the bible interactively. The creator supplies an ## Consumers -- mc-beats reads it in step 1 alongside the format profile; density tier and beat-type choices must conform, and its checklist requires composition consistency with the stated overlay style and image-type policy. -- mc-graphics and mc-ograf read it before authoring anything; it is the styling contract beyond `tokens.json`. +- mc-beats reads it on activation alongside the format profile; section 5's numbers bound the plan it writes, and its checklist requires composition consistency with the stated overlay style and image-type policy. +- mc-graphics reads it before authoring anything; it is the styling contract beyond `{brand-path}/tokens.json`, and section 2 outranks any motion recipe or template shipped with a skill. - mc-assets: the image-type policy governs lane choice per asset; the sourcing hierarchy applies. - mc-package and mc-stream-pack: thumbnail style, series templates, and the CTA section. - mc-retro and the studio agent are the writers, per above. diff --git a/skills/mc-setup/customize.toml b/skills/mc-setup/assets/studio-defaults.toml similarity index 69% rename from skills/mc-setup/customize.toml rename to skills/mc-setup/assets/studio-defaults.toml index 987aefd..42bc2e2 100644 --- a/skills/mc-setup/customize.toml +++ b/skills/mc-setup/assets/studio-defaults.toml @@ -1,22 +1,11 @@ -# DO NOT EDIT -- overwritten on every update. -# # Default studio configuration for BMad Manticore, in full. mc-setup reads # these defaults, interviews the creator, and writes the result to # [modules.manticore] in {project-root}/_bmad/custom/config.toml # (personal overrides go in config.user.toml next to it; resolve with # {project-root}/_bmad/scripts/resolve_config.py --key modules.manticore). # +# This file is the authority on the schema and on every default in it. # API keys NEVER go in any TOML; only the names of env vars that hold them. -# -# Override files for this skill itself: -# {project-root}/_bmad/custom/mc-setup.toml (team) -# {project-root}/_bmad/custom/mc-setup.user.toml (personal) - -[workflow] - -activation_steps_prepend = [] -activation_steps_append = [] -persistent_facts = [] # --- Studio defaults: the seed values for [modules.manticore] ------------- @@ -38,13 +27,14 @@ links = [] [defaults.paths] # Where your studio data lives, relative to project root unless absolute. -# brand-path holds tokens.json, voice-bible.md, blacklist.md, exemplars/, headshots/. +# brand-path holds tokens.json, production-bible.md, voice-bible.md, blacklist.md, +# craft-checklist.md, exemplars/, headshots/. brand-path = "manticore/brand" # formats-path holds your editable format profiles (learnings accumulate there). formats-path = "manticore/formats" # projects-path holds one folder per video. projects-path = "manticore/projects" -# engines-path holds your HyperFrames/OGraf engine workspaces. +# engines-path holds your HyperFrames engine workspaces. engines-path = "manticore/engines" [defaults.video] @@ -59,7 +49,7 @@ fps = 30 # beats approval, and an offered final-quality render at gate 4. The editor # timeline export and all assets (edl.json, cutplan.md, overlays) are ALWAYS # produced alongside, so you can jump into your editor at any step. The -# creator confirms this default during setup (step 3b). +# creator confirms this default during setup. self-render = true # Preview render: small and fast, for iteration. preview-height = 720 @@ -81,7 +71,8 @@ loudness-target = -14 # "high" a beat roughly every 10 to 20 seconds # "medium" a beat roughly every 20 to 45 seconds # "low" a beat roughly every 45 to 90 seconds -# Interviewed at setup; per-format overrides live in the Production Bible. +# Interviewed at setup; a format profile's frontmatter overrides the tier for +# that format, and the Production Bible mirrors the resolved value. graphics-frequency = "medium" [defaults.cta] @@ -115,8 +106,6 @@ name = "resolve" # "none" skip timeline export; you get edl.json + cutplan.md + preview.mp4 # and cut manually in your editor (Descript and others) timeline-format = "fcpxml" -# true ONLY if your editor imports OGraf packages natively (DaVinci Resolve 21+). -ograf-editable = false [defaults.transcription] @@ -153,7 +142,7 @@ image-provider = "" video-provider = "" escalation-provider = "" # Env var names for the metered API lanes. Ship blank, like -# [defaults.transcription] api-key-env: mc-setup step 5 fills the vendor's +# [defaults.transcription] api-key-env: tool registration fills the vendor's # key name only when you explicitly opt into that metered lane, and key # sourcing is discussed only inside that opt-in branch. xai-api-key-env = "" @@ -204,6 +193,89 @@ workspace = "audio-lab" # as "not available" and fall back gracefully. davinci-resolve = false +[defaults.cut] + +# Silence floor in dBFS for analyze_audio.py, passed as --noise at step 3a. +# This is NOT taste and NOT a universal: it is a property of the creator's +# room and mic. Room tone on a decent mic sits well under -30; a noisy room +# needs -35 or -40. It matters more than it looks, because the audio map is +# the timing source of truth for the whole stage AND the input to the +# transcript completeness gate: set it too high and almost nothing registers +# as silent, which starves the cutter and makes the gate misbehave. +# +# Note this is the FLOOR (what counts as silent), not the map resolution. +# analyze_audio's --map-granularity is deliberately not exposed: the +# calibration record establishes it must stay finer than every consumer, so +# it is a mechanic, not a knob. +silence-floor-db = -30.0 + +# Extra flags appended to the cutplan.py invocation. Empty means the script +# defaults: min-silence 0.30, keep-ms 200, retake window 16, run 3, +# section-run 8, section-window-s 45. +# +# NOTE: cutplan.py REFUSES (exit 2) when -o, --audio-map or --voice-bible +# appear twice, so this string cannot redirect the output or swap the timing +# source of truth. The boundary below is enforced, not just documented. +# +# Pacing (the two that change how the edit FEELS): +# --min-silence 0.30 interior silences shorter than this are the +# speaker's rhythm and are left completely alone. +# Raise it for a slower, more deliberate read. +# --keep-ms 200 breathing room kept inside each tightened silence. +# Lower is tighter; 0 is machine-gunned, don't. +# +# Section re-reads (--section-run, --section-window-s): how long a repeated +# run must be, and how close in TIME, to count as a redo rather than a +# deliberate callback. Widen the window for a rambling delivery. +# +# Interview sources: --marker-cues overrides the default cue "question from +# the interviewer" (pass --marker-cues "question from claude" for projects +# recorded against the older convention). mc-setup's marker-cue interview +# question records a non-default cue here. +# +# --blooper-cues overrides the expletive/reset vocabulary; per-creator, so +# set it here when the shipped vocabulary misses. +# +# Note: --audio-map and --voice-bible are passed by the SKILL (resolved +# paths), not from this file. +cutplan-flags = "" + +# ESCAPE HATCH for render flags the studio config does not model, e.g. +# "--segment-target-seconds 300". Encoder settings have ONE home and it is not +# here: preview-height, preview-crf, final-crf, loudnorm and loudness-target +# live in [render] / [video], and the skill emits those first with this +# string appended last. Because the last occurrence wins, restating a key +# [render] owns here silently overrides it, and which value applies then +# depends on emission order rather than on any stated rule. So do not put +# --height, --crf, or the loudness flags in these. +preview-flags = "" +final-flags = "" + +[defaults.packaging] + +# Title+thumbnail candidates to present (series projects present them as +# exactly this many A/B pairs). +candidates = 3 + +# Hard cap on on-image thumbnail hook words (face-plus-hook convention). +hook-words-max = 4 + +# Title length cap in characters (front-load what matters; longer gets cut +# off in feeds). +title-max-chars = 60 + +# Proof width in pixels for the mandatory downscale verification +# (verify_thumb.py). 120 approximates the smallest feed rendering. +verify-width = 120 + +[defaults.retro] + +# Path globs, relative to the project folder, that the wrap lane must +# never list for reclaim (in addition to the built-in never-reclaim set: +# source footage, transcripts, edl.json, cutplan, overlays, project.json, +# and the published master). +preserve = [] + # --- CLI tool registry ---------------------------------------------------- # The creator's [[tools]] entries are written into [modules.manticore] by # mc-setup, one block per agentic/generation CLI. The `notes` field is the diff --git a/skills/mc-setup/assets/voice-bible-spec.md b/skills/mc-setup/assets/voice-bible-spec.md index 0c7383e..a7ccd08 100644 --- a/skills/mc-setup/assets/voice-bible-spec.md +++ b/skills/mc-setup/assets/voice-bible-spec.md @@ -16,14 +16,34 @@ The voice bible is the primary taste file for mc-script. It lives at `{brand-pat - Filler behavior worth keeping (authenticity) vs cutting. - Speaking rate: MEASURE it from a real transcript (word count / duration) and write the number into the studio config `[owner] wpm`. On-camera educators often run 180+; the generic 145 estimate is usually wrong. 4. Every rule in the bible must cite at least one verbatim example from an exemplar. +5. Write the cadence block (below). Step 3's "filler behavior worth keeping vs cutting" is a prose finding; the cadence block is that same finding in a form the cutter can actually read. + +## The cadence block (machine-readable, required by mc-cut) + +The cut stage reads this block directly. Without it, mc-cut falls back to a deliberately tiny built-in soft-filler list, which is safe but generic. + +Put a fenced block tagged `cadence` anywhere in the file: + +```cadence +keep: so, here's the thing, which means, look, now +cut: basically, honestly, kind of, you know +``` + +- `keep` is the creator's connective glue: words that read as filler in the abstract but are this speaker's rhythm. Anything listed here is never flagged as a filler, including hard fillers, so a creator whose "hmm" is a deliberate beat can protect it. +- `cut` extends the soft-filler list with this creator's own tics. +- On conflict, `keep` wins. Preserving rhythm is the safer failure: a kept filler is a small blemish, a cut cadence word changes how the creator sounds. +- Phrases are allowed on both lists; match is case-insensitive and punctuation-stripped. + +Why this exists: the cutter reads these lists, not the prose above them. A cadence rule stated only in prose cannot stop the cutter flagging every sentence-initial "So" the creator uses as connective glue. Derive both lists from the exemplars in step 3, the same evidence-quote discipline as every other rule here. ## Consumers - mc-script reads it before weaving and checks output against it. - mc-outline uses the hook section to shape hook candidates. +- mc-cut reads the cadence block (`cutplan.py --voice-bible`) so filler detection respects the creator's rhythm. - mc-retro appends corrections when the creator flags voice misses. ## Also in the brand folder -- `exemplars/` as above. -- `headshots/`: 3 to 6 approved reference photos of the creator, the input for face-consistent thumbnail generation. Approved photos only; mc-package never uses arbitrary frames from footage. +- `{brand-path}/exemplars/` as above. +- `{brand-path}/headshots/`: 3 to 6 approved reference photos of the creator, the input for face-consistent thumbnail generation. Approved photos only; mc-package never uses arbitrary frames from footage. diff --git a/skills/mc-setup/references/bootstrap.md b/skills/mc-setup/references/bootstrap.md new file mode 100644 index 0000000..88e1ab7 --- /dev/null +++ b/skills/mc-setup/references/bootstrap.md @@ -0,0 +1,56 @@ +# Bootstrapping BMad core + +Load this when any of the three paths checked on activation is missing, which means +the project is not BMad-initialized. Manticore installs nothing itself: it runs the +bmad-method installer and then verifies the result. + +Say the project is not initialized and confirm before running anything. The installer +writes `{project-root}/_bmad/` plus IDE integration files for the chosen tool, so this +is a system change the creator agrees to first. + +## Resolve the tool id + +`claude-code` under Claude Code. Otherwise run `npx -y bmad-method install --list-tools` +and let the creator pick. + +## Install + +No `{project-root}/_bmad/` at all: + +``` +npx -y bmad-method@latest install --directory {project-root} --modules core --tools -y +``` + +Never omit either flag. A bare `-y` installs the default module set rather than just +core, and fresh `-y` installs fail without `--tools`. + +`{project-root}/_bmad/` exists but is incomplete: + +``` +npx -y bmad-method@latest install --directory {project-root} -y +``` + +That quick-update re-syncs `{project-root}/_bmad/scripts/` and keeps configured tools. +If it fails, retry with `--action update --modules core --tools ` added. + +## Verify before continuing + +- `resolve_config.py` exists under `{project-root}/_bmad/scripts/`. +- `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore` + exits 0. Empty output just means the interview has not run. + +If uv itself is missing, bootstrap uv first (see the dependencies section of SKILL.md), +then verify. + +## When it will not install + +On verification failure, stop and surface the installer output. Never hand-copy scripts +or vendor a resolver of your own; a project running against a copied resolver diverges +from every other BMad Method module in it. + +If npx, node, or the network is unavailable, or the creator declines, have them run +`npx bmad-method install` interactively from the project root (core alone is enough for +Manticore), then re-run mc-setup. + +A stale npx cache can serve an old CLI that lacks current flags. On unknown-option +errors, run `npm cache clean --force` and retry. diff --git a/skills/mc-setup/references/migration-0x.md b/skills/mc-setup/references/migration-0x.md new file mode 100644 index 0000000..7bfff32 --- /dev/null +++ b/skills/mc-setup/references/migration-0x.md @@ -0,0 +1,64 @@ +# Migrating a 0.x studio + +Load this when `[modules.manticore]` exists but is missing any of the 1.0 tables +(`[render]`, `[style]`, `[cta]`, `[live]`, `[audio]`). That studio was configured +before 1.0. + +Say so, then migrate rather than re-interviewing. Every existing value is already the +creator's answer and stays untouched; this pass only fills what 1.0 added. + +## Backfill the config + +- Write the missing tables in from `[defaults]` (render, style, cta, live, audio, plus + the mechanical `cut`, `packaging` and `retro` sub-tables), editing the existing config + surgically. +- If `[transcription] api-key-env` names a key the configured local provider never + uses, blank it. Metered keys are set only when a metered provider is chosen. +- If the `[assets]` lanes still carry pre-1.0 defaults pointing at a metered API the + creator never opted into or verified, flag it in the closing report and offer to + repoint them at a registered CLI tool. Leaving them empty is also fine: mc-assets + asks at farming time. + +## Refresh the format profiles + +For every profile in `{formats-path}` that also ships in `{skill-root}/assets/formats/`: + +``` +uv run {skill-root}/scripts/merge_profile_frontmatter.py \ + --shipped {skill-root}/assets/formats/.md \ + --studio {formats-path}/.md +``` + +It adds only the frontmatter keys new in 1.0 (`beat-types`, `density`, and any future +ones) that stages like mc-beats require. It never overwrites an existing key, the +creator's prose, or the Learnings. Then copy in any newly shipped profile that does not +exist in `{formats-path}` yet. + +## Two things that moved + +Interview footage recorded against the pre-1.0 marker cue ("question from claude") still +needs to segment. Offer the marker-cue question and record it as `cutplan-flags` in the +studio config's `[cut]` sub-table so cutplan keeps working on that footage. + +A pre-1.0 series or thumbnail template at the brand root (for example +`thumbnail-template.md`) predates the `{brand-path}/templates/.md` contract. +Offer to move it there, named for the series it describes, so mc-package finds it. + +## Then run the delta + +Interview only what 1.0 added: render consent, the video style interview, and audio +lanes. + +Copy in every shipped brand template `{brand-path}` does not have yet, above all +`{skill-root}/assets/craft-checklist.md`: a 0.x brand folder predates it and mc-script +stops on activation without it. Never overwrite what is already there. + +Scaffold `{brand-path}/production-bible.md` seeded from what already exists (tokens.json, +shipped overlays, exemplars, format-profile learnings) plus the style answers, never from +a blank slate. This studio has history; a blank bible throws it away. + +Offer without forcing: the HyperFrames graphics skills, headshot collection, the guided +voice bible, `.env.example`. + +Finish with the normal write-and-report so the migrated config is verified and the +pending gaps are named. diff --git a/skills/mc-setup/references/stack-linux.md b/skills/mc-setup/references/stack-linux.md index 50d671c..d97c358 100644 --- a/skills/mc-setup/references/stack-linux.md +++ b/skills/mc-setup/references/stack-linux.md @@ -1,6 +1,6 @@ # Linux default stack -Selected when check_deps.py reports os Linux. The GPU verdict (nvidia, amd, intel, none, unknown) splits the transcription extra and the encoder ladder. Research basis: platform and capabilities audit, 2026-07-21. +Selected when check_deps.py reports os Linux. The GPU verdict (nvidia, amd, intel, none, unknown) splits the transcription extra and the encoder ladder. ## Default stack diff --git a/skills/mc-setup/references/stack-macos.md b/skills/mc-setup/references/stack-macos.md index 07e37ef..c3f0cc8 100644 --- a/skills/mc-setup/references/stack-macos.md +++ b/skills/mc-setup/references/stack-macos.md @@ -1,6 +1,6 @@ # macOS default stack -Selected when check_deps.py reports os Darwin. Apple Silicon is the module's reference platform; everything here is the shipped default behavior, confirmed with the creator during the mc-setup interview. Research basis: platform and capabilities audit, 2026-07-21. +Selected when check_deps.py reports os Darwin. Apple Silicon is the module's reference platform; everything here is the shipped default behavior, confirmed with the creator during the mc-setup interview. ## Default stack (Apple Silicon) diff --git a/skills/mc-setup/references/stack-windows.md b/skills/mc-setup/references/stack-windows.md index 0e886e3..17da82b 100644 --- a/skills/mc-setup/references/stack-windows.md +++ b/skills/mc-setup/references/stack-windows.md @@ -1,6 +1,6 @@ # Windows default stack -Selected when check_deps.py reports os Windows. The GPU verdict (nvidia, intel, amd, none, unknown) splits the stack below. Research basis: platform and capabilities audit, 2026-07-21. +Selected when check_deps.py reports os Windows. The GPU verdict (nvidia, intel, amd, none, unknown) splits the stack below. ## Default stack (NVIDIA GPU) diff --git a/skills/mc-setup/scripts/lint_genericity.py b/skills/mc-setup/scripts/lint_genericity.py deleted file mode 100644 index e6dc3ab..0000000 --- a/skills/mc-setup/scripts/lint_genericity.py +++ /dev/null @@ -1,210 +0,0 @@ -#!/usr/bin/env python3 -# /// script -# requires-python = ">=3.11" -# /// -"""Genericity release gate: scan module content for dogfood-studio leakage. - -Nothing user-, brand-, or show-specific may ship in the module. This lint -scans the given paths (files or directories) for three classes of leak: - -1. Brand terms: personal, channel, or show names from the dogfood studio - (default: bmad, bmadcode, madison, pinkyd; case-insensitive). A hit is - allowed only when it sits inside a legitimate ecosystem usage (BMad - Method, BMad Manticore, bmad-method install commands, bmad-code-org - URLs, _bmad runtime paths, sibling module names) or inside an - ecosystem/support section of README.md or AGENTS.md. -2. Six-digit hex colors outside token example files. Grayscale values, - the shipped placeholder palette, test fixtures, .svg illustrations, and - the HTML slide decks under docs/ (self-contained illustrations) are - allowed; anything else is treated as a possible dogfood palette hex. -3. Absolute /Users/ machine paths. Never allowed. - -Usage: - uv run lint_genericity.py [ ...] - [--terms bmad,bmadcode,madison,pinkyd] - [--allow-extra REGEX] [--hex-allow RRGGBB] - -The calling skill or release checklist passes explicit paths (for the -release gate: skills/ docs/ README.md CHANGELOG.md, which covers the -format profiles under skills/mc-setup/assets/formats/); this script does -no config discovery. The lint skips its own source file and its test -suite (both necessarily contain the banned terms as data). - -Exit 0 clean, exit 1 findings, exit 2 usage error. -""" - -import argparse -import re -import sys -from pathlib import Path - -TEXT_SUFFIXES = { - ".md", ".py", ".toml", ".json", ".html", ".txt", - ".yaml", ".yml", ".css", ".js", ".svg", ".template", -} -SKIP_DIRS = {".git", "__pycache__", "node_modules", ".venv"} -# The lint's own source and its test suite necessarily contain the banned -# terms as data (the default term list and its fixtures); skip both. -# The plugin manifest legitimately carries the module author's identity and -# the module keywords; it is not shipped studio content. -SELF_NAMES = {"lint_genericity.py", "test-lint_genericity.py", "marketplace.json"} - -DEFAULT_TERMS = ["bmad", "bmadcode", "madison", "pinkyd"] - -# Legitimate ecosystem usages: a brand-term hit fully inside a match of one -# of these (case-insensitive) is allowed. -DEFAULT_ALLOW_CONTEXTS = [ - r"bmad method", - r"bmad manticore", - r"bmad-method", # npx bmad-method install commands - r"bmad-code-org", # org URLs - r"bmad-manticore", # this repo's name in URLs - r"_bmad\b", # installed runtime dir ({project-root}/_bmad/) - r"bmad[- ]core", # the installed BMad core - r"bmad-install", # installer alias - r"bmad-autopilot", # sibling modules and builder skills - r"bmad-bmm", - r"bmad-workflow-builder", - r"bmad-agent-builder", - r"bmad-agent-analyst", - r"bmad-plugins-marketplace", - r"bmad agent([- ]skill)? pattern", - r"bmad structural rules", - r"bmad ecosystem", - r"bmad-initialized", # "the project is not BMad-initialized" - r"bmad-help", # the core help skill and the merged bmad-help.csv catalog - r"bmad help convention", # the ecosystem help-catalog convention by name - r"bmad code,? llc", # trademark holder in license notices - r"bmadcode\.com", # the maintainer's official site in ecosystem links - r"@bmadcode", # the maintainer's official channel handles -] - -# README.md / AGENTS.md sections where ecosystem links, support links, and -# trademark notices (channel handles, sponsor URLs, holder names) -# legitimately appear. -ECOSYSTEM_SECTION = re.compile(r"ecosystem|support|community|license|trademark", re.IGNORECASE) -ECOSYSTEM_FILES = {"README.md", "AGENTS.md"} - -HEX_RE = re.compile(r"#([0-9A-Fa-f]{6})\b") -# Shipped placeholder palette (tokens.template.json, scaffold_ograf.py, -# html_to_png.py safe-zone guides, mc-ograf preview chrome). Generic by -# design; not dogfood brand colors. -DEFAULT_HEX_ALLOW = {"4f8cff", "2f6fe0", "7faaff", "ff3355", "ffcc00", "0c1322"} - -USERS_PATH_RE = re.compile(r"/Users/[A-Za-z0-9_.-]+") - - -def die(msg: str) -> None: - print(msg, file=sys.stderr) - sys.exit(2) - - -def iter_files(paths: list[Path]) -> list[Path]: - files: list[Path] = [] - for p in paths: - if not p.exists(): - die(f"error: {p} not found") - if p.is_file(): - files.append(p) - continue - for f in sorted(p.rglob("*")): - if f.is_file() and not (SKIP_DIRS & set(f.parts)): - files.append(f) - return [f for f in files if f.suffix.lower() in TEXT_SUFFIXES and f.name not in SELF_NAMES] - - -def is_grayscale(hexval: str) -> bool: - h = hexval.lower() - return h[0:2] == h[2:4] == h[4:6] - - -def allowed_spans(line: str, contexts: list[re.Pattern]) -> list[tuple[int, int]]: - return [m.span() for ctx in contexts for m in ctx.finditer(line)] - - -def covered(span: tuple[int, int], allowed: list[tuple[int, int]]) -> bool: - return any(a <= span[0] and span[1] <= b for a, b in allowed) - - -def scan_file(path: Path, term_res: list[re.Pattern], contexts: list[re.Pattern], - hex_allow: set[str]) -> list[str]: - try: - text = path.read_text(encoding="utf-8") - except (UnicodeDecodeError, OSError): - return [] - findings: list[str] = [] - is_ecosystem_file = path.name in ECOSYSTEM_FILES - in_ecosystem_section = False - token_file = "tokens" in path.name.lower() - in_tests = "tests" in path.parts - # Illustrations carry their own palettes by design: .svg diagrams anywhere, - # and the self-contained HTML slide decks under docs/. Skill HTML templates - # are NOT exempt; a brand hex there would ship into studio output. - is_illustration = path.suffix.lower() == ".svg" or ( - path.suffix.lower() == ".html" and "docs" in path.parts) - - for lineno, line in enumerate(text.splitlines(), start=1): - heading = re.match(r"#{1,6}\s+(.*)", line) - if is_ecosystem_file and heading: - in_ecosystem_section = bool(ECOSYSTEM_SECTION.search(heading.group(1))) - - if not (is_ecosystem_file and in_ecosystem_section): - allowed = allowed_spans(line, contexts) - for term_re in term_res: - for m in term_re.finditer(line): - if not covered(m.span(), allowed): - findings.append( - f"{path}:{lineno}: [brand-term] {m.group(0)!r} in: {line.strip()[:120]}") - - if not (token_file or in_tests or is_illustration): - for m in HEX_RE.finditer(line): - h = m.group(1).lower() - if h not in hex_allow and not is_grayscale(h): - findings.append( - f"{path}:{lineno}: [hex-color] #{m.group(1)} in: {line.strip()[:120]}") - - for m in USERS_PATH_RE.finditer(line): - findings.append( - f"{path}:{lineno}: [machine-path] {m.group(0)!r} in: {line.strip()[:120]}") - return findings - - -def main() -> None: - ap = argparse.ArgumentParser(description=__doc__, - formatter_class=argparse.RawDescriptionHelpFormatter) - ap.add_argument("paths", nargs="+", help="files or directories to scan") - ap.add_argument("--terms", default=",".join(DEFAULT_TERMS), - help="comma-separated brand terms (case-insensitive)") - ap.add_argument("--allow-extra", action="append", default=[], - help="extra allowed-context regex (repeatable)") - ap.add_argument("--hex-allow", action="append", default=[], - help="extra allowed hex value, RRGGBB without # (repeatable)") - args = ap.parse_args() - - terms = [t.strip() for t in args.terms.split(",") if t.strip()] - if not terms: - die("error: --terms is empty") - try: - term_res = [re.compile(re.escape(t), re.IGNORECASE) for t in terms] - contexts = [re.compile(c, re.IGNORECASE) - for c in DEFAULT_ALLOW_CONTEXTS + args.allow_extra] - except re.error as e: - die(f"error: bad regex: {e}") - hex_allow = DEFAULT_HEX_ALLOW | {h.lower().lstrip("#") for h in args.hex_allow} - - findings: list[str] = [] - files = iter_files([Path(p) for p in args.paths]) - for f in files: - findings.extend(scan_file(f, term_res, contexts, hex_allow)) - - for line in findings: - print(line) - if findings: - print(f"\n{len(findings)} genericity finding(s) across {len(files)} file(s). " - "Nothing user-, brand-, or show-specific may ship; fix or allowlist with a reason.") - raise SystemExit(1) - print(f"clean: {len(files)} file(s) passed the genericity gate") - - -if __name__ == "__main__": - main() diff --git a/skills/mc-setup/scripts/tests/test-lint_genericity.py b/skills/mc-setup/scripts/tests/test-lint_genericity.py deleted file mode 100644 index 146e508..0000000 --- a/skills/mc-setup/scripts/tests/test-lint_genericity.py +++ /dev/null @@ -1,149 +0,0 @@ -#!/usr/bin/env python3 -# /// script -# requires-python = ">=3.11" -# /// -"""Tests for lint_genericity.py: brand-term detection with the ecosystem -allowlist, hex-color policy (token files, grayscale, placeholder palette, -tests dirs, svg), machine-path detection, section allowances in README.md, -and the documented exit codes (0 clean, 1 findings, 2 usage error).""" -import importlib.util -import subprocess -import sys -import tempfile -import unittest -from pathlib import Path - -SCRIPT = Path(__file__).resolve().parent.parent / "lint_genericity.py" - -spec = importlib.util.spec_from_file_location("lint_genericity", SCRIPT) -lint = importlib.util.module_from_spec(spec) -spec.loader.exec_module(lint) - - -def write(tmp: str, rel: str, text: str) -> Path: - p = Path(tmp) / rel - p.parent.mkdir(parents=True, exist_ok=True) - p.write_text(text) - return p - - -def run(args: list[str]) -> subprocess.CompletedProcess: - return subprocess.run([sys.executable, str(SCRIPT), *args], capture_output=True, text=True) - - -class TestBrandTerms(unittest.TestCase): - def test_bare_term_is_flagged(self): - with tempfile.TemporaryDirectory() as tmp: - f = write(tmp, "doc.md", "Ask pinkyd about the palette.\n") - r = run([str(f)]) - self.assertEqual(r.returncode, 1, r.stdout + r.stderr) - self.assertIn("[brand-term]", r.stdout) - self.assertIn("pinkyd", r.stdout) - - def test_case_insensitive(self): - with tempfile.TemporaryDirectory() as tmp: - f = write(tmp, "doc.md", "Madison's studio config.\n") - r = run([str(f)]) - self.assertEqual(r.returncode, 1, r.stdout + r.stderr) - - def test_ecosystem_usages_allowed(self): - allowed = ( - "Install with `npx bmad-method install`.\n" - "BMad Manticore is a BMad Method module.\n" - "Config lives in `{project-root}/_bmad/custom/config.toml`.\n" - "See https://github.com/bmad-code-org/bmad-manticore for source.\n" - "The project is not BMad-initialized.\n" - "It reads the installed BMad core scripts.\n" - ) - with tempfile.TemporaryDirectory() as tmp: - f = write(tmp, "doc.md", allowed) - r = run([str(f)]) - self.assertEqual(r.returncode, 0, r.stdout + r.stderr) - - def test_readme_ecosystem_section_allowed_but_other_sections_flagged(self): - text = ( - "# Title\n\nbmadcode leaked here.\n\n" - "## Part of the BMad ecosystem\n\nFollow https://x.com/BMadCode\n" - ) - with tempfile.TemporaryDirectory() as tmp: - f = write(tmp, "README.md", text) - r = run([str(f)]) - self.assertEqual(r.returncode, 1, r.stdout + r.stderr) - self.assertIn(":3:", r.stdout) - self.assertNotIn(":7:", r.stdout) - - def test_section_allowance_only_in_readme_or_agents(self): - text = "## Support BMad\n\nFollow bmadcode everywhere.\n" - with tempfile.TemporaryDirectory() as tmp: - f = write(tmp, "guide.md", text) - r = run([str(f)]) - self.assertEqual(r.returncode, 1, r.stdout + r.stderr) - - -class TestHexColors(unittest.TestCase): - def test_chromatic_hex_flagged(self): - with tempfile.TemporaryDirectory() as tmp: - f = write(tmp, "style.md", "Use #ba2f8c for callouts.\n") - r = run([str(f)]) - self.assertEqual(r.returncode, 1, r.stdout + r.stderr) - self.assertIn("[hex-color]", r.stdout) - - def test_grayscale_and_placeholder_palette_allowed(self): - with tempfile.TemporaryDirectory() as tmp: - f = write(tmp, "style.md", "Neutral #111111, #ffffff, accent #4f8cff.\n") - r = run([str(f)]) - self.assertEqual(r.returncode, 0, r.stdout + r.stderr) - - def test_token_files_tests_dirs_and_svg_exempt(self): - with tempfile.TemporaryDirectory() as tmp: - write(tmp, "tokens.template.json", '{"accent": "#ba2f8c"}\n') - write(tmp, "scripts/tests/test-x.py", 'FIXTURE = "#ba2f8c"\n') - write(tmp, "diagram.svg", '\n') - r = run([tmp]) - self.assertEqual(r.returncode, 0, r.stdout + r.stderr) - - def test_hex_allow_flag_extends_allowlist(self): - with tempfile.TemporaryDirectory() as tmp: - f = write(tmp, "style.md", "Use #ba2f8c here.\n") - r = run([str(f), "--hex-allow", "ba2f8c"]) - self.assertEqual(r.returncode, 0, r.stdout + r.stderr) - - -class TestMachinePaths(unittest.TestCase): - def test_users_path_flagged_even_in_svg_and_tests(self): - with tempfile.TemporaryDirectory() as tmp: - write(tmp, "scripts/tests/test-x.py", 'p = "/Users/someone/footage.mov"\n') - r = run([tmp]) - self.assertEqual(r.returncode, 1, r.stdout + r.stderr) - self.assertIn("[machine-path]", r.stdout) - - -class TestCli(unittest.TestCase): - def test_clean_tree_exits_0(self): - with tempfile.TemporaryDirectory() as tmp: - write(tmp, "doc.md", "Plain generic module content.\n") - r = run([tmp]) - self.assertEqual(r.returncode, 0, r.stdout + r.stderr) - self.assertIn("clean", r.stdout) - - def test_missing_path_exits_2(self): - r = run(["/nonexistent/path-for-lint-test"]) - self.assertEqual(r.returncode, 2, r.stdout + r.stderr) - - def test_skips_own_source_its_tests_and_binaries(self): - with tempfile.TemporaryDirectory() as tmp: - write(tmp, "lint_genericity.py", 'TERMS = ["bmadcode", "pinkyd"]\n') - write(tmp, "tests/test-lint_genericity.py", 'FIXTURE = "pinkyd"\n') - write(tmp, "clip.mov", "binary-ish pinkyd content\n") - r = run([tmp]) - self.assertEqual(r.returncode, 0, r.stdout + r.stderr) - - def test_custom_terms(self): - with tempfile.TemporaryDirectory() as tmp: - f = write(tmp, "doc.md", "myshowname appears here.\n") - r = run([str(f), "--terms", "myshowname"]) - self.assertEqual(r.returncode, 1, r.stdout + r.stderr) - - -if __name__ == "__main__": - unittest.main() diff --git a/skills/mc-setup/scripts/tests/test-merge_profile_frontmatter.py b/skills/mc-setup/scripts/tests/test-merge_profile_frontmatter.py index 3e5c18d..8cfa456 100644 --- a/skills/mc-setup/scripts/tests/test-merge_profile_frontmatter.py +++ b/skills/mc-setup/scripts/tests/test-merge_profile_frontmatter.py @@ -35,7 +35,7 @@ STUDIO = """--- format: talking-head stages: [new, braindump, outline, script, record, cut, beats, assets, graphics, package, final, retro] -engine_overlays: ograf +engine_overlays: html generated_broll: banned --- @@ -76,7 +76,7 @@ def test_missing_keys_merged_existing_and_body_untouched(self): merged = self.studio.read_text() self.assertIn("beat-types: [popup, diagram, lower-third, stat-card, cta]", merged) self.assertIn('medium: "20-45s"', merged) - self.assertIn("engine_overlays: ograf", merged) # studio value wins + self.assertIn("engine_overlays: html", merged) # studio value wins self.assertIn("generated_broll: banned", merged) # studio value wins self.assertNotIn("Shipped prose", merged) self.assertIn("the creator's hard-won learning stays put", merged) diff --git a/skills/mc-stream-pack/SKILL.md b/skills/mc-stream-pack/SKILL.md index 3953baa..faa4465 100644 --- a/skills/mc-stream-pack/SKILL.md +++ b/skills/mc-stream-pack/SKILL.md @@ -1,21 +1,48 @@ --- name: mc-stream-pack -description: Produce a complete branded livestream asset pack for OBS (scenes, stinger, lower thirds) from brand tokens. Use with the livestream-pack format or when the creator asks for stream assets. +description: Build a branded livestream asset pack for OBS. Use on the livestream-pack format, or when the user says "stream assets", "OBS pack", or "scenes and stinger". --- # mc-stream-pack -Brand tokens in, complete pack out. Spec lives in the `livestream-pack` format profile; this skill executes it. +Brand tokens in, complete pack out. The `livestream-pack` format profile is the spec; this skill executes it. The outcome is a `graphics/` folder the creator loads into OBS and goes live from without you in the room, which is the bar: every asset verified, and HANDOFF.md saying where each one goes. -## Steps +## Resolution rules -1. Load the studio config (`uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`; empty means mc-setup has not run: stop and route the creator there) and this skill's own surface (`uv run {project-root}/_bmad/scripts/resolve_customization.py --skill {skill-root}`; run `{workflow.activation_steps_prepend}` now, `{workflow.activation_steps_append}` after this step, and hold `{workflow.persistent_facts}` as standing context). Resolve `paths` values against `{project-root}`. Read `project.json` (stage `stream-pack`), the `livestream-pack` format profile, `{brand-path}/tokens.json`, and `{brand-path}/production-bible.md` when it exists (the styling contract beyond tokens: overlay and popup aesthetic for scenes and lower thirds, per-series template sections, and the CTA section). The OGraf standards apply to the lower thirds; the mc-ograf skill enforces them in step 2. -2. Build the pack the profile specifies (scene list, reactivity, render formats, and durations come from the profile, not from here): static scenes as self-contained local HTML in `graphics/scenes/`, all styling from tokens.json; the stinger as one HyperFrames comp in `{engines-path}/hyperframes/`, rendered to both formats the profile names; lower thirds and topic cards via the mc-ograf skill (never reach into its folder). Baked alpha deliverables headed for OBS browser or stinger use on any platform get a WebM VP9 alpha variant produced and verified in one step: `uv run {skill-root}/scripts/render_verify.py graphics/.mov --transcode-webm graphics/.webm` (checks default to yuva420p; add `--expect-res`/`--expect-fps`/`--expect-dur` from the profile). When `[live] tool` is vmix or other, know the targets: vMix rejects MP4 stingers and prefers PNG sequences, so deliver a PNG sequence (`ffmpeg -i .mov -pix_fmt rgba graphics/-png/%04d.png`) or the ProRes 4444 MOV instead of WebM; Wirecast takes the ProRes 4444 MOV directly. Sound for the pack (the stinger whoosh, a Starting Soon music bed) routes through the mc-audio service skill the same way; deliver the wavs alongside the scenes with OBS wiring noted in HANDOFF.md. This live lane does not require `[editor] ograf-editable`; OBS/SPX-GC is editor-independent. -3. Verify, not vibes: run the profile's verification section. Scene screenshots land in `graphics/_verify/` and every one is visually checked; stinger checks run via `uv run {skill-root}/scripts/render_verify.py`. Stinger and baked-asset WebM variants are verified with `--pixfmt yuva420p`, or produced and verified in one step via `--transcode-webm` as in step 2. -4. Write `graphics/HANDOFF.md`: OBS setup steps per asset (browser source URLs/sizes, stinger transition settings). Update project.json artifacts and advance stage per the profile's stages list (next after `stream-pack`, normally `final`): the creator loads the pack in OBS and approves the look live. +- Bare paths resolve against `{video-path}`, the current video project at `{projects-path}//`. +- `{skill-root}` → this skill's installed directory; files in it always carry it (`{skill-root}/scripts/render_verify.py`). +- `{project-root}` → the project working directory. + +## On Activation + +1. Load the studio config: `uv run {project-root}/_bmad/scripts/resolve_config.py --project-root {project-root} --key modules.manticore`. Empty means mc-setup has not run: stop and route the creator there. Resolve `paths` values against `{project-root}`. +2. Read `project.json` (stage `stream-pack`) and the `livestream-pack` format profile. +3. Read `{brand-path}/tokens.json`. If it does not exist, tell the creator it is missing and that styling the pack cannot happen without it, then route to mc-setup and stop. +4. Read `{brand-path}/production-bible.md`. If it does not exist, tell the creator it is missing and that the styling contract beyond tokens (the overlay and popup aesthetic for scenes and lower thirds, the per-series template sections, and the CTA section) cannot happen without it, then route to mc-setup and stop. + +## Build the pack + +The profile owns the scene list, reactivity, render formats, and durations. Scenes, lower thirds, and topic cards are self-contained local HTML in `graphics/scenes/`, styled entirely from tokens.json. The stinger is one HyperFrames comp in `{engines-path}/hyperframes/`, rendered to both formats the profile names. + +Baked alpha deliverables headed for OBS browser or stinger use on any platform get their WebM VP9 alpha variant produced and verified in one step: + +`uv run {skill-root}/scripts/render_verify.py graphics/.mov --transcode-webm graphics/.webm` + +Checks default to yuva420p; add `--expect-res`/`--expect-fps`/`--expect-dur` from the profile. + +When `[live] tool` is vmix or other, skip WebM: vMix rejects MP4 stingers and prefers PNG sequences, so deliver a sequence (`ffmpeg -i .mov -pix_fmt rgba graphics/-png/%04d.png`) or the ProRes 4444 MOV. Wirecast takes the ProRes 4444 MOV directly. + +Sound for the pack (the stinger whoosh, a Starting Soon music bed) routes through the mc-audio service skill; deliver the wavs alongside the scenes with their OBS wiring noted in HANDOFF.md. + +## Verify + +Run the profile's verification section. Scene screenshots land in `graphics/_verify/`. Stinger renders check via `uv run {skill-root}/scripts/render_verify.py`; the stinger and baked-asset WebM variants add `--pixfmt yuva420p`, or arrive already verified from the `--transcode-webm` call above. + +## Hand off + +Write `graphics/HANDOFF.md`: per asset, the OBS setup steps (browser source URLs and sizes, stinger transition settings). Then update project.json artifacts and advance stage per the profile's stages list (next after `stream-pack`, normally `final`), where the creator loads the pack in OBS and approves the look live. ## Checklist - Every scene screenshot visually checked; no scene ships unseen. -- Countdown actually resets on scene re-activation (test via the obsstudio event or document it as OBS-only behavior). -- No NodeCG, no alert plumbing in v1. +- Countdown actually resets on scene re-activation (test via the obsstudio event, or document it as OBS-only behavior). diff --git a/skills/mc-stream-pack/customize.toml b/skills/mc-stream-pack/customize.toml deleted file mode 100644 index 0e54866..0000000 --- a/skills/mc-stream-pack/customize.toml +++ /dev/null @@ -1,20 +0,0 @@ -# DO NOT EDIT -- overwritten on every update. -# -# Customization surface for mc-stream-pack. -# Override files (not edited here): -# {project-root}/_bmad/custom/mc-stream-pack.toml (team) -# {project-root}/_bmad/custom/mc-stream-pack.user.toml (personal) -# -# Studio-wide config (owner, paths, editor, transcription, assets, tools) -# lives in [modules.manticore] in {project-root}/_bmad/custom/config.toml, -# maintained by mc-setup. This file holds only this skill's own surface. - -[workflow] - -# Steps to run before / after the standard activation. -activation_steps_prepend = [] -activation_steps_append = [] - -# Persistent facts held for the whole run: literal sentences, or -# "file:{project-root}/..." paths whose contents are loaded as facts. -persistent_facts = [] diff --git a/skills/module-help.csv b/skills/module-help.csv index 041c665..53b73ee 100644 --- a/skills/module-help.csv +++ b/skills/module-help.csv @@ -7,12 +7,11 @@ BMad Manticore,mc-new,New Project,NP,"Scaffold a project from a format profile: BMad Manticore,mc-braindump,Braindump,BD,"Interview that captures the idea in the creator's exact words, verbatim. Camera-rolling mode turns the interview itself into usable footage.",,,1-write,mc-new,mc-outline,false,projects-path,*/braindump* BMad Manticore,mc-outline,Outline,OL,"3 hooks plus one outline plus the packaging promise. GATE 1: hard stop for creator approval. Footage-first projects skip the write phase entirely.",,,1-write,mc-braindump,mc-script,false,projects-path,*/outline.md BMad Manticore,mc-script,Script,SC,"Weave the script from the creator's own braindump words, lint against the blacklist, craft QA. The creator records it however they always record.",,,1-write,mc-outline,mc-cut,false,projects-path,*/script.md -BMad Manticore,mc-cut,Cut,CT,"Word-level transcript, cut plan with taste calls. GATE 2: hard stop. Every approval produces a preview render and the editor timeline export; the final-quality render is offered at gate 4.",,,2-cut,mc-script,mc-beats,true,projects-path,*/cut/edl.json +BMad Manticore,mc-cut,Cut,CT,"Verified word-level transcript, audio-based cut with dead-air tightening, and an editorial pass that recommends content cuts. GATE 2: hard stop on both the mechanical and content calls. Every approval produces a validated preview render and the editor timeline export; the final-quality render is offered at gate 4.",,,2-cut,mc-script,mc-beats,true,projects-path,*/cut/edl.json BMad Manticore,mc-beats,Graphics Beats,BT,"Riff treatment ideas with the creator, then the beat table anchored to spoken words under creativity mandates and the density tier, with the CTA placement pass. GATE 3: hard stop.",,,3-graphics,mc-cut,mc-graphics,false,projects-path,*/beats/beats.md BMad Manticore,mc-graphics,Build Graphics,GX,"Execute the approved beat table in HyperFrames / HTML / design-prompting; frame-verified alpha overlays plus an editor HANDOFF.",,,3-graphics,mc-beats,mc-assets,false,projects-path,*/graphics/* BMad Manticore,mc-assets,Farm Assets,FA,"Source and farm the stills and b-roll the beat table calls for through registered CLI tools (metered APIs opt-in), real verified imagery first.",,,3-graphics,mc-beats,mc-package,false,projects-path,*/assets/manifest.json BMad Manticore,mc-audio,Farm Sound,AU,"Service skill, no stage or gate: local-first TTS narration and two-host dialogue (Kokoro-82M), instrumental beds (MusicGen-small), SFX (AudioLDM2). Called from graphics, stream packs, and voiceover narration, or directly.",,,anytime,,,false,,*/manifest.json -BMad Manticore,mc-ograf,OGraf Graphics,OG,"Service skill: editable broadcast graphics where the target supports them (DaVinci Resolve 21+ editor lane, OBS/SPX-GC live lane). Everyone else gets baked alpha.",,,anytime,,,false,, BMad Manticore,mc-package,Package,PK,"Titles, thumbnails verified at 120px, description, CTAs, dual-timeline chapters, SRT/VTT captions and a publishable transcript from the edited timeline, series A/B pairs, live-event mode. May start any time after gate 1; offer it during dead time between stages.",,,4-package,mc-outline,mc-retro,false,projects-path,*/packaging/* BMad Manticore,mc-stream-pack,Stream Pack,LS,"A complete branded livestream asset pack for OBS (scenes, stinger, lower thirds) from brand tokens; the livestream-pack format lane.",,,anytime,,,false,projects-path,*/graphics/scenes/* BMad Manticore,mc-retro,Retro,RT,"After publishing: one round of notes edits the format profile, the bibles, and the brand files so the next video starts smarter, then the post-publish wrap.",,,5-wrap,mc-package,,false,brand-path,production-bible.md diff --git a/skills/module.yaml b/skills/module.yaml index ddfac2b..39930e9 100644 --- a/skills/module.yaml +++ b/skills/module.yaml @@ -7,7 +7,7 @@ default_selected: false # Agent roster, essence only. External skills (help catalog, party-mode) # read these descriptors to route and display agents. Full persona and -# behavior live in the skill's customize.toml. +# behavior live in the skill's SKILL.md. agents: - code: mc-agent name: Manny