From 364325c7919cef646253e776f08c10ef929f77bc Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Sat, 26 Sep 2026 23:11:06 +0000 Subject: [PATCH 01/17] Ground image-generate prompts in documentation and fail unaligned artwork. Image elements now share the same fail-closed contract as subject-beat coverage: scene-spec-generate feeds source snippets into the LLM, image prompts must use documented terms, and image-generate wraps the Images API call with narration/source plus a validate gate. Co-authored-by: jmjava --- AGENTS.md | 7 +- README.md | 27 +++++-- src/docgen/cli.py | 3 + src/docgen/config.py | 44 +++++++++++ src/docgen/image_generate.py | 115 +++++++++++++++++++++++++++-- src/docgen/init.py | 1 + src/docgen/scene_spec.py | 58 +++++++++++++++ src/docgen/scene_spec_generate.py | 85 ++++++++++++++++++--- src/docgen/validate.py | 63 +++++++++++++++- src/docgen/yaml_generate.py | 1 + tests/test_config.py | 27 ++++++- tests/test_image_generate.py | 102 +++++++++++++++++++++++++ tests/test_scene_spec.py | 34 +++++++++ tests/test_scene_spec_generate.py | 54 ++++++++++++++ tests/test_validate.py | 97 +++++++++++++++++++++++- tests/test_validate_timing_sync.py | 7 +- tests/test_yaml_generate.py | 1 + 17 files changed, 691 insertions(+), 35 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 4fa24aa..e14875f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -70,13 +70,13 @@ Commands registered on the **`docgen`** CLI include: - **`freeze`** — ``docgen freeze`` builds the **`docgen-gui`** onedir (`pip install 'docgen[packaging]'`). Optional ``--smoke`` runs the binary headless. Do not run a full freeze in routine pytest; set ``DOCGEN_FREEZE_SMOKE=1`` for the optional test. - **`tts`** — text-to-speech for segment files (OpenAI or xAI `/v1/tts`). - **`timestamps`** — word/segment timing (`timing.json`). Default engine **`local`** aligns the known narration text against the mp3 offline (ffmpeg silencedetect, no API); **`--engine whisper`** uses OpenAI whisper-1 or xAI `/v1/stt` when `ai.provider` is grok. Both emit the same Whisper-shaped blocks. Failed ffmpeg silencedetect raises `AlignmentError` (empty stderr is not treated as full-span speech). OpenAI whisper-1 word/segment `start`/`end` must be finite JSON numbers (bool/NaN raise `AIError`). Grok `/v1/stt` rejects empty word tokens and inverted `end < start` intervals (`AIError`). Empty ``segments.all`` raises ``TimestampError`` (same as TTS) and does not leave a stale ``timing.json`` as success. -- **`image-generate`** — render scene-spec **image elements** (`image:` + `prompt:` boxes) via OpenAI Images or xAI Imagine into the bundle (also runs for missing assets inside `generate-all`). +- **`image-generate`** — render scene-spec **image elements** (`image:` + `prompt:` boxes) via OpenAI Images or xAI Imagine into the bundle (also runs for missing assets inside `generate-all`). Default **`align_with_docs`** wraps each prompt with narration + source snippets and fails when the authored prompt shares no documented terms. - **`manim`** — render Manim scenes declared in config. - **`compose`** — mux narration audio with visual sources via ffmpeg. With no segment ids, uses ``segments.all`` (same as ``generate-all``), not ``segments.default``. A ``type: mixed`` row raises ``ComposeError`` if any listed source is missing (no silent subset mux). -- **`validate`** / **`validate --pre-push`** — drift, narration lint, Manim hints, **`timing_sync`**, **`story_end`** (last paced reveal vs audio end; hard fail), **`scene_assets`** (pre-render: stuck-board cadence, frame-budget overlaps, `MANIM_FONT` consistency, stale helpers / stale compiled class including hand-edited generated-region labels and `run_time` — hard fail; also a `generate-all` gate before Manim), **`av_sync`** (hard fail on `--pre-push` / `generate-all`; prefers scene-spec labels as OCR anchors), **`subject_beat_coverage`** (declarative specs vs narration topic beats; hard fail when enabled), and related visual-sync checks (`ocr_scan`, `layout`, `freeze_ratio` — hard fail on `--pre-push` / `generate-all`). Missing tesseract fails `ocr_scan` / `av_sync` / `layout` (not skip-PASS). Missing audio or an LFS pointer fails `timing_sync` and recording media gates (`stream_presence`, `av_drift`, `ocr_scan`, `av_sync`) (not skip-PASS). Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` for `type: manim` (not skip-PASS). +- **`validate`** / **`validate --pre-push`** — drift, narration lint, Manim hints, **`timing_sync`**, **`story_end`** (last paced reveal vs audio end; hard fail), **`scene_assets`** (pre-render: stuck-board cadence, frame-budget overlaps, `MANIM_FONT` consistency, stale helpers / stale compiled class including hand-edited generated-region labels and `run_time` — hard fail; also a `generate-all` gate before Manim), **`av_sync`** (hard fail on `--pre-push` / `generate-all`; prefers scene-spec labels as OCR anchors), **`subject_beat_coverage`** (declarative specs vs narration topic beats; hard fail when enabled), **`image_prompt_alignment`** (image-element prompts must share documented terms with narration/source; hard fail when enabled), and related visual-sync checks (`ocr_scan`, `layout`, `freeze_ratio` — hard fail on `--pre-push` / `generate-all`). Missing tesseract fails `ocr_scan` / `av_sync` / `layout` (not skip-PASS). Missing audio or an LFS pointer fails `timing_sync` and recording media gates (`stream_presence`, `av_drift`, `ocr_scan`, `av_sync`) (not skip-PASS). Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` / `image_prompt_alignment` for `type: manim` (not skip-PASS). - **`lint`** — narration lint helper. - **`narration-generate`** — LLM-assisted narration from hints and repo context; optional **`--revise --revision-notes`** for in-place edits (same contract as the wizard Revise button). -- **`scene-spec-generate`** — LLM emits declarative **`*.scene.yaml`**; enforces frame budget + **subject-beat coverage** (dwell OK; cover topic shifts; reject invented labels). +- **`scene-spec-generate`** — LLM emits declarative **`*.scene.yaml`**; enforces frame budget + **subject-beat coverage** (dwell OK; cover topic shifts; reject invented labels) and **image-prompt alignment** (image `prompt:` must use documented terms from narration/source). - **`scene-compile`** — compile specs into **`scenes.py`** (generated regions only). - **`yaml-generate`** — merge defaults and hint wiring into **`docgen.yaml`**. - **`clean-bundle`** — remove regenerable outputs per policy. @@ -91,6 +91,7 @@ Commands registered on the **`docgen`** CLI include: - **Manim / `scenes.py` (marker blocks):** Fix generators under `src/docgen/**` (`manim_scene_support.py`, `scene_spec.py`, `scene_spec_generate.py`, `validate`, `yaml_generate`, tests). **Do not** patch generated classes inside a consumer's **`animations/scenes.py`** between **`BEGIN/END GENERATED SCENE`** markers; re-run **`scene-spec-generate`** / **`scene-compile --retime`** and **`manim`** instead. Preferred consumer order: narration → TTS → timestamps → scene-spec/compile → Manim → compose. - **Beat sync (fail-closed):** when `timing.json` has words, every story box label must match a spoken phrase (`wait_word`); unmatched labels and leftover LLM indices are rejected. Opt out with ``pace: none``. Legacy row-level ``wait_segment`` is upgraded to ``wait_word`` and written back on ``scene-compile`` (direct ``compile_scene_class`` still rejects leftover ``wait_segment``). Fuzzy containment matching is not used. **`scene-compile` clamps FadeIn / page-fade `run_time` against the next word start** so `_TimedScene._clock` cannot race past waits (issue #66 — do not emit cascading first-board dumps). After a reveal, a **dwell** slot may play `Indicate` / `Circumscribe` when the gap to the next `wait_word` is long enough (also clamped). Long holds emit additional mid-hold pulses (`timed_wait` + emphasis) so the board does not freeze after the first Indicate. Optional box fields: `shape` (rounded/pill/diamond), `reveal` (fade/grow/slide), `emphasis` (none/pulse/ring). Page transitions FadeOut revealed boxes, not the parent `VGroup`. `scene-compile` refreshes stale `_box` / `_arrow` / `_TimedScene` helpers in `scenes.py`. - **Subject-beat coverage:** implemented in `scene_spec.layout_density_violations` / `cluster_subject_beats`; enforced by **`scene-spec-generate`** and **`validate`** (`validation.subject_beat_coverage.enabled`, default true). Not a blind label count. +- **Image-prompt alignment:** implemented in `scene_spec.image_prompt_alignment_violations`; **`scene-spec-generate`** and **`image-generate`** reject prompts that share no documented terms; **`validate`** (`validation.image_prompt_alignment.enabled`, default true) is the same gate. `image-generate` also wraps the Images API prompt with narration/source (`image_generation.align_with_docs`, default true). - **`docgen benchmark` is required** after clock / compile / `_TimedScene` / dwell changes. Pytest string assertions are not a substitute. Do not remove the CI `benchmark` job. See **Required gate** below. - Prefer **stable CLI / library contracts** and **documented exit codes** so CI can depend on them. - **`narration_from_source`:** hints in config + **`docgen narration-generate`** — owner-supplied context paths, not opaque bulk edits to outputs. diff --git a/README.md b/README.md index aec8a80..e6802e1 100644 --- a/README.md +++ b/README.md @@ -55,21 +55,26 @@ If you still need the legacy behaviour, pin a pre-removal commit - **Image assets in Manim scenes** — a scene-spec box may be an **image element** (`image: images/.png` + `prompt:`); `docgen image-generate` renders the prompt via OpenAI Images (default `gpt-image-1`) or xAI Imagine - (`grok-imagine-image-2.0` when `ai.provider` is `grok`). The compiled scene - shows it with the `_image` helper. `generate-all` fills in missing assets - automatically. + (`grok-imagine-image-2.0` when `ai.provider` is `grok`). By default the + authored prompt is **grounded** in the segment narration plus + `manim_scene_generation` source snippets, and prompts that share no + documented terms fail closed (`image_prompt_alignment`). The compiled scene + shows the asset with the `_image` helper. `generate-all` fills in missing + assets automatically. - **ffmpeg composition** — combine narration audio and Manim video into final segments, with a freeze-tail guard. - **Validation** — A/V drift, freeze ratio, OCR error scan, layout, narration lint, Manim scene lint, **timing_sync** (stale `timing.json` vs regenerated mp3 — hard fail), **story_end** (paced visual story finishes long before narration — - hard fail), and **av_sync** (OCR check that scene-spec label anchors appear on - screen near their spoken time — hard fail on `--pre-push` / `generate-all`). + hard fail), **image_prompt_alignment** (image-element `prompt:` must share + documented terms with narration/source — hard fail), and **av_sync** (OCR + check that scene-spec label anchors appear on screen near their spoken time + — hard fail on `--pre-push` / `generate-all`). Missing tesseract fails `ocr_scan` / `av_sync` / `layout` instead of skip-PASS. Missing audio or an LFS pointer fails `timing_sync` and recording media gates (`stream_presence`, `av_drift`, `ocr_scan`, `av_sync`) instead of skip-PASS. - Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` for - `type: manim` instead of skip-PASS. A hand-edited generated-region label + Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` / + `image_prompt_alignment` for `type: manim` instead of skip-PASS. A hand-edited generated-region label or `run_time` in `scenes.py` fails `scene_assets` compile_sync (clock / benchmark still execute a fresh `compile_scene_class`, not the on-disk file). - **GitHub Pages** — auto-generate `index.html`, deploy workflow, LFS rules, @@ -128,7 +133,9 @@ using the Cursor key: image_generation: model: gpt-image-1 # or dall-e-3, gpt-image-1-mini, … size: 1536x1024 + align_with_docs: true # wrap prompts with narration/source; fail invented terms # quality: high + # style: "…" # optional override of the educational-diagram prefix ``` ```bash @@ -241,7 +248,7 @@ docgen --repo /path/to/your-project generate-all | `docgen freeze [--dist DIR] [--smoke]` | PyInstaller onedir for **`docgen-gui` only** (`pip install 'docgen[packaging]'`). Not the full Manim CLI | | `docgen tts [--segment 01] [--dry-run]` | Generate TTS audio | | `docgen timestamps [--engine local\|whisper]` | Extract word/segment timestamps from TTS audio → `timing.json` (default `local`: offline narration-text alignment; `whisper`: OpenAI transcription). Empty `segments.all` is `TimestampError` (stale `timing.json` is not success) | -| `docgen image-generate [--segment 01 \| --all \| --spec PATH] [--force] [--dry-run] [--model …] [--size …]` | Generate scene-spec image assets (`image:` + `prompt:` boxes) via the OpenAI Images API into the bundle | +| `docgen image-generate [--segment 01 \| --all \| --spec PATH] [--force] [--dry-run] [--model …] [--size …]` | Generate scene-spec image assets (`image:` + `prompt:` boxes) via the OpenAI Images API; grounds prompts in narration/source and fails unaligned prompts unless `image_generation.align_with_docs` is false | | `docgen manim [--scene StackDAGScene]` | Render Manim animations | | `docgen compose [01 02 03] [--ffmpeg-timeout 900]` | Compose segments (audio + video). Omit ids to walk `segments.all` (same as `generate-all`); a mapped segment with missing audio/visuals is a hard fail | | `docgen validate [--max-drift 2.75] [--pre-push]` | Run all validation checks | @@ -344,7 +351,9 @@ timestamps: image_generation: # scene-spec image elements (docgen image-generate) model: gpt-image-1 # Cursor/OpenAI Images; Grok remaps gpt-image-* to Imagine size: 1536x1024 + align_with_docs: true # ground prompts in narration/source (default) # quality: high # optional, model-specific + # style: "…" # optional Images-API prefix override manim: quality: 1080p30 # supports 480p15, 720p30, 1080p30, 1080p60, 1440p30, 1440p60, 2160p60 @@ -355,6 +364,8 @@ manim: validation: subject_beat_coverage: enabled: true # scene-spec-generate + validate: cover narration topic beats + image_prompt_alignment: + enabled: true # image-element prompts must use documented terms compose: ffmpeg_timeout_sec: 300 # can also be overridden with: docgen compose --ffmpeg-timeout N diff --git a/src/docgen/cli.py b/src/docgen/cli.py index 9501b4c..49c075b 100644 --- a/src/docgen/cli.py +++ b/src/docgen/cli.py @@ -1142,6 +1142,9 @@ def image_generate_cmd( if res.status == "dry-run": click.echo(f"[image-generate] {target.name}: would generate {res.relpath}") click.echo(f" prompt: {res.prompt}") + if res.effective_prompt and res.effective_prompt != res.prompt: + click.echo(" aligned prompt:") + click.echo(res.effective_prompt) elif res.status == "exists": click.echo(f"[image-generate] {target.name}: {res.relpath} exists (skip; use --force)") else: diff --git a/src/docgen/config.py b/src/docgen/config.py index 9f37b47..184cd44 100644 --- a/src/docgen/config.py +++ b/src/docgen/config.py @@ -331,6 +331,7 @@ def __post_init__(self) -> None: "story_end", "narration_lint", "subject_beat_coverage", + "image_prompt_alignment", ): self._sub_block(validation, nested, label=f"validation.{nested}") # List-valued keys: a string must not be iterated as characters. @@ -485,6 +486,14 @@ def __post_init__(self) -> None: require_yaml_string(ig["size"], label="image_generation.size", source=src) if ig.get("quality") is not None: require_yaml_string(ig["quality"], label="image_generation.quality", source=src) + if ig.get("align_with_docs") is not None: + require_yaml_bool( + ig["align_with_docs"], + label="image_generation.align_with_docs", + source=src, + ) + if ig.get("style") is not None: + require_yaml_string(ig["style"], label="image_generation.style", source=src) ai = self._block("ai") if ai.get("provider") is not None: require_yaml_string(ai["provider"], label="ai.provider", source=src) @@ -766,6 +775,17 @@ def __post_init__(self) -> None: label="validation.subject_beat_coverage.enabled", source=src, ) + ipa = self._sub_block( + validation, + "image_prompt_alignment", + label="validation.image_prompt_alignment", + ) + if ipa.get("enabled") is not None: + require_yaml_bool( + ipa["enabled"], + label="validation.image_prompt_alignment.enabled", + source=src, + ) def _source_label(self) -> str: return self.yaml_path.name if self.yaml_path else "docgen.yaml" @@ -950,10 +970,34 @@ def image_generation_config(self) -> dict[str, Any]: defaults: dict[str, Any] = { "model": "gpt-image-1", "size": "1536x1024", + "align_with_docs": True, } defaults.update(self._block("image_generation")) return defaults + @property + def image_align_with_docs(self) -> bool: + """When true (default), ground image prompts in narration/source and fail unaligned ones.""" + block = self._block("image_generation") + if "align_with_docs" in block: + return bool(block.get("align_with_docs")) + return True + + @property + def image_prompt_alignment_enabled(self) -> bool: + """When true (default), validate + scene-spec-generate enforce image-prompt grounding. + + Config: ``validation.image_prompt_alignment.enabled`` (bool). + """ + block = self._sub_block( + self._block("validation"), + "image_prompt_alignment", + label="validation.image_prompt_alignment", + ) + if "enabled" in block: + return bool(block.get("enabled")) + return True + # -- Manim ----------------------------------------------------------------- @property diff --git a/src/docgen/image_generate.py b/src/docgen/image_generate.py index b41722b..b78a627 100644 --- a/src/docgen/image_generate.py +++ b/src/docgen/image_generate.py @@ -11,7 +11,11 @@ ``docgen image-generate`` scans specs, calls the Images API (OpenAI or xAI Imagine) for elements whose asset is missing (or ``--force``), and writes PNG bytes to -``/``. ``docgen manim`` then loads the asset via the +``/``. By default the authored ``prompt`` is **grounded** in +the segment narration plus ``manim_scene_generation`` source snippets so the +image model sees the documentation, not only a short caption. Prompts that +share no documented terms fail closed (same check as ``validate`` / +``image_prompt_alignment``). ``docgen manim`` then loads the asset via the ``_image`` helper in ``scenes.py``. """ @@ -20,16 +24,28 @@ import base64 from dataclasses import dataclass from pathlib import Path -from typing import TYPE_CHECKING, Callable +from typing import TYPE_CHECKING, Any, Callable from docgen.openai_retry import call_with_rate_limit_retries -from docgen.scene_spec import iter_image_elements, load_scene_spec +from docgen.scene_spec import ( + image_prompt_alignment_violations, + iter_image_elements, + load_scene_spec, +) if TYPE_CHECKING: from docgen.config import Config DEFAULT_IMAGE_MODEL = "gpt-image-1" DEFAULT_IMAGE_SIZE = "1536x1024" +DEFAULT_IMAGE_STYLE = ( + "Clean educational diagram for a documentation video. Flat vector " + "illustration, high contrast, no watermark, no signature, no decorative " + "fake UI or invented product logos. Any readable text must be short ASCII " + "copied from the documented subject. Do not add components that are not " + "named in the documentation." +) +_ALIGN_CORPUS_EXCERPT = 2200 class ImageGenerationError(RuntimeError): @@ -42,6 +58,7 @@ class ImageAssetResult: path: Path status: str # "generated" | "exists" | "dry-run" prompt: str + effective_prompt: str = "" def generate_image_bytes( @@ -128,6 +145,69 @@ def generate_image_bytes( ) +def _excerpt(text: str, limit: int = _ALIGN_CORPUS_EXCERPT) -> str: + collapsed = " ".join((text or "").split()) + if len(collapsed) <= limit: + return collapsed + cut = collapsed[: limit - 1] + if " " in cut: + cut = cut.rsplit(" ", 1)[0] + return cut + "…" + + +def build_aligned_image_prompt( + authored: str, + *, + corpus_text: str = "", + label: str = "", + style: str = DEFAULT_IMAGE_STYLE, +) -> str: + """Wrap an authored scene-spec prompt with documentation + style constraints.""" + parts: list[str] = [(style or "").strip() or DEFAULT_IMAGE_STYLE, ""] + corpus = (corpus_text or "").strip() + if corpus: + parts.append( + "Documented subject (use these terms; do not invent names, logos, " + "or extra components):" + ) + parts.append(_excerpt(corpus)) + parts.append("") + lab = (label or "").strip() + if lab: + parts.append(f"On-screen timing label (must remain accurate): {lab}") + parts.append("") + parts.append("Illustration request:") + parts.append((authored or "").strip()) + return "\n".join(parts).strip() + "\n" + + +def collect_alignment_corpus(cfg: "Config", spec: dict[str, Any]) -> str: + """Narration + scene-generation hints + source snippets for one spec.""" + parts: list[str] = [] + seg_id = str(spec.get("segment_id") or "").strip() + if seg_id: + found = cfg.find_segment_asset(cfg.narration_dir, seg_id, ".md") + if found is not None and found.is_file(): + try: + parts.append(found.read_text(encoding="utf-8")) + except OSError: + pass + from docgen.manim_scene_support import ( + collect_source_snippets, + merged_scene_generation_settings, + ) + + settings = merged_scene_generation_settings(cfg, seg_id) + for h in settings.hints: + if str(h).strip(): + parts.append(str(h).strip()) + for label, text in collect_source_snippets(cfg, settings, extra_paths=[]): + body = str(text or "").strip() + if body: + parts.append(f"{label}\n{body}") + return "\n\n".join(p for p in parts if str(p).strip()) + + def _resolve_asset_path(cfg: "Config", relpath: str) -> Path: p = Path(relpath) if p.is_absolute() or ".." in p.parts: @@ -163,15 +243,36 @@ def generate_images_for_spec( size = (size_override or "").strip() or str(icfg.get("size") or DEFAULT_IMAGE_SIZE) quality = icfg.get("quality") quality = str(quality).strip() if quality else None + align = bool(icfg.get("align_with_docs", True)) + style = str(icfg.get("style") or "").strip() or DEFAULT_IMAGE_STYLE + corpus = collect_alignment_corpus(cfg, spec) if align else "" + + if align: + issues = image_prompt_alignment_violations(spec, corpus_text=corpus) + if issues: + joined = "\n ".join(issues) + raise ImageGenerationError( + f"{spec_path}: image prompt alignment failed — rewrite each " + f"`prompt` so it uses documented terms from narration/source " + f"(or set image_generation.align_with_docs: false):\n {joined}" + ) results: list[ImageAssetResult] = [] for el in elements: rel = str(el["image"]).strip() prompt = str(el.get("prompt") or "").strip() out = _resolve_asset_path(cfg, rel) + label = str(el.get("label") or "").strip() + effective = ( + build_aligned_image_prompt( + prompt, corpus_text=corpus, label=label, style=style + ) + if align and prompt + else prompt + ) if out.is_file() and not force: - results.append(ImageAssetResult(rel, out, "exists", prompt)) + results.append(ImageAssetResult(rel, out, "exists", prompt, effective)) continue if not prompt: raise ImageGenerationError( @@ -179,7 +280,7 @@ def generate_images_for_spec( f"({out}); add the file to the bundle or set a prompt in the spec." ) if dry_run: - results.append(ImageAssetResult(rel, out, "dry-run", prompt)) + results.append(ImageAssetResult(rel, out, "dry-run", prompt, effective)) continue fn = image_fn or ( @@ -187,14 +288,14 @@ def generate_images_for_spec( prompt=p, model=model, size=size, quality=quality, cfg=cfg ) ) - data = fn(prompt) + data = fn(effective) if not data: raise ImageGenerationError( f"{spec_path}: image element {rel!r} — provider returned empty bytes" ) out.parent.mkdir(parents=True, exist_ok=True) out.write_bytes(data) - results.append(ImageAssetResult(rel, out, "generated", prompt)) + results.append(ImageAssetResult(rel, out, "generated", prompt, effective)) return results diff --git a/src/docgen/init.py b/src/docgen/init.py index 8f16208..2a9d583 100644 --- a/src/docgen/init.py +++ b/src/docgen/init.py @@ -338,6 +338,7 @@ def _write_config(plan: InitPlan) -> str: "image_generation": { "model": "gpt-image-1", # Cursor/OpenAI Images; override with --model "size": "1536x1024", + "align_with_docs": True, # ground prompts in narration/source; fail invented terms }, "compose": { "ffmpeg_timeout_sec": 300, diff --git a/src/docgen/scene_spec.py b/src/docgen/scene_spec.py index ed7d94c..4af804d 100644 --- a/src/docgen/scene_spec.py +++ b/src/docgen/scene_spec.py @@ -1339,6 +1339,64 @@ def content_tokens(text: str) -> set[str]: return {w for w in words if len(w) > 2 and w not in _CONTENT_STOPWORDS} +# Visual-style words an image prompt may use without being documented terms. +_IMAGE_STYLE_TOKENS = frozenset( + """ + diagram diagrams illustration illustrations icon icons flat clean vector + isometric cartoon watercolor sketch photo photographic realistic abstract + artwork image images picture pictures visual background foreground style + colored colour color palette high contrast simple minimal educational + documentary video watermark signature logo logos chrome banner poster + scene board rounded pill diamond arrow arrows flow flowchart infographic + thumbnail render rendering drawing draw depict showing shows show + depicting depicts labeled labelled label labels text white black blue + green orange red gray grey dark light bright soft hard wide tall small + large tiny big thin thick line lines box boxes node nodes panel panels + layout grid row rows column columns + """.split() +) + + +def image_prompt_alignment_violations( + spec: dict[str, Any], + *, + corpus_text: str, +) -> list[str]: + """Reject image prompts that share no documented terms with narration/source. + + Style adjectives (``flat``, ``diagram``, ``illustration``, …) are ignored. + A missing corpus (no narration and no source) is a no-op — there is nothing + to align against. Specs with no image elements, or image elements that omit + ``prompt`` (committed assets), also pass. + """ + elements = iter_image_elements(spec) + if not elements: + return [] + corpus = content_tokens(corpus_text) + if not corpus: + return [] + issues: list[str] = [] + for el in elements: + rel = str(el.get("image") or "").strip() or "(unnamed image)" + prompt = str(el.get("prompt") or "").strip() + if not prompt: + continue + toks = content_tokens(prompt) + substance = toks - _IMAGE_STYLE_TOKENS + if not substance: + issues.append( + f"{rel}: image prompt is only visual style (no documented subject terms)" + ) + continue + if not (substance & corpus): + preview = prompt if len(prompt) <= 80 else prompt[:77] + "..." + issues.append( + f"{rel}: image prompt shares no documented terms with " + f"narration/source: {preview!r}" + ) + return issues + + def cluster_subject_beats( sentences: list[str], *, diff --git a/src/docgen/scene_spec_generate.py b/src/docgen/scene_spec_generate.py index 693d07e..73b9410 100644 --- a/src/docgen/scene_spec_generate.py +++ b/src/docgen/scene_spec_generate.py @@ -36,6 +36,7 @@ sanitize_pacing_conflicts, compile_scene_class, cluster_subject_beats, + image_prompt_alignment_violations, layout_budget_violations, layout_density_violations, layout_stack_budget, @@ -90,10 +91,14 @@ may instead be an image element with: - image: bundle-relative asset path, e.g. ``images/.png`` (no absolute paths, no "..") - width / height: positive numbers (frame budget rules above apply; images count like boxes) - - prompt: string — a clear visual description; ``docgen image-generate`` renders it via the OpenAI Images API - - label: optional single word from the narration used as the timing anchor for the reveal + - prompt: string — a clear visual description grounded in the narration **and** SOURCE + DOCUMENTATION; ``docgen image-generate`` renders it and rejects prompts that invent + undocumented product names or share no documented terms + - label: optional spoken phrase from the narration used as the timing anchor for the reveal Image elements must NOT carry ``color`` or ``font_size``. Prefer labeled boxes for diagrams; use images -only for illustrative artwork the hints explicitly request. +only for illustrative artwork the hints explicitly request. Image ``prompt`` text must name +the same concepts as the narration/source (not generic "a diagram" / invented architecture). +Any words you expect to appear *inside* the artwork must be short ASCII copied from the docs. Optional per-box (**Whisper ``words`` only**); omit if unsure — compile fills from each box ``label`` → first transcript match: - wait_word: non-negative int — index into ``timing.json`` → ``words``; that box waits until that token's **start**, then fades in (**one box at a time** within each row). @@ -251,6 +256,22 @@ def build_scene_spec_user_message( if str(h).strip(): parts.append(f"- {str(h).strip()}") + if source_snippets: + parts.append("") + parts.append("--- SOURCE DOCUMENTATION ---") + parts.append( + "Use these excerpts when authoring box labels **and** image prompts. " + "Do not invent product names, APIs, or architecture that is not in this " + "source or the narration." + ) + for label, text in source_snippets: + body = str(text or "").strip() + if not body: + continue + parts.append("") + parts.append(f"### {label}") + parts.append(body) + if reference_scenes: parts.append("") parts.append( @@ -449,6 +470,7 @@ def _parse_and_harden_llm_spec( raw: str, enforce_density: bool, density_slack: int = 0, + corpus_text: str = "", ) -> dict[str, Any]: """Parse YAML, auto-layout, validate schema/budget/(optional) density, compile-lint.""" body = strip_yaml_fences(raw) @@ -499,6 +521,17 @@ def _parse_and_harden_llm_spec( f"segment {seg_id}: scene spec failed subject-beat coverage:\n {joined}\nDraft: {draft}" ) + if getattr(cfg, "image_prompt_alignment_enabled", True): + align_issues = image_prompt_alignment_violations( + merged_spec, corpus_text=corpus_text or narration_text + ) + if align_issues: + draft = _save_draft(cfg, seg_id, body) + joined = "\n ".join(align_issues) + raise SceneGenerationError( + f"segment {seg_id}: image prompt alignment failed:\n {joined}\nDraft: {draft}" + ) + try: _, _ = linted_class_block_from_spec(cfg, merged_spec, timing_key=seg_name) except SceneGenerationError as exc: @@ -536,6 +569,15 @@ def generate_scene_spec( existing = scenes_path.read_text(encoding="utf-8") if scenes_path.exists() else "" reference_scenes = extract_reference_classes(existing) snippets = collect_source_snippets(cfg, settings, extra_paths=extra_paths) + corpus_parts = [narration_text] + for h in list(settings.hints) + list(extra_hints): + if str(h).strip(): + corpus_parts.append(str(h).strip()) + for label, text in snippets: + body = str(text or "").strip() + if body: + corpus_parts.append(f"{label}\n{body}") + corpus_text = "\n\n".join(p for p in corpus_parts if str(p).strip()) system_prompt = scene_spec_system_prompt(cfg, seg_id) user_message = build_scene_spec_user_message( @@ -578,13 +620,23 @@ def generate_scene_spec( for attempt in range(3): msg = user_message if attempt > 0 and last_sparse is not None: - msg = ( - f"{user_message}\n\n--- RETRY: SUBJECT-BEAT COVERAGE FAILED ---\n" - f"{last_sparse}\n" - f"Cover each of the {n_beats} subject beats with a spoken-phrase label. " - "Hold the board across sentences in the same beat; add a new label only " - "when the topic shifts. Do not invent unspoken diagram terms." - ) + err = str(last_sparse) + if "image prompt alignment" in err: + msg = ( + f"{user_message}\n\n--- RETRY: IMAGE PROMPT ALIGNMENT FAILED ---\n" + f"{last_sparse}\n" + "Rewrite each image prompt so it uses documented terms from the " + "narration and SOURCE DOCUMENTATION. Do not invent product names " + "or generic 'a diagram' artwork with no subject." + ) + else: + msg = ( + f"{user_message}\n\n--- RETRY: SUBJECT-BEAT COVERAGE FAILED ---\n" + f"{last_sparse}\n" + f"Cover each of the {n_beats} subject beats with a spoken-phrase label. " + "Hold the board across sentences in the same beat; add a new label only " + "when the topic shifts. Do not invent unspoken diagram terms." + ) try: raw = invoke( system_prompt=system_prompt, @@ -612,12 +664,22 @@ def generate_scene_spec( raw=raw, enforce_density=True, density_slack=0, + corpus_text=corpus_text, ) last_sparse = None break except SceneGenerationError as exc: - if "subject-beat coverage" not in str(exc): + err = str(exc) + retryable = ( + "subject-beat coverage" in err or "image prompt alignment" in err + ) + if not retryable: raise + if "image prompt alignment" in err: + if attempt >= 2: + raise + last_sparse = exc + continue # Near-miss: accept without another LLM call when close enough. try: merged_spec = _parse_and_harden_llm_spec( @@ -630,6 +692,7 @@ def generate_scene_spec( raw=raw, enforce_density=True, density_slack=near_miss_slack, + corpus_text=corpus_text, ) last_sparse = None break diff --git a/src/docgen/validate.py b/src/docgen/validate.py index e79031b..ebb8519 100644 --- a/src/docgen/validate.py +++ b/src/docgen/validate.py @@ -362,6 +362,7 @@ def validate_segment( if self.config.visual_map.get(seg_id, {}).get("type") == "manim": report.checks.append(self._check_manim_scene_lint()) report.checks.append(self._check_subject_beat_coverage(seg_id)) + report.checks.append(self._check_image_prompt_alignment(seg_id)) report.checks.append(self._check_scene_assets(seg_id)) return report.to_dict() @@ -372,9 +373,9 @@ def run_pre_push(self) -> None: Missing recordings are reported as warnings, not failures — a project that hasn't generated videos yet should still be pushable. Quality checks on *existing* recordings — including visual-sync (``av_sync``, - ``subject_beat_coverage``, ``ocr_scan``, ``layout``, ``freeze_ratio``) - — and narration lint are hard failures so ``generate-all`` cannot - print ``Pipeline complete`` after a desynced mux. + ``subject_beat_coverage``, ``image_prompt_alignment``, ``ocr_scan``, + ``layout``, ``freeze_ratio``) — and narration lint are hard failures so + ``generate-all`` cannot print ``Pipeline complete`` after a desynced mux. """ reports = self.run_all() hard_fail = False @@ -646,6 +647,62 @@ def _check_subject_beat_coverage(self, seg_id: str) -> CheckResult: ["Subject beats covered; no invented unspoken labels"], ) + def _check_image_prompt_alignment(self, seg_id: str) -> CheckResult: + """Ensure scene-spec image prompts share documented terms (not generic art).""" + if not self.config.image_prompt_alignment_enabled: + return CheckResult( + "image_prompt_alignment", + True, + ["validation.image_prompt_alignment disabled in config (skipped)"], + ) + + seg_name = self.config.resolve_segment_name(seg_id) + spec_path = self.config.animations_dir / "specs" / f"{seg_name}.scene.yaml" + if not spec_path.is_file(): + return _missing_manim_spec_result("image_prompt_alignment", spec_path.name) + + import yaml + + from docgen.image_generate import collect_alignment_corpus + from docgen.scene_spec import image_prompt_alignment_violations, iter_image_elements + + try: + raw = yaml.safe_load(spec_path.read_text(encoding="utf-8")) + except (OSError, yaml.YAMLError) as exc: + return CheckResult( + "image_prompt_alignment", + False, + [f"could not load {spec_path.name}: {exc}"], + ) + if not isinstance(raw, dict): + return CheckResult( + "image_prompt_alignment", + False, + [f"{spec_path.name}: root must be a mapping"], + ) + if not iter_image_elements(raw): + return CheckResult( + "image_prompt_alignment", + True, + ["No image elements (skipped)"], + ) + + corpus = collect_alignment_corpus(self.config, raw) + issues = image_prompt_alignment_violations(raw, corpus_text=corpus) + if issues: + return CheckResult("image_prompt_alignment", False, issues) + if not corpus.strip(): + return CheckResult( + "image_prompt_alignment", + True, + ["No narration/source corpus yet — image prompts not checked"], + ) + return CheckResult( + "image_prompt_alignment", + True, + ["Image prompts share documented terms with narration/source"], + ) + def _check_scene_assets(self, seg_id: str) -> CheckResult: """Pre-render stuck / overlap / font / compile-sync gate (no video required).""" sa_cfg = self.config.scene_assets_config diff --git a/src/docgen/yaml_generate.py b/src/docgen/yaml_generate.py index c8163ac..0e87256 100644 --- a/src/docgen/yaml_generate.py +++ b/src/docgen/yaml_generate.py @@ -220,6 +220,7 @@ def merge_defaults( raw["image_generation"] = { "model": "gpt-image-1", "size": "1536x1024", + "align_with_docs": True, } changes.append( "image_generation: added model gpt-image-1 (Cursor/OpenAI Images; override with --model)" diff --git a/tests/test_config.py b/tests/test_config.py index cb79082..217355a 100644 --- a/tests/test_config.py +++ b/tests/test_config.py @@ -1136,7 +1136,8 @@ def test_from_yaml_validation_enable_bools_allowed(tmp_path: Path) -> None: " scene_assets:\n enabled: true\n" " story_end:\n enabled: false\n" " layout:\n check_overlap: false\n" - " subject_beat_coverage:\n enabled: false\n", + " subject_beat_coverage:\n enabled: false\n" + " image_prompt_alignment:\n enabled: false\n", encoding="utf-8", ) c = Config.from_yaml(p) @@ -1148,3 +1149,27 @@ def test_from_yaml_validation_enable_bools_allowed(tmp_path: Path) -> None: assert c.story_end_config["enabled"] is False assert c.layout_config["check_overlap"] is False assert c.subject_beat_coverage_enabled is False + assert c.image_prompt_alignment_enabled is False + + +def test_from_yaml_int_image_prompt_alignment_enabled_raises(tmp_path: Path) -> None: + p = tmp_path / "docgen.yaml" + p.write_text( + "validation:\n image_prompt_alignment:\n enabled: 0\n", + encoding="utf-8", + ) + with pytest.raises( + ConfigError, + match="validation.image_prompt_alignment.enabled must be a YAML boolean", + ): + Config.from_yaml(p) + + +def test_from_yaml_int_image_align_with_docs_raises(tmp_path: Path) -> None: + p = tmp_path / "docgen.yaml" + p.write_text("image_generation:\n align_with_docs: 1\n", encoding="utf-8") + with pytest.raises( + ConfigError, + match="image_generation.align_with_docs must be a YAML boolean", + ): + Config.from_yaml(p) diff --git a/tests/test_image_generate.py b/tests/test_image_generate.py index ea0df85..9f1b916 100644 --- a/tests/test_image_generate.py +++ b/tests/test_image_generate.py @@ -9,7 +9,9 @@ from docgen.config import Config from docgen.image_generate import ( + DEFAULT_IMAGE_STYLE, ImageGenerationError, + build_aligned_image_prompt, generate_images_for_spec, generate_missing_images_for_bundle, spec_files_for_bundle, @@ -82,6 +84,8 @@ def test_dry_run_reports_without_writing(cfg: Config) -> None: results = generate_images_for_spec(cfg, spec, dry_run=True, image_fn=lambda p: _PNG_BYTES) assert [r.status for r in results] == ["dry-run"] assert results[0].prompt == "a diagram" + assert DEFAULT_IMAGE_STYLE.split(".")[0] in results[0].effective_prompt + assert "Illustration request:" in results[0].effective_prompt assert not (cfg.base_dir / "images" / "arch.png").exists() @@ -125,6 +129,104 @@ def test_no_specs_dir_is_noop(cfg: Config) -> None: assert generate_missing_images_for_bundle(cfg, image_fn=lambda p: _PNG_BYTES) == [] +def test_aligned_prompt_is_what_the_provider_sees(cfg: Config) -> None: + spec = _write_spec( + cfg.animations_dir / "specs" / "01-x.scene.yaml", + prompt="clean diagram of the bootstrap pipeline", + ) + (cfg.narration_dir).mkdir(parents=True, exist_ok=True) + (cfg.narration_dir / "1.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", encoding="utf-8" + ) + seen: list[str] = [] + + def _capture(prompt: str) -> bytes: + seen.append(prompt) + return _PNG_BYTES + + generate_images_for_spec(cfg, spec, image_fn=_capture) + assert seen + assert "bootstrap pipeline" in seen[0] + assert "Documented subject" in seen[0] + assert seen[0] != "clean diagram of the bootstrap pipeline" + + +def test_unaligned_prompt_fails_before_provider(cfg: Config) -> None: + spec = _write_spec( + cfg.animations_dir / "specs" / "01-x.scene.yaml", + prompt="isometric render of the WidgetX orchestrator", + ) + cfg.narration_dir.mkdir(parents=True, exist_ok=True) + (cfg.narration_dir / "1.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", encoding="utf-8" + ) + with pytest.raises(ImageGenerationError, match="image prompt alignment"): + generate_images_for_spec(cfg, spec, image_fn=lambda p: _PNG_BYTES) + assert not (cfg.base_dir / "images" / "arch.png").exists() + + +def test_align_with_docs_false_sends_authored_prompt(tmp_path: Path) -> None: + (tmp_path / "docgen.yaml").write_text( + yaml.dump( + { + "segments": {"all": ["1"]}, + "image_generation": {"align_with_docs": False}, + } + ), + encoding="utf-8", + ) + cfg = Config.from_yaml(tmp_path / "docgen.yaml") + spec = _write_spec( + cfg.animations_dir / "specs" / "01-x.scene.yaml", + prompt="isometric render of the WidgetX orchestrator", + ) + cfg.narration_dir.mkdir(parents=True, exist_ok=True) + (cfg.narration_dir / "1.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", encoding="utf-8" + ) + seen: list[str] = [] + generate_images_for_spec(cfg, spec, image_fn=lambda p: seen.append(p) or _PNG_BYTES) + assert seen == ["isometric render of the WidgetX orchestrator"] + + +def test_source_docs_ground_prompt_without_narration(tmp_path: Path) -> None: + (tmp_path / "docs").mkdir() + (tmp_path / "docs" / "arch.md").write_text( + "The checkout service owns the cart lock.\n", encoding="utf-8" + ) + (tmp_path / "docgen.yaml").write_text( + yaml.dump( + { + "repo_root": ".", + "segments": {"all": ["1"]}, + "manim_scene_generation": {"context": {"paths": ["docs/arch.md"]}}, + } + ), + encoding="utf-8", + ) + cfg = Config.from_yaml(tmp_path / "docgen.yaml") + spec = _write_spec( + cfg.animations_dir / "specs" / "01-x.scene.yaml", + prompt="clean diagram of the checkout service", + ) + seen: list[str] = [] + generate_images_for_spec(cfg, spec, image_fn=lambda p: seen.append(p) or _PNG_BYTES) + assert seen + assert "checkout service" in seen[0] + assert "cart lock" in seen[0] + + +def test_build_aligned_image_prompt_includes_corpus_and_label() -> None: + out = build_aligned_image_prompt( + "clean diagram of the bootstrap pipeline", + corpus_text="The bootstrap pipeline seeds the cluster.", + label="bootstrap", + ) + assert "bootstrap pipeline" in out + assert "On-screen timing label" in out + assert "Illustration request:" in out + + def test_cli_image_generate_all_fails_when_manim_has_no_specs(tmp_path: Path) -> None: from click.testing import CliRunner diff --git a/tests/test_scene_spec.py b/tests/test_scene_spec.py index 6471196..16cf2a0 100644 --- a/tests/test_scene_spec.py +++ b/tests/test_scene_spec.py @@ -17,6 +17,7 @@ disk_spec_with_merged_wait_words, cluster_subject_beats, count_spec_labels, + image_prompt_alignment_violations, iter_paced_label_anchors, last_paced_reveal_time, layout_budget_violations, @@ -569,6 +570,39 @@ def test_layout_stack_budget_decreases_with_larger_title_font() -> None: assert b_small > b_large +def test_image_prompt_alignment_requires_documented_terms() -> None: + spec = { + "title": {"text": "T", "font_size": 36, "color": "C_WHITE"}, + "rows": [ + { + "run_time": 1.0, + "boxes": [ + { + "image": "images/arch.png", + "width": 4.0, + "height": 2.5, + "prompt": "clean flat diagram of the bootstrap pipeline", + } + ], + } + ], + } + narr = "The bootstrap pipeline seeds the cluster." + assert image_prompt_alignment_violations(spec, corpus_text=narr) == [] + + spec["rows"][0]["boxes"][0]["prompt"] = "a clean flat illustration" + issues = image_prompt_alignment_violations(spec, corpus_text=narr) + assert issues + assert any("only visual style" in msg for msg in issues) + + spec["rows"][0]["boxes"][0]["prompt"] = "isometric render of the WidgetX orchestrator" + issues = image_prompt_alignment_violations(spec, corpus_text=narr) + assert issues + assert any("shares no documented terms" in msg for msg in issues) + + assert image_prompt_alignment_violations(spec, corpus_text="") == [] + + def test_subject_beat_coverage_allows_dwell_rejects_missed_topics() -> None: narr = ( "The bootstrap pipeline seeds the cluster. " diff --git a/tests/test_scene_spec_generate.py b/tests/test_scene_spec_generate.py index 32f0a10..c294c5f 100644 --- a/tests/test_scene_spec_generate.py +++ b/tests/test_scene_spec_generate.py @@ -370,6 +370,60 @@ def test_user_message_includes_computed_layout_stack_budgets() -> None: assert "13.22" in msg # horizontal safe width (FRAME_WIDTH - 1.0) +def test_user_message_includes_source_snippets() -> None: + msg = build_scene_spec_user_message( + seg_id="01", + seg_name="01-x", + class_name="XScene", + narration_text="The bootstrap pipeline seeds the cluster.", + timing_enrichment="(no timing)", + hints=[], + extra_hints=[], + reference_scenes="", + source_snippets=[("docs/architecture.md", "The bootstrap pipeline writes a lockfile.")], + ) + assert "SOURCE DOCUMENTATION" in msg + assert "docs/architecture.md" in msg + assert "writes a lockfile" in msg + assert "Do not invent product names" in msg + + +def test_generate_rejects_unaligned_image_prompt(tmp_path: Path) -> None: + cfg = _bundle(tmp_path) + + def fake_llm(**_kwargs: object) -> str: + return """```yaml +segment_id: "08" +class_name: ExtrasScene +title: + text: "Synthetic" + font_size: 40 + color: C_WHITE +rows: + - run_time: 1.2 + boxes: + - label: "Hello" + color: C_ORANGE + width: 4.0 + height: 1.0 + font_size: 20 + - image: images/widget.png + width: 4.0 + height: 2.0 + prompt: "isometric render of the WidgetX orchestrator" +```""" + + with pytest.raises(SceneGenerationError, match="image prompt alignment"): + generate_scene_spec( + cfg, + "08", + extra_paths=[], + extra_hints=[], + dry_run=False, + llm=fake_llm, + ) + + def test_scene_spec_generate_all_uses_default_when_all_missing(tmp_path: Path) -> None: from click.testing import CliRunner diff --git a/tests/test_validate.py b/tests/test_validate.py index e2fb80e..6395904 100644 --- a/tests/test_validate.py +++ b/tests/test_validate.py @@ -623,6 +623,101 @@ def test_validate_passes_covered_beats(self, cfg_dir: Path) -> None: assert check["passed"], check["details"] +class TestImagePromptAlignmentValidate: + def test_validate_flags_unaligned_image_prompt(self, cfg_dir: Path) -> None: + cfg_raw = yaml.safe_load((cfg_dir / "docgen.yaml").read_text(encoding="utf-8")) + cfg_raw["visual_map"]["01"] = {"type": "manim", "source": "Scene01.mp4"} + cfg_raw.setdefault("segment_names", {})["01"] = "01-test" + (cfg_dir / "docgen.yaml").write_text(yaml.dump(cfg_raw), encoding="utf-8") + (cfg_dir / "narration" / "01-test.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", + encoding="utf-8", + ) + specs = cfg_dir / "animations" / "specs" + specs.mkdir(parents=True, exist_ok=True) + (specs / "01-test.scene.yaml").write_text( + yaml.dump( + { + "segment_id": "01", + "class_name": "DemoScene", + "title": {"text": "T", "font_size": 36, "color": "C_WHITE"}, + "rows": [ + { + "run_time": 1.0, + "boxes": [ + { + "label": "bootstrap pipeline", + "color": "C_ORANGE", + "width": 4.0, + "height": 1.0, + "font_size": 18, + }, + { + "image": "images/arch.png", + "width": 4.0, + "height": 2.0, + "prompt": "isometric render of the WidgetX orchestrator", + }, + ], + } + ], + } + ), + encoding="utf-8", + ) + config = Config.from_yaml(cfg_dir / "docgen.yaml") + report = Validator(config).validate_segment("01") + check = next(c for c in report["checks"] if c["name"] == "image_prompt_alignment") + assert not check["passed"] + assert any("documented terms" in d for d in check["details"]) + + def test_validate_passes_grounded_image_prompt(self, cfg_dir: Path) -> None: + cfg_raw = yaml.safe_load((cfg_dir / "docgen.yaml").read_text(encoding="utf-8")) + cfg_raw["visual_map"]["01"] = {"type": "manim", "source": "Scene01.mp4"} + cfg_raw.setdefault("segment_names", {})["01"] = "01-test" + (cfg_dir / "docgen.yaml").write_text(yaml.dump(cfg_raw), encoding="utf-8") + (cfg_dir / "narration" / "01-test.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", + encoding="utf-8", + ) + specs = cfg_dir / "animations" / "specs" + specs.mkdir(parents=True, exist_ok=True) + (specs / "01-test.scene.yaml").write_text( + yaml.dump( + { + "segment_id": "01", + "class_name": "DemoScene", + "title": {"text": "T", "font_size": 36, "color": "C_WHITE"}, + "rows": [ + { + "run_time": 1.0, + "boxes": [ + { + "label": "bootstrap pipeline", + "color": "C_ORANGE", + "width": 4.0, + "height": 1.0, + "font_size": 18, + }, + { + "image": "images/arch.png", + "width": 4.0, + "height": 2.0, + "prompt": "clean diagram of the bootstrap pipeline", + }, + ], + } + ], + } + ), + encoding="utf-8", + ) + config = Config.from_yaml(cfg_dir / "docgen.yaml") + report = Validator(config).validate_segment("01") + check = next(c for c in report["checks"] if c["name"] == "image_prompt_alignment") + assert check["passed"], check["details"] + + # ── ffprobe JSON probes honor returncode ────────────────────────────── class _FakeProbe: @@ -686,7 +781,7 @@ def test_drift_pass_when_ffprobe_succeeds(self, config, tmp_path, monkeypatch): @pytest.mark.parametrize( "check_name", - ("av_sync", "subject_beat_coverage", "ocr_scan", "layout", "freeze_ratio"), + ("av_sync", "subject_beat_coverage", "image_prompt_alignment", "ocr_scan", "layout", "freeze_ratio"), ) def test_run_pre_push_visual_sync_fail_is_hard(check_name: str, capsys) -> None: """Visual-sync FAILs must be FAIL + SystemExit, not WARN (leftover #2).""" diff --git a/tests/test_validate_timing_sync.py b/tests/test_validate_timing_sync.py index 2775ca4..26e264e 100644 --- a/tests/test_validate_timing_sync.py +++ b/tests/test_validate_timing_sync.py @@ -435,7 +435,10 @@ def _write_paced_hand_authored_scenes(cfg: Config) -> None: ) -@pytest.mark.parametrize("check_name", ("story_end", "subject_beat_coverage")) +@pytest.mark.parametrize( + "check_name", + ("story_end", "subject_beat_coverage", "image_prompt_alignment"), +) def test_hand_authored_paced_manim_without_spec_fails(cfg, check_name: str) -> None: """Manim + paced timed_play + no *.scene.yaml must fail, not skip-PASS.""" _write_paced_hand_authored_scenes(cfg) @@ -446,6 +449,8 @@ def test_hand_authored_paced_manim_without_spec_fails(cfg, check_name: str) -> N v = Validator(cfg) if check_name == "story_end": check = v._check_story_end("01") + elif check_name == "image_prompt_alignment": + check = v._check_image_prompt_alignment("01") else: check = v._check_subject_beat_coverage("01") assert check.name == check_name diff --git a/tests/test_yaml_generate.py b/tests/test_yaml_generate.py index 8041e4f..5501f5f 100644 --- a/tests/test_yaml_generate.py +++ b/tests/test_yaml_generate.py @@ -85,6 +85,7 @@ def test_merge_defaults_adds_archive_exclude(tmp_path: Path) -> None: assert raw["ai"]["provider"] == "openai" assert raw["image_generation"]["model"] == "gpt-image-1" assert raw["image_generation"]["size"] == "1536x1024" + assert raw["image_generation"]["align_with_docs"] is True def test_merge_defaults_idempotent_archive(tmp_path: Path) -> None: From 1df0651d4feae2e82c726f04568b63b706ab4ea1 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Sat, 26 Sep 2026 23:12:21 +0000 Subject: [PATCH 02/17] Fix the scene-spec-generate alignment test to use a spoken image label. The image stem was treated as an invented subject-beat label and failed coverage before the new prompt-alignment gate could run. Co-authored-by: jmjava --- tests/test_scene_spec_generate.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tests/test_scene_spec_generate.py b/tests/test_scene_spec_generate.py index c294c5f..199c0d7 100644 --- a/tests/test_scene_spec_generate.py +++ b/tests/test_scene_spec_generate.py @@ -407,9 +407,10 @@ def fake_llm(**_kwargs: object) -> str: width: 4.0 height: 1.0 font_size: 20 - - image: images/widget.png + - image: images/hello.png width: 4.0 height: 2.0 + label: "Hello" prompt: "isometric render of the WidgetX orchestrator" ```""" From 6212a134ae1c33341b7d47c3042390bded24fe57 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Sat, 26 Sep 2026 23:21:26 +0000 Subject: [PATCH 03/17] Review generated image pixels against the documentation. Prompt grounding only checked the caption. image-generate now OCRs the PNG for invented labels and vision-reviews it (OpenAI / Grok / Claude), retries once with the critique, and deletes a failing asset. validate adds image_asset_alignment (OCR on by default; vision opt-in). Co-authored-by: jmjava --- AGENTS.md | 5 +- README.md | 24 +++-- src/docgen/ai_client.py | 104 +++++++++++++++++++- src/docgen/config.py | 55 +++++++++++ src/docgen/image_align.py | 150 +++++++++++++++++++++++++++++ src/docgen/image_generate.py | 121 +++++++++++++++++++++-- src/docgen/init.py | 1 + src/docgen/scene_spec.py | 28 ++++++ src/docgen/validate.py | 105 +++++++++++++++++++- src/docgen/yaml_generate.py | 1 + tests/test_ai_client.py | 64 ++++++++++++ tests/test_config.py | 29 +++++- tests/test_image_align.py | 73 ++++++++++++++ tests/test_image_generate.py | 72 ++++++++++++++ tests/test_scene_spec.py | 16 +++ tests/test_validate.py | 60 +++++++++++- tests/test_validate_timing_sync.py | 4 +- tests/test_yaml_generate.py | 1 + 18 files changed, 889 insertions(+), 24 deletions(-) create mode 100644 src/docgen/image_align.py create mode 100644 tests/test_image_align.py diff --git a/AGENTS.md b/AGENTS.md index e14875f..5b141a8 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -70,10 +70,10 @@ Commands registered on the **`docgen`** CLI include: - **`freeze`** — ``docgen freeze`` builds the **`docgen-gui`** onedir (`pip install 'docgen[packaging]'`). Optional ``--smoke`` runs the binary headless. Do not run a full freeze in routine pytest; set ``DOCGEN_FREEZE_SMOKE=1`` for the optional test. - **`tts`** — text-to-speech for segment files (OpenAI or xAI `/v1/tts`). - **`timestamps`** — word/segment timing (`timing.json`). Default engine **`local`** aligns the known narration text against the mp3 offline (ffmpeg silencedetect, no API); **`--engine whisper`** uses OpenAI whisper-1 or xAI `/v1/stt` when `ai.provider` is grok. Both emit the same Whisper-shaped blocks. Failed ffmpeg silencedetect raises `AlignmentError` (empty stderr is not treated as full-span speech). OpenAI whisper-1 word/segment `start`/`end` must be finite JSON numbers (bool/NaN raise `AIError`). Grok `/v1/stt` rejects empty word tokens and inverted `end < start` intervals (`AIError`). Empty ``segments.all`` raises ``TimestampError`` (same as TTS) and does not leave a stale ``timing.json`` as success. -- **`image-generate`** — render scene-spec **image elements** (`image:` + `prompt:` boxes) via OpenAI Images or xAI Imagine into the bundle (also runs for missing assets inside `generate-all`). Default **`align_with_docs`** wraps each prompt with narration + source snippets and fails when the authored prompt shares no documented terms. +- **`image-generate`** — render scene-spec **image elements** (`image:` + `prompt:` boxes) via OpenAI Images or xAI Imagine into the bundle (also runs for missing assets inside `generate-all`). Default **`align_with_docs`** wraps each prompt with narration + source snippets and fails when the authored prompt shares no documented terms. After the PNG is written, **OCR** plus a **vision review** (`align_review`, default true) check the pixels against the same corpus and retry once on FAIL. - **`manim`** — render Manim scenes declared in config. - **`compose`** — mux narration audio with visual sources via ffmpeg. With no segment ids, uses ``segments.all`` (same as ``generate-all``), not ``segments.default``. A ``type: mixed`` row raises ``ComposeError`` if any listed source is missing (no silent subset mux). -- **`validate`** / **`validate --pre-push`** — drift, narration lint, Manim hints, **`timing_sync`**, **`story_end`** (last paced reveal vs audio end; hard fail), **`scene_assets`** (pre-render: stuck-board cadence, frame-budget overlaps, `MANIM_FONT` consistency, stale helpers / stale compiled class including hand-edited generated-region labels and `run_time` — hard fail; also a `generate-all` gate before Manim), **`av_sync`** (hard fail on `--pre-push` / `generate-all`; prefers scene-spec labels as OCR anchors), **`subject_beat_coverage`** (declarative specs vs narration topic beats; hard fail when enabled), **`image_prompt_alignment`** (image-element prompts must share documented terms with narration/source; hard fail when enabled), and related visual-sync checks (`ocr_scan`, `layout`, `freeze_ratio` — hard fail on `--pre-push` / `generate-all`). Missing tesseract fails `ocr_scan` / `av_sync` / `layout` (not skip-PASS). Missing audio or an LFS pointer fails `timing_sync` and recording media gates (`stream_presence`, `av_drift`, `ocr_scan`, `av_sync`) (not skip-PASS). Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` / `image_prompt_alignment` for `type: manim` (not skip-PASS). +- **`validate`** / **`validate --pre-push`** — drift, narration lint, Manim hints, **`timing_sync`**, **`story_end`** (last paced reveal vs audio end; hard fail), **`scene_assets`** (pre-render: stuck-board cadence, frame-budget overlaps, `MANIM_FONT` consistency, stale helpers / stale compiled class including hand-edited generated-region labels and `run_time` — hard fail; also a `generate-all` gate before Manim), **`av_sync`** (hard fail on `--pre-push` / `generate-all`; prefers scene-spec labels as OCR anchors), **`subject_beat_coverage`** (declarative specs vs narration topic beats; hard fail when enabled), **`image_prompt_alignment`** (image-element prompts must share documented terms with narration/source; hard fail when enabled), **`image_asset_alignment`** (OCR of generated PNGs vs docs; optional vision when `validation.image_asset_alignment.review`; hard fail when enabled), and related visual-sync checks (`ocr_scan`, `layout`, `freeze_ratio` — hard fail on `--pre-push` / `generate-all`). Missing tesseract fails `ocr_scan` / `av_sync` / `layout` / `image_asset_alignment` (not skip-PASS). Missing audio or an LFS pointer fails `timing_sync` and recording media gates (`stream_presence`, `av_drift`, `ocr_scan`, `av_sync`) (not skip-PASS). Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` / `image_prompt_alignment` / `image_asset_alignment` for `type: manim` (not skip-PASS). - **`lint`** — narration lint helper. - **`narration-generate`** — LLM-assisted narration from hints and repo context; optional **`--revise --revision-notes`** for in-place edits (same contract as the wizard Revise button). - **`scene-spec-generate`** — LLM emits declarative **`*.scene.yaml`**; enforces frame budget + **subject-beat coverage** (dwell OK; cover topic shifts; reject invented labels) and **image-prompt alignment** (image `prompt:` must use documented terms from narration/source). @@ -92,6 +92,7 @@ Commands registered on the **`docgen`** CLI include: - **Beat sync (fail-closed):** when `timing.json` has words, every story box label must match a spoken phrase (`wait_word`); unmatched labels and leftover LLM indices are rejected. Opt out with ``pace: none``. Legacy row-level ``wait_segment`` is upgraded to ``wait_word`` and written back on ``scene-compile`` (direct ``compile_scene_class`` still rejects leftover ``wait_segment``). Fuzzy containment matching is not used. **`scene-compile` clamps FadeIn / page-fade `run_time` against the next word start** so `_TimedScene._clock` cannot race past waits (issue #66 — do not emit cascading first-board dumps). After a reveal, a **dwell** slot may play `Indicate` / `Circumscribe` when the gap to the next `wait_word` is long enough (also clamped). Long holds emit additional mid-hold pulses (`timed_wait` + emphasis) so the board does not freeze after the first Indicate. Optional box fields: `shape` (rounded/pill/diamond), `reveal` (fade/grow/slide), `emphasis` (none/pulse/ring). Page transitions FadeOut revealed boxes, not the parent `VGroup`. `scene-compile` refreshes stale `_box` / `_arrow` / `_TimedScene` helpers in `scenes.py`. - **Subject-beat coverage:** implemented in `scene_spec.layout_density_violations` / `cluster_subject_beats`; enforced by **`scene-spec-generate`** and **`validate`** (`validation.subject_beat_coverage.enabled`, default true). Not a blind label count. - **Image-prompt alignment:** implemented in `scene_spec.image_prompt_alignment_violations`; **`scene-spec-generate`** and **`image-generate`** reject prompts that share no documented terms; **`validate`** (`validation.image_prompt_alignment.enabled`, default true) is the same gate. `image-generate` also wraps the Images API prompt with narration/source (`image_generation.align_with_docs`, default true). +- **Image-asset alignment:** OCR + vision review of the generated PNG (`docgen.image_align`). **`image-generate`** fails closed and retries once (`image_generation.align_review`, default true). **`validate`** (`validation.image_asset_alignment`, OCR on by default, vision opt-in) is the same pixel gate. - **`docgen benchmark` is required** after clock / compile / `_TimedScene` / dwell changes. Pytest string assertions are not a substitute. Do not remove the CI `benchmark` job. See **Required gate** below. - Prefer **stable CLI / library contracts** and **documented exit codes** so CI can depend on them. - **`narration_from_source`:** hints in config + **`docgen narration-generate`** — owner-supplied context paths, not opaque bulk edits to outputs. diff --git a/README.md b/README.md index e6802e1..0e6e018 100644 --- a/README.md +++ b/README.md @@ -58,23 +58,27 @@ If you still need the legacy behaviour, pin a pre-removal commit (`grok-imagine-image-2.0` when `ai.provider` is `grok`). By default the authored prompt is **grounded** in the segment narration plus `manim_scene_generation` source snippets, and prompts that share no - documented terms fail closed (`image_prompt_alignment`). The compiled scene - shows the asset with the `_image` helper. `generate-all` fills in missing - assets automatically. + documented terms fail closed (`image_prompt_alignment`). After the PNG is + written, OCR rejects invented on-image labels and a vision model reviews + the pixels against the same docs (`image_asset_alignment` / + `image_generation.align_review`). The compiled scene shows the asset with + the `_image` helper. `generate-all` fills in missing assets automatically. - **ffmpeg composition** — combine narration audio and Manim video into final segments, with a freeze-tail guard. - **Validation** — A/V drift, freeze ratio, OCR error scan, layout, narration lint, Manim scene lint, **timing_sync** (stale `timing.json` vs regenerated mp3 — hard fail), **story_end** (paced visual story finishes long before narration — hard fail), **image_prompt_alignment** (image-element `prompt:` must share - documented terms with narration/source — hard fail), and **av_sync** (OCR + documented terms with narration/source — hard fail), **image_asset_alignment** + (OCR of generated PNGs vs docs; optional vision review — hard fail), and **av_sync** (OCR check that scene-spec label anchors appear on screen near their spoken time — hard fail on `--pre-push` / `generate-all`). Missing tesseract fails `ocr_scan` / `av_sync` / `layout` instead of skip-PASS. Missing audio or an LFS pointer fails `timing_sync` and recording media gates (`stream_presence`, `av_drift`, `ocr_scan`, `av_sync`) instead of skip-PASS. Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` / - `image_prompt_alignment` for `type: manim` instead of skip-PASS. A hand-edited generated-region label + `image_prompt_alignment` / `image_asset_alignment` for `type: manim` + instead of skip-PASS. A hand-edited generated-region label or `run_time` in `scenes.py` fails `scene_assets` compile_sync (clock / benchmark still execute a fresh `compile_scene_class`, not the on-disk file). - **GitHub Pages** — auto-generate `index.html`, deploy workflow, LFS rules, @@ -134,6 +138,8 @@ image_generation: model: gpt-image-1 # or dall-e-3, gpt-image-1-mini, … size: 1536x1024 align_with_docs: true # wrap prompts with narration/source; fail invented terms + align_review: true # vision-review the PNG; retry once on FAIL + # review_model: gpt-4o # quality: high # style: "…" # optional override of the educational-diagram prefix ``` @@ -248,7 +254,7 @@ docgen --repo /path/to/your-project generate-all | `docgen freeze [--dist DIR] [--smoke]` | PyInstaller onedir for **`docgen-gui` only** (`pip install 'docgen[packaging]'`). Not the full Manim CLI | | `docgen tts [--segment 01] [--dry-run]` | Generate TTS audio | | `docgen timestamps [--engine local\|whisper]` | Extract word/segment timestamps from TTS audio → `timing.json` (default `local`: offline narration-text alignment; `whisper`: OpenAI transcription). Empty `segments.all` is `TimestampError` (stale `timing.json` is not success) | -| `docgen image-generate [--segment 01 \| --all \| --spec PATH] [--force] [--dry-run] [--model …] [--size …]` | Generate scene-spec image assets (`image:` + `prompt:` boxes) via the OpenAI Images API; grounds prompts in narration/source and fails unaligned prompts unless `image_generation.align_with_docs` is false | +| `docgen image-generate [--segment 01 \| --all \| --spec PATH] [--force] [--dry-run] [--model …] [--size …]` | Generate scene-spec image assets (`image:` + `prompt:` boxes) via the OpenAI Images API; grounds prompts in narration/source, then OCR + vision-reviews the PNG unless `align_with_docs` / `align_review` is false | | `docgen manim [--scene StackDAGScene]` | Render Manim animations | | `docgen compose [01 02 03] [--ffmpeg-timeout 900]` | Compose segments (audio + video). Omit ids to walk `segments.all` (same as `generate-all`); a mapped segment with missing audio/visuals is a hard fail | | `docgen validate [--max-drift 2.75] [--pre-push]` | Run all validation checks | @@ -352,6 +358,8 @@ image_generation: # scene-spec image elements (docgen image-generate) model: gpt-image-1 # Cursor/OpenAI Images; Grok remaps gpt-image-* to Imagine size: 1536x1024 align_with_docs: true # ground prompts in narration/source (default) + align_review: true # vision-review generated PNGs (default) + # review_model: gpt-4o # quality: high # optional, model-specific # style: "…" # optional Images-API prefix override @@ -366,6 +374,10 @@ validation: enabled: true # scene-spec-generate + validate: cover narration topic beats image_prompt_alignment: enabled: true # image-element prompts must use documented terms + image_asset_alignment: + enabled: true # OCR generated PNGs vs narration/source + ocr: true + review: false # set true to vision-review existing assets in validate compose: ffmpeg_timeout_sec: 300 # can also be overridden with: docgen compose --ffmpeg-timeout N diff --git a/src/docgen/ai_client.py b/src/docgen/ai_client.py index 16a3bb2..536ae97 100644 --- a/src/docgen/ai_client.py +++ b/src/docgen/ai_client.py @@ -9,7 +9,8 @@ Chat + images for OpenAI/Grok go through the ``openai`` SDK. xAI is ``base_url=https://api.x.ai/v1`` plus model aliases. Anthropic chat uses -``POST https://api.anthropic.com/v1/messages`` (no TTS/images there). +``POST https://api.anthropic.com/v1/messages`` (no image *generation* there; +Claude can still *review* a PNG via ``chat_completion_with_image``). TTS/STT: OpenAI ``/v1/audio/speech`` + ``whisper-1``, or xAI ``/v1/tts`` and ``/v1/stt``. @@ -393,6 +394,89 @@ def _create() -> Any: return text +def chat_completion_with_image( + *, + system_prompt: str, + user_message: str, + image_bytes: bytes, + media_type: str = "image/png", + model: str, + temperature: float, + cfg: "Config | None" = None, +) -> str: + """Vision chat: attach one image plus text (OpenAI, Grok, or Anthropic).""" + import base64 + + settings = resolve_ai_settings(cfg) + resolved = resolve_chat_model(model, settings) + if not image_bytes: + raise AIError("vision chat needs non-empty image bytes") + mime = (media_type or "image/png").strip() or "image/png" + b64 = base64.b64encode(image_bytes).decode("ascii") + if settings.is_anthropic: + return _anthropic_chat( + system_prompt=system_prompt, + user_message=user_message, + model=resolved, + temperature=temperature, + settings=settings, + image_b64=b64, + image_media_type=mime, + ) + + import openai + + from docgen.openai_retry import call_with_rate_limit_retries + + client = openai_client(cfg) + + def _create() -> Any: + return client.chat.completions.create( + model=resolved, + messages=[ + {"role": "system", "content": system_prompt}, + { + "role": "user", + "content": [ + {"type": "text", "text": user_message}, + { + "type": "image_url", + "image_url": {"url": f"data:{mime};base64,{b64}"}, + }, + ], + }, + ], + temperature=float(temperature), + ) + + try: + response = call_with_rate_limit_retries(_create) + except openai.AuthenticationError as exc: + raise AIError( + f"{_vendor(settings)} rejected {settings.api_key_env} (authentication failed): {exc}. " + f"{settings.auth_help()}" + ) from exc + except openai.PermissionDeniedError as exc: + raise AIError( + f"{_vendor(settings)} permission denied for model {resolved!r}: {exc}." + ) from exc + except openai.RateLimitError as exc: + raise AIError( + f"{_vendor(settings)} rate-limited for model {resolved!r}: {exc}." + ) from exc + except openai.APIConnectionError as exc: + raise AIError( + f"{_vendor(settings)} connection error: {exc} — re-run when connectivity is restored." + ) from exc + try: + text = (response.choices[0].message.content or "").strip() + except (IndexError, AttributeError): + text = "" + if not text: + raise AIError(f"{_vendor(settings)} vision chat returned no text content.") + return text + + def synthesize_speech( *, text: str, @@ -556,15 +640,31 @@ def _anthropic_chat( model: str, temperature: float, settings: AISettings, + image_b64: str | None = None, + image_media_type: str = "image/png", ) -> str: if not settings.api_key: raise AIError(f"Anthropic chat needs ANTHROPIC_API_KEY. {settings.auth_help()}") + if image_b64: + user_content: Any = [ + { + "type": "image", + "source": { + "type": "base64", + "media_type": image_media_type or "image/png", + "data": image_b64, + }, + }, + {"type": "text", "text": user_message}, + ] + else: + user_content = user_message payload = { "model": model, "max_tokens": ANTHROPIC_MAX_TOKENS, "temperature": float(temperature), "system": system_prompt, - "messages": [{"role": "user", "content": user_message}], + "messages": [{"role": "user", "content": user_content}], } raw = _http_json( _anthropic_messages_url(settings), diff --git a/src/docgen/config.py b/src/docgen/config.py index 184cd44..0aba5d2 100644 --- a/src/docgen/config.py +++ b/src/docgen/config.py @@ -332,6 +332,7 @@ def __post_init__(self) -> None: "narration_lint", "subject_beat_coverage", "image_prompt_alignment", + "image_asset_alignment", ): self._sub_block(validation, nested, label=f"validation.{nested}") # List-valued keys: a string must not be iterated as characters. @@ -494,6 +495,22 @@ def __post_init__(self) -> None: ) if ig.get("style") is not None: require_yaml_string(ig["style"], label="image_generation.style", source=src) + if ig.get("align_review") is not None: + require_yaml_bool( + ig["align_review"], + label="image_generation.align_review", + source=src, + ) + if ig.get("review_model") is not None: + require_yaml_string( + ig["review_model"], label="image_generation.review_model", source=src + ) + if ig.get("align_review_retries") is not None: + require_yaml_number( + ig["align_review_retries"], + label="image_generation.align_review_retries", + source=src, + ) ai = self._block("ai") if ai.get("provider") is not None: require_yaml_string(ai["provider"], label="ai.provider", source=src) @@ -786,6 +803,29 @@ def __post_init__(self) -> None: label="validation.image_prompt_alignment.enabled", source=src, ) + iaa = self._sub_block( + validation, + "image_asset_alignment", + label="validation.image_asset_alignment", + ) + if iaa.get("enabled") is not None: + require_yaml_bool( + iaa["enabled"], + label="validation.image_asset_alignment.enabled", + source=src, + ) + if iaa.get("ocr") is not None: + require_yaml_bool( + iaa["ocr"], + label="validation.image_asset_alignment.ocr", + source=src, + ) + if iaa.get("review") is not None: + require_yaml_bool( + iaa["review"], + label="validation.image_asset_alignment.review", + source=src, + ) def _source_label(self) -> str: return self.yaml_path.name if self.yaml_path else "docgen.yaml" @@ -971,6 +1011,8 @@ def image_generation_config(self) -> dict[str, Any]: "model": "gpt-image-1", "size": "1536x1024", "align_with_docs": True, + "align_review": True, + "align_review_retries": 1, } defaults.update(self._block("image_generation")) return defaults @@ -998,6 +1040,19 @@ def image_prompt_alignment_enabled(self) -> bool: return bool(block.get("enabled")) return True + @property + def image_asset_alignment_config(self) -> dict[str, Any]: + """Pixel checks on generated scene images (OCR + optional vision).""" + defaults: dict[str, Any] = {"enabled": True, "ocr": True, "review": False} + defaults.update( + self._sub_block( + self._block("validation"), + "image_asset_alignment", + label="validation.image_asset_alignment", + ) + ) + return defaults + # -- Manim ----------------------------------------------------------------- @property diff --git a/src/docgen/image_align.py b/src/docgen/image_align.py new file mode 100644 index 0000000..b31dc4d --- /dev/null +++ b/src/docgen/image_align.py @@ -0,0 +1,150 @@ +"""Pixel-level alignment of generated scene images against documentation. + +Prompt grounding (``image_prompt_alignment``) only checks the caption. This +module looks at the **PNG**: + +* **OCR** — readable tokens that are not in narration/source fail (invented + on-image labels). Empty OCR is fine (many diagrams have no text). +* **Vision review** — a chat model that accepts images (OpenAI, Grok, or + Claude) answers PASS/FAIL against the same documentation corpus. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path +from typing import TYPE_CHECKING, Callable + +if TYPE_CHECKING: + from docgen.config import Config + +DEFAULT_REVIEW_MODEL = "gpt-4o" + +_REVIEW_SYSTEM = """You review one illustration for a documentation video. + +Decide whether the IMAGE depicts the DOCUMENTED SUBJECT (narration + source). +PASS when the picture is a reasonable, simplified depiction of those concepts. +FAIL when it shows a different subject, invented product/API names, decorative +text that is not copied from the docs, fake UI chrome, or generic clip-art +unrelated to the documentation. + +Reply with EXACTLY two lines and nothing else: +PASS + +or +FAIL + +""" + + +@dataclass(frozen=True) +class ImageReviewResult: + passed: bool + reason: str + ocr_text: str = "" + + +def ocr_image_text(path: Path) -> str | None: + """OCR a still image. ``None`` if unreadable or tesseract is unavailable. + + Empty string means the file decoded but no text was found. + """ + try: + import cv2 + import pytesseract + + pytesseract.get_tesseract_version() + except Exception: + return None + img = cv2.imread(str(path)) + if img is None: + return None + gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) + _, thresh = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU) + try: + return str(pytesseract.image_to_string(thresh) or "") + except Exception: + return None + + +def parse_review_verdict(text: str) -> ImageReviewResult: + """Parse a PASS/FAIL vision reply. Unparseable text fails closed.""" + lines = [ln.strip() for ln in (text or "").splitlines() if ln.strip()] + if not lines: + return ImageReviewResult(False, "vision review returned empty text") + verdict = lines[0].split()[0].upper().strip(".:") + reason = lines[1] if len(lines) > 1 else " ".join(lines[1:]) or lines[0] + if verdict == "PASS": + return ImageReviewResult(True, reason) + if verdict == "FAIL": + return ImageReviewResult(False, reason) + preview = (text or "").strip() + if len(preview) > 80: + preview = preview[:77] + "..." + return ImageReviewResult(False, f"vision review reply was not PASS/FAIL: {preview!r}") + + +def build_review_user_message( + *, + corpus_text: str, + authored_prompt: str = "", + label: str = "", +) -> str: + parts = [ + "DOCUMENTED SUBJECT:", + (corpus_text or "").strip() or "(none)", + ] + if (authored_prompt or "").strip(): + parts.extend(["", "AUTHORED IMAGE PROMPT:", authored_prompt.strip()]) + if (label or "").strip(): + parts.extend(["", "ON-SCREEN LABEL:", label.strip()]) + parts.extend(["", "Review the attached image."]) + return "\n".join(parts) + + +def _media_type_for(path: Path) -> str: + ext = path.suffix.lower() + return { + ".jpg": "image/jpeg", + ".jpeg": "image/jpeg", + ".webp": "image/webp", + ".gif": "image/gif", + }.get(ext, "image/png") + + +def review_image_against_docs( + path: Path, + *, + corpus_text: str, + authored_prompt: str = "", + label: str = "", + cfg: "Config | None" = None, + model: str = "", + chat_fn: Callable[..., str] | None = None, +) -> ImageReviewResult: + """Send ``path`` + documentation to a vision-capable chat model.""" + from docgen.ai_client import AIError, chat_completion_with_image + + try: + raw = path.read_bytes() + except OSError as exc: + return ImageReviewResult(False, f"could not read image {path}: {exc}") + if not raw: + return ImageReviewResult(False, f"image {path} is empty") + user = build_review_user_message( + corpus_text=corpus_text, authored_prompt=authored_prompt, label=label + ) + invoke = chat_fn or chat_completion_with_image + try: + text = invoke( + system_prompt=_REVIEW_SYSTEM, + user_message=user, + image_bytes=raw, + media_type=_media_type_for(path), + model=(model or "").strip() or DEFAULT_REVIEW_MODEL, + temperature=0.0, + cfg=cfg, + ) + except (AIError, RuntimeError, OSError, TypeError, ValueError) as exc: + return ImageReviewResult(False, f"vision review failed: {exc}") + return parse_review_verdict(text) diff --git a/src/docgen/image_generate.py b/src/docgen/image_generate.py index b78a627..980cbb8 100644 --- a/src/docgen/image_generate.py +++ b/src/docgen/image_generate.py @@ -15,8 +15,11 @@ the segment narration plus ``manim_scene_generation`` source snippets so the image model sees the documentation, not only a short caption. Prompts that share no documented terms fail closed (same check as ``validate`` / -``image_prompt_alignment``). ``docgen manim`` then loads the asset via the -``_image`` helper in ``scenes.py``. +``image_prompt_alignment``). After the PNG is written, **OCR** rejects +invented on-image labels and a **vision review** (OpenAI / Grok / Claude) +checks the pixels against the same corpus (``image_generation.align_review``). +A failed review retries once with the critique, then deletes the asset. +``docgen manim`` then loads the asset via the ``_image`` helper in ``scenes.py``. """ from __future__ import annotations @@ -26,8 +29,14 @@ from pathlib import Path from typing import TYPE_CHECKING, Any, Callable +from docgen.image_align import ( + ImageReviewResult, + ocr_image_text, + review_image_against_docs, +) from docgen.openai_retry import call_with_rate_limit_retries from docgen.scene_spec import ( + image_ocr_alignment_violations, image_prompt_alignment_violations, iter_image_elements, load_scene_spec, @@ -227,6 +236,8 @@ def generate_images_for_spec( model_override: str | None = None, size_override: str | None = None, image_fn: Callable[[str], bytes] | None = None, + review_fn: Callable[..., ImageReviewResult] | None = None, + ocr_fn: Callable[[Path], str | None] | None = None, ) -> list[ImageAssetResult]: """Generate missing image assets referenced by one ``*.scene.yaml``. @@ -234,7 +245,9 @@ def generate_images_for_spec( missing **and** has no ``prompt`` fails loud — either commit the file or give the toolchain a prompt to generate it from. - ``image_fn`` is an injection point for tests (prompt → PNG bytes). + ``image_fn`` / ``review_fn`` / ``ocr_fn`` are injection points for tests. + Live vision review runs when ``align_review`` is on and ``image_fn`` is + not injected (or ``review_fn`` is provided). """ spec = load_scene_spec(spec_path) elements = iter_image_elements(spec) @@ -244,8 +257,16 @@ def generate_images_for_spec( quality = icfg.get("quality") quality = str(quality).strip() if quality else None align = bool(icfg.get("align_with_docs", True)) + align_review = bool(icfg.get("align_review", True)) + try: + pixel_retries = int(icfg.get("align_review_retries", 1) or 0) + except (TypeError, ValueError): + pixel_retries = 1 + pixel_retries = max(0, pixel_retries) + review_model = str(icfg.get("review_model") or "").strip() style = str(icfg.get("style") or "").strip() or DEFAULT_IMAGE_STYLE corpus = collect_alignment_corpus(cfg, spec) if align else "" + live_review = align and align_review and (review_fn is not None or image_fn is None) if align: issues = image_prompt_alignment_violations(spec, corpus_text=corpus) @@ -288,17 +309,95 @@ def generate_images_for_spec( prompt=p, model=model, size=size, quality=quality, cfg=cfg ) ) - data = fn(effective) - if not data: + critique = "" + kept = False + last_issues: list[str] = [] + for attempt in range(1 + pixel_retries): + to_send = effective + if critique: + to_send = ( + f"{effective}\n\n--- PIXEL REVIEW FAILED ---\n{critique}\n" + "Redraw so the image matches the documented subject. " + "Do not invent labels or extra components." + ) + data = fn(to_send) + if not data: + raise ImageGenerationError( + f"{spec_path}: image element {rel!r} — provider returned empty bytes" + ) + out.parent.mkdir(parents=True, exist_ok=True) + out.write_bytes(data) + last_issues = _pixel_alignment_issues( + out, + relpath=rel, + corpus=corpus, + authored_prompt=prompt, + label=label, + cfg=cfg, + align=align, + live_review=live_review, + review_model=review_model, + review_fn=review_fn, + ocr_fn=ocr_fn, + ) + if not last_issues: + kept = True + break + critique = "\n".join(last_issues) + if not kept: + out.unlink(missing_ok=True) + joined = "\n ".join(last_issues) raise ImageGenerationError( - f"{spec_path}: image element {rel!r} — provider returned empty bytes" + f"{spec_path}: image {rel!r} failed pixel alignment " + f"(OCR / vision review):\n {joined}" ) - out.parent.mkdir(parents=True, exist_ok=True) - out.write_bytes(data) results.append(ImageAssetResult(rel, out, "generated", prompt, effective)) return results +def _pixel_alignment_issues( + path: Path, + *, + relpath: str, + corpus: str, + authored_prompt: str, + label: str, + cfg: "Config", + align: bool, + live_review: bool, + review_model: str, + review_fn: Callable[..., ImageReviewResult] | None, + ocr_fn: Callable[[Path], str | None] | None, +) -> list[str]: + if not align or not corpus.strip(): + return [] + issues: list[str] = [] + scanned = (ocr_fn or ocr_image_text)(path) + if scanned: + issues.extend( + image_ocr_alignment_violations( + scanned, corpus_text=corpus, relpath=relpath + ) + ) + if live_review: + if review_fn is not None: + verdict = review_fn( + path, corpus_text=corpus, authored_prompt=authored_prompt, label=label + ) + else: + verdict = review_image_against_docs( + path, + corpus_text=corpus, + authored_prompt=authored_prompt, + label=label, + cfg=cfg, + model=review_model, + ) + if not verdict.passed: + issues.append(f"{relpath}: vision review FAIL — {verdict.reason}") + return issues + + def spec_files_for_bundle(cfg: "Config") -> list[Path]: """All committed ``animations/specs/*.scene.yaml`` files, sorted.""" specs_dir = cfg.animations_dir / "specs" @@ -311,6 +410,8 @@ def generate_missing_images_for_bundle( cfg: "Config", *, image_fn: Callable[[str], bytes] | None = None, + review_fn: Callable[..., ImageReviewResult] | None = None, + ocr_fn: Callable[[Path], str | None] | None = None, ) -> list[str]: """Generate only **missing** image assets across all bundle specs. @@ -319,7 +420,9 @@ def generate_missing_images_for_bundle( """ msgs: list[str] = [] for spec_path in spec_files_for_bundle(cfg): - for res in generate_images_for_spec(cfg, spec_path, image_fn=image_fn): + for res in generate_images_for_spec( + cfg, spec_path, image_fn=image_fn, review_fn=review_fn, ocr_fn=ocr_fn + ): if res.status == "generated": msgs.append(f"{spec_path.name}: generated {res.relpath}") return msgs diff --git a/src/docgen/init.py b/src/docgen/init.py index 2a9d583..98ed0d6 100644 --- a/src/docgen/init.py +++ b/src/docgen/init.py @@ -339,6 +339,7 @@ def _write_config(plan: InitPlan) -> str: "model": "gpt-image-1", # Cursor/OpenAI Images; override with --model "size": "1536x1024", "align_with_docs": True, # ground prompts in narration/source; fail invented terms + "align_review": True, # vision-review the PNG against the same docs }, "compose": { "ffmpeg_timeout_sec": 300, diff --git a/src/docgen/scene_spec.py b/src/docgen/scene_spec.py index 4af804d..d22fbd2 100644 --- a/src/docgen/scene_spec.py +++ b/src/docgen/scene_spec.py @@ -1397,6 +1397,34 @@ def image_prompt_alignment_violations( return issues +def image_ocr_alignment_violations( + ocr_text: str, + *, + corpus_text: str, + relpath: str = "image", +) -> list[str]: + """Reject OCR tokens that look like invented documented terms. + + Style words and a missing corpus are ignored. A single short OCR ghost + (``<6`` letters) is tolerated; two confident invented tokens, or one + token of length ≥ 6, fail. + """ + corpus = content_tokens(corpus_text) + if not corpus: + return [] + substance = content_tokens(ocr_text) - _IMAGE_STYLE_TOKENS + if not substance: + return [] + invented = {t for t in substance if t not in corpus} + confident = {t for t in invented if len(t) >= 4} + if len(confident) >= 2 or any(len(t) >= 6 for t in confident): + sample = ", ".join(repr(x) for x in sorted(confident)[:6]) + return [ + f"{relpath}: on-image OCR has terms not in narration/source: {sample}" + ] + return [] + + def cluster_subject_beats( sentences: list[str], *, diff --git a/src/docgen/validate.py b/src/docgen/validate.py index ebb8519..0c8b672 100644 --- a/src/docgen/validate.py +++ b/src/docgen/validate.py @@ -363,6 +363,7 @@ def validate_segment( report.checks.append(self._check_manim_scene_lint()) report.checks.append(self._check_subject_beat_coverage(seg_id)) report.checks.append(self._check_image_prompt_alignment(seg_id)) + report.checks.append(self._check_image_asset_alignment(seg_id)) report.checks.append(self._check_scene_assets(seg_id)) return report.to_dict() @@ -373,8 +374,9 @@ def run_pre_push(self) -> None: Missing recordings are reported as warnings, not failures — a project that hasn't generated videos yet should still be pushable. Quality checks on *existing* recordings — including visual-sync (``av_sync``, - ``subject_beat_coverage``, ``image_prompt_alignment``, ``ocr_scan``, - ``layout``, ``freeze_ratio``) — and narration lint are hard failures so + ``subject_beat_coverage``, ``image_prompt_alignment``, + ``image_asset_alignment``, ``ocr_scan``, ``layout``, ``freeze_ratio``) + — and narration lint are hard failures so ``generate-all`` cannot print ``Pipeline complete`` after a desynced mux. """ reports = self.run_all() @@ -703,6 +705,105 @@ def _check_image_prompt_alignment(self, seg_id: str) -> CheckResult: ["Image prompts share documented terms with narration/source"], ) + def _check_image_asset_alignment(self, seg_id: str) -> CheckResult: + """OCR (and optional vision) of generated scene-spec PNGs vs documentation.""" + iaa = self.config.image_asset_alignment_config + if not iaa.get("enabled", True): + return CheckResult( + "image_asset_alignment", + True, + ["validation.image_asset_alignment disabled in config (skipped)"], + ) + + seg_name = self.config.resolve_segment_name(seg_id) + spec_path = self.config.animations_dir / "specs" / f"{seg_name}.scene.yaml" + if not spec_path.is_file(): + return _missing_manim_spec_result("image_asset_alignment", spec_path.name) + + import yaml + + from docgen.image_align import ocr_image_text, review_image_against_docs + from docgen.image_generate import collect_alignment_corpus + from docgen.scene_spec import image_ocr_alignment_violations, iter_image_elements + + try: + raw = yaml.safe_load(spec_path.read_text(encoding="utf-8")) + except (OSError, yaml.YAMLError) as exc: + return CheckResult( + "image_asset_alignment", + False, + [f"could not load {spec_path.name}: {exc}"], + ) + if not isinstance(raw, dict): + return CheckResult( + "image_asset_alignment", + False, + [f"{spec_path.name}: root must be a mapping"], + ) + elements = iter_image_elements(raw) + if not elements: + return CheckResult( + "image_asset_alignment", + True, + ["No image elements (skipped)"], + ) + + corpus = collect_alignment_corpus(self.config, raw) + if not corpus.strip(): + return CheckResult( + "image_asset_alignment", + True, + ["No narration/source corpus yet — image pixels not checked"], + ) + + issues: list[str] = [] + scanned_any = False + for el in elements: + rel = str(el.get("image") or "").strip() + if not rel: + continue + asset = self.config.base_dir / rel + if not asset.is_file(): + continue + scanned_any = True + if iaa.get("ocr", True): + unavail = _tesseract_unavailable_detail() + if unavail: + return CheckResult("image_asset_alignment", False, [unavail]) + ocr_text = ocr_image_text(asset) + if ocr_text is None: + issues.append(f"{rel}: could not OCR image (unreadable file)") + elif ocr_text: + issues.extend( + image_ocr_alignment_violations( + ocr_text, corpus_text=corpus, relpath=rel + ) + ) + if iaa.get("review"): + verdict = review_image_against_docs( + asset, + corpus_text=corpus, + authored_prompt=str(el.get("prompt") or ""), + label=str(el.get("label") or ""), + cfg=self.config, + ) + if not verdict.passed: + issues.append(f"{rel}: vision review FAIL — {verdict.reason}") + + if issues: + return CheckResult("image_asset_alignment", False, issues) + if not scanned_any: + return CheckResult( + "image_asset_alignment", + True, + ["Image assets not on disk yet (skipped) — run `docgen image-generate`"], + ) + return CheckResult( + "image_asset_alignment", + True, + ["Generated image pixels match documented terms"], + ) + def _check_scene_assets(self, seg_id: str) -> CheckResult: """Pre-render stuck / overlap / font / compile-sync gate (no video required).""" sa_cfg = self.config.scene_assets_config diff --git a/src/docgen/yaml_generate.py b/src/docgen/yaml_generate.py index 0e87256..82272ef 100644 --- a/src/docgen/yaml_generate.py +++ b/src/docgen/yaml_generate.py @@ -221,6 +221,7 @@ def merge_defaults( "model": "gpt-image-1", "size": "1536x1024", "align_with_docs": True, + "align_review": True, } changes.append( "image_generation: added model gpt-image-1 (Cursor/OpenAI Images; override with --model)" diff --git a/tests/test_ai_client.py b/tests/test_ai_client.py index aece8ae..e9a7757 100644 --- a/tests/test_ai_client.py +++ b/tests/test_ai_client.py @@ -15,6 +15,7 @@ DEFAULT_GROK_IMAGE_MODEL, GROK_BASE_URL, chat_completion, + chat_completion_with_image, openai_client, resolve_ai_settings, resolve_chat_model, @@ -137,6 +138,41 @@ class _Resp: assert captured["model"] == DEFAULT_GROK_CHAT_MODEL +def test_chat_completion_with_image_sends_data_uri( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setenv("CURSOR_API_KEY", "sk-proj-cursor") + captured: dict = {} + + class _Msg: + content = "PASS\nok" + + class _Choice: + message = _Msg() + + class _Resp: + choices = [_Choice()] + + fake = MagicMock() + fake.chat.completions.create.side_effect = lambda **kw: captured.update(kw) or _Resp() + + with patch("docgen.ai_client.openai_client", return_value=fake): + out = chat_completion_with_image( + system_prompt="sys", + user_message="review this", + image_bytes=b"png-bytes", + media_type="image/png", + model="gpt-4o", + temperature=0.0, + cfg=_cfg(tmp_path, {}), + ) + assert out == "PASS\nok" + content = captured["messages"][1]["content"] + assert content[0]["type"] == "text" + assert content[1]["type"] == "image_url" + assert content[1]["image_url"]["url"].startswith("data:image/png;base64,") + + def test_openai_empty_chat_content_raises( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, clear_ai_env: None ) -> None: @@ -583,6 +619,34 @@ def _http(url: str, *, data: bytes, headers: dict, **_kwargs) -> bytes: assert captured["payload"]["model"] == DEFAULT_ANTHROPIC_CHAT_MODEL +def test_anthropic_vision_chat_posts_image_block( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, clear_ai_env: None +) -> None: + monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-test") + cfg = _cfg(tmp_path, {"ai": {"provider": "anthropic"}}) + captured: dict = {} + + def _http(url: str, *, data: bytes, headers: dict, **_kwargs) -> bytes: + captured["payload"] = json.loads(data.decode()) + return json.dumps({"content": [{"type": "text", "text": "PASS\nok"}]}).encode() + + with patch("docgen.ai_client._http_with_retries", side_effect=_http): + out = chat_completion_with_image( + system_prompt="sys", + user_message="review", + image_bytes=b"png-bytes", + media_type="image/png", + model="claude-sonnet-4-5", + temperature=0.0, + cfg=cfg, + ) + assert out == "PASS\nok" + content = captured["payload"]["messages"][0]["content"] + assert content[0]["type"] == "image" + assert content[0]["source"]["type"] == "base64" + assert content[1]["type"] == "text" + + def test_anthropic_chat_honors_base_url( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, clear_ai_env: None ) -> None: diff --git a/tests/test_config.py b/tests/test_config.py index 217355a..bc88250 100644 --- a/tests/test_config.py +++ b/tests/test_config.py @@ -1137,7 +1137,8 @@ def test_from_yaml_validation_enable_bools_allowed(tmp_path: Path) -> None: " story_end:\n enabled: false\n" " layout:\n check_overlap: false\n" " subject_beat_coverage:\n enabled: false\n" - " image_prompt_alignment:\n enabled: false\n", + " image_prompt_alignment:\n enabled: false\n" + " image_asset_alignment:\n enabled: false\n ocr: true\n review: false\n", encoding="utf-8", ) c = Config.from_yaml(p) @@ -1150,6 +1151,9 @@ def test_from_yaml_validation_enable_bools_allowed(tmp_path: Path) -> None: assert c.layout_config["check_overlap"] is False assert c.subject_beat_coverage_enabled is False assert c.image_prompt_alignment_enabled is False + assert c.image_asset_alignment_config["enabled"] is False + assert c.image_asset_alignment_config["ocr"] is True + assert c.image_asset_alignment_config["review"] is False def test_from_yaml_int_image_prompt_alignment_enabled_raises(tmp_path: Path) -> None: @@ -1165,6 +1169,29 @@ def test_from_yaml_int_image_prompt_alignment_enabled_raises(tmp_path: Path) -> Config.from_yaml(p) +def test_from_yaml_int_image_asset_alignment_enabled_raises(tmp_path: Path) -> None: + p = tmp_path / "docgen.yaml" + p.write_text( + "validation:\n image_asset_alignment:\n enabled: 0\n", + encoding="utf-8", + ) + with pytest.raises( + ConfigError, + match="validation.image_asset_alignment.enabled must be a YAML boolean", + ): + Config.from_yaml(p) + + +def test_from_yaml_int_image_align_review_raises(tmp_path: Path) -> None: + p = tmp_path / "docgen.yaml" + p.write_text("image_generation:\n align_review: 1\n", encoding="utf-8") + with pytest.raises( + ConfigError, + match="image_generation.align_review must be a YAML boolean", + ): + Config.from_yaml(p) + + def test_from_yaml_int_image_align_with_docs_raises(tmp_path: Path) -> None: p = tmp_path / "docgen.yaml" p.write_text("image_generation:\n align_with_docs: 1\n", encoding="utf-8") diff --git a/tests/test_image_align.py b/tests/test_image_align.py new file mode 100644 index 0000000..5842b8a --- /dev/null +++ b/tests/test_image_align.py @@ -0,0 +1,73 @@ +"""Pixel-level image alignment: OCR verdicts and vision PASS/FAIL parsing.""" + +from __future__ import annotations + +from pathlib import Path + +from docgen.image_align import ( + build_review_user_message, + parse_review_verdict, + review_image_against_docs, +) + + +def test_parse_review_verdict_pass_and_fail() -> None: + ok = parse_review_verdict("PASS\nDepicts the checkout service lock.") + assert ok.passed is True + assert "checkout" in ok.reason + bad = parse_review_verdict("FAIL\nShows a WidgetX console instead.") + assert bad.passed is False + assert "WidgetX" in bad.reason + + +def test_parse_review_verdict_unparseable_fails_closed() -> None: + out = parse_review_verdict("looks fine to me") + assert out.passed is False + assert "PASS/FAIL" in out.reason + + +def test_parse_review_verdict_empty_fails_closed() -> None: + out = parse_review_verdict(" ") + assert out.passed is False + + +def test_build_review_user_message_includes_docs() -> None: + msg = build_review_user_message( + corpus_text="The checkout service owns the cart lock.", + authored_prompt="clean diagram of the checkout service", + label="checkout service", + ) + assert "DOCUMENTED SUBJECT" in msg + assert "cart lock" in msg + assert "AUTHORED IMAGE PROMPT" in msg + assert "ON-SCREEN LABEL" in msg + + +def test_review_image_against_docs_uses_chat_fn(tmp_path: Path) -> None: + png = tmp_path / "x.png" + png.write_bytes(b"\x89PNG\r\n\x1a\nnot-a-real-png") + captured: dict = {} + + def _chat(**kwargs: object) -> str: + captured.update(kwargs) + return "PASS\nMatches the checkout service." + + verdict = review_image_against_docs( + png, + corpus_text="The checkout service owns the cart lock.", + authored_prompt="diagram of checkout", + chat_fn=_chat, + ) + assert verdict.passed is True + assert captured["image_bytes"] == png.read_bytes() + assert "checkout service" in str(captured["user_message"]) + + +def test_review_empty_file_fails(tmp_path: Path) -> None: + png = tmp_path / "empty.png" + png.write_bytes(b"") + verdict = review_image_against_docs( + png, corpus_text="checkout service", chat_fn=lambda **_k: "PASS\nok" + ) + assert verdict.passed is False + assert "empty" in verdict.reason diff --git a/tests/test_image_generate.py b/tests/test_image_generate.py index 9f1b916..6e52fb0 100644 --- a/tests/test_image_generate.py +++ b/tests/test_image_generate.py @@ -8,6 +8,7 @@ import yaml from docgen.config import Config +from docgen.image_align import ImageReviewResult from docgen.image_generate import ( DEFAULT_IMAGE_STYLE, ImageGenerationError, @@ -216,6 +217,77 @@ def test_source_docs_ground_prompt_without_narration(tmp_path: Path) -> None: assert "cart lock" in seen[0] +def test_ocr_invented_text_deletes_asset(cfg: Config) -> None: + spec = _write_spec( + cfg.animations_dir / "specs" / "01-x.scene.yaml", + prompt="clean diagram of the bootstrap pipeline", + ) + cfg.narration_dir.mkdir(parents=True, exist_ok=True) + (cfg.narration_dir / "1.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", encoding="utf-8" + ) + with pytest.raises(ImageGenerationError, match="pixel alignment"): + generate_images_for_spec( + cfg, + spec, + image_fn=lambda p: _PNG_BYTES, + ocr_fn=lambda _path: "WidgetX Orchestrator console", + ) + assert not (cfg.base_dir / "images" / "arch.png").exists() + + +def test_vision_fail_retries_then_keeps_pass(cfg: Config) -> None: + spec = _write_spec( + cfg.animations_dir / "specs" / "01-x.scene.yaml", + prompt="clean diagram of the bootstrap pipeline", + ) + cfg.narration_dir.mkdir(parents=True, exist_ok=True) + (cfg.narration_dir / "1.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", encoding="utf-8" + ) + seen: list[str] = [] + reviews = iter( + [ + ImageReviewResult(False, "shows a generic city skyline"), + ImageReviewResult(True, "now shows the bootstrap pipeline"), + ] + ) + + def _review(*_a: object, **_k: object) -> ImageReviewResult: + return next(reviews) + + generate_images_for_spec( + cfg, + spec, + image_fn=lambda p: seen.append(p) or _PNG_BYTES, + review_fn=_review, + ocr_fn=lambda _path: "", + ) + assert len(seen) == 2 + assert "PIXEL REVIEW FAILED" in seen[1] + assert (cfg.base_dir / "images" / "arch.png").read_bytes() == _PNG_BYTES + + +def test_vision_fail_exhausted_deletes_asset(cfg: Config) -> None: + spec = _write_spec( + cfg.animations_dir / "specs" / "01-x.scene.yaml", + prompt="clean diagram of the bootstrap pipeline", + ) + cfg.narration_dir.mkdir(parents=True, exist_ok=True) + (cfg.narration_dir / "1.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", encoding="utf-8" + ) + with pytest.raises(ImageGenerationError, match="vision review FAIL"): + generate_images_for_spec( + cfg, + spec, + image_fn=lambda p: _PNG_BYTES, + review_fn=lambda *_a, **_k: ImageReviewResult(False, "wrong subject"), + ocr_fn=lambda _path: "", + ) + assert not (cfg.base_dir / "images" / "arch.png").exists() + + def test_build_aligned_image_prompt_includes_corpus_and_label() -> None: out = build_aligned_image_prompt( "clean diagram of the bootstrap pipeline", diff --git a/tests/test_scene_spec.py b/tests/test_scene_spec.py index 16cf2a0..f8f00bc 100644 --- a/tests/test_scene_spec.py +++ b/tests/test_scene_spec.py @@ -17,6 +17,7 @@ disk_spec_with_merged_wait_words, cluster_subject_beats, count_spec_labels, + image_ocr_alignment_violations, image_prompt_alignment_violations, iter_paced_label_anchors, last_paced_reveal_time, @@ -603,6 +604,21 @@ def test_image_prompt_alignment_requires_documented_terms() -> None: assert image_prompt_alignment_violations(spec, corpus_text="") == [] +def test_image_ocr_alignment_flags_invented_on_image_text() -> None: + narr = "The bootstrap pipeline seeds the cluster." + assert image_ocr_alignment_violations( + "bootstrap pipeline", corpus_text=narr, relpath="images/arch.png" + ) == [] + assert image_ocr_alignment_violations("", corpus_text=narr) == [] + issues = image_ocr_alignment_violations( + "WidgetX Orchestrator console", + corpus_text=narr, + relpath="images/arch.png", + ) + assert issues + assert any("OCR" in msg for msg in issues) + + def test_subject_beat_coverage_allows_dwell_rejects_missed_topics() -> None: narr = ( "The bootstrap pipeline seeds the cluster. " diff --git a/tests/test_validate.py b/tests/test_validate.py index 6395904..0ca1dd0 100644 --- a/tests/test_validate.py +++ b/tests/test_validate.py @@ -717,6 +717,64 @@ def test_validate_passes_grounded_image_prompt(self, cfg_dir: Path) -> None: check = next(c for c in report["checks"] if c["name"] == "image_prompt_alignment") assert check["passed"], check["details"] + def test_validate_flags_invented_ocr_on_asset(self, cfg_dir: Path, monkeypatch) -> None: + cfg_raw = yaml.safe_load((cfg_dir / "docgen.yaml").read_text(encoding="utf-8")) + cfg_raw["visual_map"]["01"] = {"type": "manim", "source": "Scene01.mp4"} + cfg_raw.setdefault("segment_names", {})["01"] = "01-test" + (cfg_dir / "docgen.yaml").write_text(yaml.dump(cfg_raw), encoding="utf-8") + (cfg_dir / "narration" / "01-test.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", + encoding="utf-8", + ) + specs = cfg_dir / "animations" / "specs" + specs.mkdir(parents=True, exist_ok=True) + (specs / "01-test.scene.yaml").write_text( + yaml.dump( + { + "segment_id": "01", + "class_name": "DemoScene", + "title": {"text": "T", "font_size": 36, "color": "C_WHITE"}, + "rows": [ + { + "run_time": 1.0, + "boxes": [ + { + "label": "bootstrap pipeline", + "color": "C_ORANGE", + "width": 4.0, + "height": 1.0, + "font_size": 18, + }, + { + "image": "images/arch.png", + "width": 4.0, + "height": 2.0, + "prompt": "clean diagram of the bootstrap pipeline", + }, + ], + } + ], + } + ), + encoding="utf-8", + ) + asset = cfg_dir / "images" / "arch.png" + asset.parent.mkdir(parents=True, exist_ok=True) + asset.write_bytes(b"\x89PNG\r\n\x1a\nfake") + monkeypatch.setattr( + "docgen.image_align.ocr_image_text", + lambda _path: "WidgetX Orchestrator console", + ) + monkeypatch.setattr( + "docgen.validate._tesseract_unavailable_detail", + lambda: None, + ) + config = Config.from_yaml(cfg_dir / "docgen.yaml") + report = Validator(config).validate_segment("01") + check = next(c for c in report["checks"] if c["name"] == "image_asset_alignment") + assert not check["passed"] + assert any("OCR" in d for d in check["details"]) + # ── ffprobe JSON probes honor returncode ────────────────────────────── @@ -781,7 +839,7 @@ def test_drift_pass_when_ffprobe_succeeds(self, config, tmp_path, monkeypatch): @pytest.mark.parametrize( "check_name", - ("av_sync", "subject_beat_coverage", "image_prompt_alignment", "ocr_scan", "layout", "freeze_ratio"), + ("av_sync", "subject_beat_coverage", "image_prompt_alignment", "image_asset_alignment", "ocr_scan", "layout", "freeze_ratio"), ) def test_run_pre_push_visual_sync_fail_is_hard(check_name: str, capsys) -> None: """Visual-sync FAILs must be FAIL + SystemExit, not WARN (leftover #2).""" diff --git a/tests/test_validate_timing_sync.py b/tests/test_validate_timing_sync.py index 26e264e..eb75544 100644 --- a/tests/test_validate_timing_sync.py +++ b/tests/test_validate_timing_sync.py @@ -437,7 +437,7 @@ def _write_paced_hand_authored_scenes(cfg: Config) -> None: @pytest.mark.parametrize( "check_name", - ("story_end", "subject_beat_coverage", "image_prompt_alignment"), + ("story_end", "subject_beat_coverage", "image_prompt_alignment", "image_asset_alignment"), ) def test_hand_authored_paced_manim_without_spec_fails(cfg, check_name: str) -> None: """Manim + paced timed_play + no *.scene.yaml must fail, not skip-PASS.""" @@ -451,6 +451,8 @@ def test_hand_authored_paced_manim_without_spec_fails(cfg, check_name: str) -> N check = v._check_story_end("01") elif check_name == "image_prompt_alignment": check = v._check_image_prompt_alignment("01") + elif check_name == "image_asset_alignment": + check = v._check_image_asset_alignment("01") else: check = v._check_subject_beat_coverage("01") assert check.name == check_name diff --git a/tests/test_yaml_generate.py b/tests/test_yaml_generate.py index 5501f5f..0bf7781 100644 --- a/tests/test_yaml_generate.py +++ b/tests/test_yaml_generate.py @@ -86,6 +86,7 @@ def test_merge_defaults_adds_archive_exclude(tmp_path: Path) -> None: assert raw["image_generation"]["model"] == "gpt-image-1" assert raw["image_generation"]["size"] == "1536x1024" assert raw["image_generation"]["align_with_docs"] is True + assert raw["image_generation"]["align_review"] is True def test_merge_defaults_idempotent_archive(tmp_path: Path) -> None: From 287a2120d258388198fe67abd1d621d0aa199160 Mon Sep 17 00:00:00 2001 From: John Menke Date: Sun, 27 Sep 2026 14:30:57 -0400 Subject: [PATCH 04/17] Fail CI when hotspots, dead code, or mutation survivors grow (#174) * Add maintainability ratchets that fail when hotspots, dead code, or survivors grow. Keep the existing pull-request complexity diff. Hotspots fail only when a changed Python file is both complex and frequently changed. Vulture and mutmut compare to committed baselines and do not rewrite them. Co-authored-by: Cursor * Run path_filters mutation against its own tests. The CI mutmut job was executing the full suite, including a narration test that calls the live API. --------- Co-authored-by: Cursor --- .github/workflows/ci.yml | 51 +++++- .gitignore | 1 + config/mutation-path-filters-baseline.json | 12 ++ config/vulture-baseline.txt | 8 + pyproject.toml | 26 ++++ scripts/check-hotspots.py | 171 +++++++++++++++++++++ scripts/check-mutation-score.py | 52 +++++++ scripts/check-vulture.py | 95 ++++++++++++ scripts/test-check-hotspots.sh | 72 +++++++++ scripts/test-check-mutation.sh | 35 +++++ scripts/test-check-vulture.sh | 41 +++++ tests/test_consumer_import_fitness.py | 49 ++++++ 12 files changed, 612 insertions(+), 1 deletion(-) create mode 100644 config/mutation-path-filters-baseline.json create mode 100644 config/vulture-baseline.txt create mode 100644 scripts/check-hotspots.py create mode 100644 scripts/check-mutation-score.py create mode 100644 scripts/check-vulture.py create mode 100644 scripts/test-check-hotspots.sh create mode 100644 scripts/test-check-mutation.sh create mode 100644 scripts/test-check-vulture.sh create mode 100644 tests/test_consumer_import_fitness.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 6d7d1b0..c2364ba 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -20,12 +20,22 @@ jobs: - uses: actions/setup-python@v6 with: python-version: "3.12" - - run: pip install ruff lizard==1.24.0 + - run: pip install ruff lizard==1.24.0 vulture==2.16 - run: ruff check src/ tests/ - name: Complexity proving test run: bash scripts/test-check-complexity.sh - name: PR-diff complexity run: python3 scripts/check-complexity.py --base origin/main --paths src + - name: Hotspot proving test + run: bash scripts/test-check-hotspots.sh + - name: PR-diff hotspots + run: python3 scripts/check-hotspots.py --base origin/main --paths src + - name: Vulture proving test + run: bash scripts/test-check-vulture.sh + - name: Vulture baseline + run: python3 scripts/check-vulture.py src tests + - name: Mutation score proving test + run: bash scripts/test-check-mutation.sh unit: runs-on: ubuntu-latest @@ -93,3 +103,42 @@ jobs: - run: pip install --no-cache-dir "." - name: Validate scratch init bundle run: bash scripts/ci-validate-scratch-bundle.sh + + # mutmut on src/docgen/path_filters.py only. The job reads the committed + # survived ceiling and does not rewrite it. + mutation: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v6 + - uses: actions/setup-python@v6 + with: + python-version: "3.12" + - name: Install system dependencies + run: | + set -euo pipefail + attempt=1 + max_attempts=4 + backoff=4 + while [ "$attempt" -le "$max_attempts" ]; do + echo "apt attempt $attempt/$max_attempts" + if timeout 600s sudo apt-get -o Acquire::Retries=3 update \ + && timeout 600s sudo apt-get -o Acquire::Retries=3 install -y --no-install-recommends ffmpeg tesseract-ocr; then + exit 0 + fi + if [ "$attempt" -eq "$max_attempts" ]; then + echo "apt failed after $max_attempts attempts" + exit 1 + fi + echo "apt failed, retrying in ${backoff}s..." + sleep "$backoff" + backoff=$((backoff * 2)) + attempt=$((attempt + 1)) + done + - run: pip install --no-cache-dir ".[dev]" "mutmut==3.8.0" + - name: Mutation score for path_filters.py + env: + PYTHONPATH: ${{ github.workspace }}/src + run: | + mutmut run --max-children 1 + mutmut export-cicd-stats + python3 scripts/check-mutation-score.py diff --git a/.gitignore b/.gitignore index 1ceadc8..5178bf9 100644 --- a/.gitignore +++ b/.gitignore @@ -32,3 +32,4 @@ tests/.bin-cache/ htmlcov/ .mypy_cache/ .ruff_cache/ +mutants/ diff --git a/config/mutation-path-filters-baseline.json b/config/mutation-path-filters-baseline.json new file mode 100644 index 0000000..5b993ff --- /dev/null +++ b/config/mutation-path-filters-baseline.json @@ -0,0 +1,12 @@ +{ + "module": "src/docgen/path_filters.py", + "killed": 11, + "survived": 0, + "total": 11, + "no_tests": 0, + "skipped": 0, + "suspicious": 0, + "timeout": 0, + "check_was_interrupted_by_user": 0, + "segfault": 0 +} diff --git a/config/vulture-baseline.txt b/config/vulture-baseline.txt new file mode 100644 index 0000000..be7fa31 --- /dev/null +++ b/config/vulture-baseline.txt @@ -0,0 +1,8 @@ +# High-confidence vulture findings (vulture 2.16, min-confidence 80). +# Each line is: +# check-vulture.py fails when a count rises or a new key appears. +# CI reads this file and must not rewrite it. +1 src/docgen/cli.py unused variable 'param' +2 src/docgen/scene_clock_harness.py unused variable 'other' +1 src/docgen/scene_spec_generate.py unused variable 'source_snippets' +20 tests/test_ai_client.py unused variable 'clear_ai_env' diff --git a/pyproject.toml b/pyproject.toml index 4bd52c9..02ec56d 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -66,3 +66,29 @@ target-version = "py310" # selection that fails the existing codebase; keep CI on the historical E/F gate. [tool.ruff.lint] select = ["E4", "E7", "E9", "F"] + +# One module. validate.py's generated mutant module is about 4.6MB and the +# runner was OOM-killed above 12G, so that file is not the CI target. +# path_filters.py is the module that can finish. source_paths copies the +# package so imports resolve; only_mutate keeps the run off the rest of src. +[tool.mutmut] +source_paths = ["src"] +only_mutate = ["src/docgen/path_filters.py"] +pytest_add_cli_args_test_selection = ["tests/test_path_filters.py"] +also_copy = [ + ".github", + ".cursor", + "scripts", + "AGENTS.md", + "README.md", + "NOTICE", + "SESSION_NOTES.md", + "packaging", + "docs", + "milestones", + "third_party", + "issues", + "specs-on-hold", + "session-notes", + "config", +] diff --git a/scripts/check-hotspots.py b/scripts/check-hotspots.py new file mode 100644 index 0000000..655d5d0 --- /dev/null +++ b/scripts/check-hotspots.py @@ -0,0 +1,171 @@ +#!/usr/bin/env python3 +"""Hotspot gate. + +Fails when a changed Python file is both complex and in the top change-frequency +set. A change to a quiet file passes, including when that file is complex. A +change to a frequent file that is not complex passes. + +Complex means any function with CCN > 10 or NLOC > 80 (same limits as +check-complexity.py). Frequency is the number of commits on --base that touch +the file, among Python files present at that revision. The top set is the +--top files by that count; ties at the cutoff are included. +""" + +from __future__ import annotations + +import argparse +import csv +import io +import subprocess +import sys +import tempfile +from pathlib import Path + +DEFAULT_CCN = 10 +DEFAULT_NLOC = 80 +DEFAULT_TOP = 10 + + +def git(*args: str, cwd: Path) -> str: + return subprocess.check_output(["git", *args], cwd=cwd, text=True).rstrip("\n") + + +def changed_python_files(repo: Path, base: str, paths: list[str]) -> list[str]: + rels = git( + "diff", + "--name-only", + "--diff-filter=ACMR", + f"{base}...HEAD", + "--", + *paths, + cwd=repo, + ) + files = [] + for line in rels.splitlines(): + line = line.strip() + if line.endswith(".py"): + files.append(line) + return files + + +def python_files_at(repo: Path, rev: str, paths: list[str]) -> set[str]: + listed = git("ls-tree", "-r", "--name-only", rev, "--", *paths, cwd=repo) + return {line.strip() for line in listed.splitlines() if line.strip().endswith(".py")} + + +def commit_counts(repo: Path, rev: str, paths: list[str]) -> dict[str, int]: + log = git("log", rev, "--pretty=format:", "--name-only", "--", *paths, cwd=repo) + counts: dict[str, int] = {} + for line in log.splitlines(): + line = line.strip() + if not line.endswith(".py"): + continue + counts[line] = counts.get(line, 0) + 1 + return counts + + +def top_frequency(counts: dict[str, int], top_n: int) -> set[str]: + ranked = sorted(counts.items(), key=lambda item: (-item[1], item[0])) + if not ranked or top_n < 1: + return set() + cutoff = ranked[min(top_n, len(ranked)) - 1][1] + return {path for path, count in ranked if count >= cutoff and count > 0} + + +def lizard_rows(source: str, filename: str) -> list[dict[str, str]]: + if not source.strip(): + return [] + with tempfile.TemporaryDirectory() as tmp: + dest = Path(tmp) / Path(filename).name + dest.write_text(source, encoding="utf-8") + csv_path = Path(tmp) / "out.csv" + subprocess.run( + ["lizard", "-l", "python", "-C", "999", "-L", "999999", "-o", str(csv_path), str(dest)], + check=True, + capture_output=True, + text=True, + ) + text = csv_path.read_text(encoding="utf-8") + rows = [] + for rec in csv.reader(io.StringIO(text)): + if len(rec) < 9: + continue + rows.append({"nloc": rec[0], "ccn": rec[1], "name": rec[7]}) + return rows + + +def function_peaks(repo: Path, rel: str) -> tuple[int, int] | None: + try: + source = git("show", f"HEAD:{rel}", cwd=repo) + except subprocess.CalledProcessError: + return None + rows = lizard_rows(source, rel) + if not rows: + return (0, 0) + max_ccn = max(int(row["ccn"]) for row in rows) + max_nloc = max(int(row["nloc"]) for row in rows) + return max_ccn, max_nloc + + +def hotspot_failures( + repo: Path, + changed: list[str], + hot: set[str], + counts: dict[str, int], + ccn_limit: int, + nloc_limit: int, +) -> list[str]: + failures = [] + for rel in changed: + if rel not in hot: + continue + peaks = function_peaks(repo, rel) + if peaks is None: + continue + ccn, nloc = peaks + if ccn > ccn_limit or nloc > nloc_limit: + failures.append( + f"HOTSPOT {rel} commits={counts.get(rel, 0)} " + f"maxCCN={ccn} maxNLOC={nloc} (limits {ccn_limit}/{nloc_limit})" + ) + return failures + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--base", default="origin/main") + parser.add_argument("--repo", default=".") + parser.add_argument("--ccn", type=int, default=DEFAULT_CCN) + parser.add_argument("--nloc", type=int, default=DEFAULT_NLOC) + parser.add_argument("--top", type=int, default=DEFAULT_TOP) + parser.add_argument("--paths", nargs="*", default=["."]) + args = parser.parse_args() + repo = Path(args.repo).resolve() + try: + changed = changed_python_files(repo, args.base, args.paths) + present = python_files_at(repo, args.base, args.paths) + counts = commit_counts(repo, args.base, args.paths) + except subprocess.CalledProcessError as exc: + print(f"git failed: {exc}", file=sys.stderr) + return 2 + if not changed: + print("check-hotspots: no changed Python files") + return 0 + ranked = {path: count for path, count in counts.items() if path in present} + hot = top_frequency(ranked, args.top) + try: + failures = hotspot_failures(repo, changed, hot, counts, args.ccn, args.nloc) + except subprocess.CalledProcessError as exc: + print(f"lizard failed: {exc}", file=sys.stderr) + return 2 + if failures: + print("check-hotspots: FAIL", file=sys.stderr) + for line in failures: + print(line, file=sys.stderr) + return 1 + print(f"check-hotspots: PASS ({len(changed)} file(s), {len(hot)} hotspot path(s))") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/check-mutation-score.py b/scripts/check-mutation-score.py new file mode 100644 index 0000000..1086e7e --- /dev/null +++ b/scripts/check-mutation-score.py @@ -0,0 +1,52 @@ +#!/usr/bin/env python3 +"""Compare a mutmut export to the committed survived ceiling. + +Reads mutants/mutmut-cicd-stats.json and config/mutation-path-filters-baseline.json. +Fails when survived mutants increase. Does not write the baseline. +""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +DEFAULT_BASELINE = ROOT / "config" / "mutation-path-filters-baseline.json" +DEFAULT_STATS = ROOT / "mutants" / "mutmut-cicd-stats.json" + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--baseline", type=Path, default=DEFAULT_BASELINE) + parser.add_argument("--stats", type=Path, default=DEFAULT_STATS) + args = parser.parse_args() + if not args.baseline.is_file(): + print(f"missing mutation baseline: {args.baseline}", file=sys.stderr) + return 2 + if not args.stats.is_file(): + print(f"missing mutmut stats: {args.stats}", file=sys.stderr) + return 2 + baseline = json.loads(args.baseline.read_text(encoding="utf-8")) + stats = json.loads(args.stats.read_text(encoding="utf-8")) + if int(stats.get("check_was_interrupted_by_user") or 0) > 0: + print("check-mutation-score: FAIL interrupted run", file=sys.stderr) + return 1 + if int(stats.get("total") or 0) <= 0: + print("check-mutation-score: FAIL empty mutmut export", file=sys.stderr) + return 1 + allowed = int(baseline["survived"]) + survived = int(stats["survived"]) + if survived > allowed: + print( + f"check-mutation-score: FAIL survived {allowed} -> {survived}", + file=sys.stderr, + ) + return 1 + print(f"check-mutation-score: PASS survived {survived} (ceiling {allowed})") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/check-vulture.py b/scripts/check-vulture.py new file mode 100644 index 0000000..03f3c74 --- /dev/null +++ b/scripts/check-vulture.py @@ -0,0 +1,95 @@ +#!/usr/bin/env python3 +"""Fail when vulture reports a new high-confidence finding. + +The committed baseline counts today's findings. A higher count, or a finding +that is not listed, fails. Removed findings are allowed. This script never +writes the baseline. +""" + +from __future__ import annotations + +import argparse +import re +import subprocess +import sys +from collections import Counter +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +MIN_CONFIDENCE = "80" +_LINE = re.compile(r"^(.*?):\d+: (.*?)(?: \(\d+% confidence\))?$") + + +def finding_key(line: str) -> str | None: + match = _LINE.match(line.strip()) + if match is None: + return None + return f"{match.group(1)} {match.group(2)}" + + +def parse_baseline(text: str) -> dict[str, int]: + counts: dict[str, int] = {} + for raw in text.splitlines(): + line = raw.strip() + if not line or line.startswith("#"): + continue + count_text, key = line.split(" ", 1) + counts[key] = int(count_text) + return counts + + +def current_counts(paths: list[str], cwd: Path) -> Counter[str]: + proc = subprocess.run( + [sys.executable, "-m", "vulture", *paths, "--min-confidence", MIN_CONFIDENCE], + cwd=cwd, + capture_output=True, + text=True, + check=False, + ) + if proc.returncode not in (0, 3): + sys.stderr.write(proc.stderr) + raise RuntimeError(f"vulture exited {proc.returncode}") + counts: Counter[str] = Counter() + for line in proc.stdout.splitlines(): + key = finding_key(line) + if key is not None: + counts[key] += 1 + return counts + + +def new_findings(current: Counter[str], baseline: dict[str, int]) -> list[str]: + failures = [] + for key, count in sorted(current.items()): + allowed = baseline.get(key, 0) + if count > allowed: + failures.append(f"NEW {key} count {allowed} -> {count}") + return failures + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--baseline", type=Path, default=ROOT / "config" / "vulture-baseline.txt") + parser.add_argument("paths", nargs="*") + args = parser.parse_args() + paths = args.paths or ["src", "tests"] + if not args.baseline.is_file(): + print(f"missing vulture baseline: {args.baseline}", file=sys.stderr) + return 2 + baseline = parse_baseline(args.baseline.read_text(encoding="utf-8")) + try: + current = current_counts(paths, ROOT) + except RuntimeError as exc: + print(exc, file=sys.stderr) + return 2 + failures = new_findings(current, baseline) + if failures: + print("check-vulture: FAIL", file=sys.stderr) + for line in failures: + print(line, file=sys.stderr) + return 1 + print(f"check-vulture: PASS ({len(current)} finding key(s))") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/test-check-hotspots.sh b/scripts/test-check-hotspots.sh new file mode 100644 index 0000000..e8d6d36 --- /dev/null +++ b/scripts/test-check-hotspots.sh @@ -0,0 +1,72 @@ +#!/usr/bin/env bash +# Proving test: a quiet complex file passes; a frequent complex file fails. +set -euo pipefail +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +ROOT="$(cd "${SCRIPT_DIR}/.." && pwd)" +CHECK="${SCRIPT_DIR}/check-hotspots.py" +TMP="$(mktemp -d)" +trap 'rm -rf "$TMP"' EXIT + +commit() { + git -C "$TMP" -c user.email="sa@example.test" -c user.name="SA" -c commit.gpgsign=false commit -qm "$1" +} + +git -C "$TMP" init -q +complex() { + cat <<'PY' +def branchy(x): + if x == 1: return 1 + if x == 2: return 2 + if x == 3: return 3 + if x == 4: return 4 + if x == 5: return 5 + if x == 6: return 6 + if x == 7: return 7 + if x == 8: return 8 + if x == 9: return 9 + if x == 10: return 10 + if x == 11: return 11 + return 0 +PY +} +complex > "$TMP/hot_complex.py" +printf 'def simple(x):\n return x + 1\n' > "$TMP/hot_simple.py" +complex > "$TMP/quiet_complex.py" +git -C "$TMP" add hot_complex.py hot_simple.py quiet_complex.py +commit base +for n in 2 3 4; do + printf '\n# touch %s\n' "$n" >> "$TMP/hot_complex.py" + printf '\n# touch %s\n' "$n" >> "$TMP/hot_simple.py" + git -C "$TMP" add hot_complex.py hot_simple.py + commit "touch $n" +done + +printf '\n# quiet edit\n' >> "$TMP/quiet_complex.py" +git -C "$TMP" add quiet_complex.py +commit "quiet edit" +if ! python3 "$CHECK" --repo "$TMP" --base HEAD~1 --top 2 --paths .; then + echo "expected PASS for a quiet complex file" >&2 + exit 1 +fi + +printf '\n# simple edit\n' >> "$TMP/hot_simple.py" +git -C "$TMP" add hot_simple.py +commit "simple edit" +if ! python3 "$CHECK" --repo "$TMP" --base HEAD~1 --top 2 --paths .; then + echo "expected PASS for a frequent file that is not complex" >&2 + exit 1 +fi + +printf '\n# hotspot edit\n' >> "$TMP/hot_complex.py" +git -C "$TMP" add hot_complex.py +commit "hotspot edit" +if python3 "$CHECK" --repo "$TMP" --base HEAD~1 --top 2 --paths .; then + echo "expected FAIL for a frequent complex file" >&2 + exit 1 +fi + +if ! grep -q 'scripts/check-hotspots.py' "$ROOT/.github/workflows/ci.yml"; then + echo "CI does not run check-hotspots.py" >&2 + exit 1 +fi +echo "test-check-hotspots: PASS" diff --git a/scripts/test-check-mutation.sh b/scripts/test-check-mutation.sh new file mode 100644 index 0000000..d58c0ec --- /dev/null +++ b/scripts/test-check-mutation.sh @@ -0,0 +1,35 @@ +#!/usr/bin/env bash +# Proving test: a higher survived count fails. The checker does not rewrite the baseline. +set -euo pipefail +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +ROOT="$(cd "${SCRIPT_DIR}/.." && pwd)" +CHECK="${SCRIPT_DIR}/check-mutation-score.py" +TMP="$(mktemp -d)" +trap 'rm -rf "$TMP"' EXIT + +BASELINE="$TMP/baseline.json" +STATS="$TMP/stats.json" +printf '%s\n' '{"module":"fixture","survived":2,"total":10}' > "$BASELINE" +printf '%s\n' '{"survived":2,"total":10,"check_was_interrupted_by_user":0}' > "$STATS" +before="$(cksum "$BASELINE")" +if ! python3 "$CHECK" --baseline "$BASELINE" --stats "$STATS"; then + echo "expected PASS when survived stays at the ceiling" >&2 + exit 1 +fi +after="$(cksum "$BASELINE")" +if [ "$before" != "$after" ]; then + echo "checker rewrote the baseline" >&2 + exit 1 +fi + +printf '%s\n' '{"survived":3,"total":11,"check_was_interrupted_by_user":0}' > "$STATS" +if python3 "$CHECK" --baseline "$BASELINE" --stats "$STATS"; then + echo "expected FAIL when survived rises" >&2 + exit 1 +fi + +if grep -n 'mutation-path-filters-baseline.json' "$ROOT/.github/workflows/ci.yml"; then + echo "CI must not rewrite the mutation baseline" >&2 + exit 1 +fi +echo "test-check-mutation: PASS" diff --git a/scripts/test-check-vulture.sh b/scripts/test-check-vulture.sh new file mode 100644 index 0000000..3327cec --- /dev/null +++ b/scripts/test-check-vulture.sh @@ -0,0 +1,41 @@ +#!/usr/bin/env bash +# Proving test: a baselined finding passes; a new finding fails. CI must not rewrite the baseline. +set -euo pipefail +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +ROOT="$(cd "${SCRIPT_DIR}/.." && pwd)" +CHECK="${SCRIPT_DIR}/check-vulture.py" +TMP="$(mktemp -d)" +trap 'rm -rf "$TMP"' EXIT + +OK="$TMP/ok.py" +BAD="$TMP/bad.py" +BASELINE="$TMP/baseline.txt" +printf 'def keep(known_dead):\n return 1\n' > "$OK" +printf "1 %s unused variable 'known_dead'\n" "$OK" > "$BASELINE" +if ! python3 "$CHECK" --baseline "$BASELINE" "$OK"; then + echo "expected PASS for a baselined unused variable" >&2 + exit 1 +fi + +printf 'def keep(known_dead):\n return 1\ndef again(known_dead):\n return 2\n' > "$OK" +if python3 "$CHECK" --baseline "$BASELINE" "$OK"; then + echo "expected FAIL when a baselined finding's count rises" >&2 + exit 1 +fi + +printf 'def keep(known_dead):\n return 1\n' > "$OK" +printf 'def other(new_dead):\n return 1\n' > "$BAD" +if python3 "$CHECK" --baseline "$BASELINE" "$OK" "$BAD"; then + echo "expected FAIL for a finding that is not in the baseline" >&2 + exit 1 +fi + +if grep -n 'make-whitelist' "$ROOT/.github/workflows/ci.yml"; then + echo "CI must not rewrite the vulture baseline" >&2 + exit 1 +fi +if ! grep -q 'scripts/check-vulture.py' "$ROOT/.github/workflows/ci.yml"; then + echo "CI does not run check-vulture.py" >&2 + exit 1 +fi +echo "test-check-vulture: PASS" diff --git a/tests/test_consumer_import_fitness.py b/tests/test_consumer_import_fitness.py new file mode 100644 index 0000000..13a6cca --- /dev/null +++ b/tests/test_consumer_import_fitness.py @@ -0,0 +1,49 @@ +"""Fitness function: src/docgen does not import a consumer repository.""" + +from __future__ import annotations + +import ast +import re +from pathlib import Path + +SRC = Path(__file__).resolve().parents[1] / "src" / "docgen" + +_CONSUMER_NAMES = ("course-builder", "course_builder", "tekton-dag", "tekton_dag") +_HARDCODED_BUNDLE = re.compile(r"(?:^|[\s\"'])/(?:[\w.-]+/){2,}docs/demos\b") + + +def _consumer_ref(text: str) -> str | None: + for name in _CONSUMER_NAMES: + if name in text: + return name + if _HARDCODED_BUNDLE.search(text): + return "hardcoded bundle" + return None + + +def _imported_names(node: ast.AST) -> list[str]: + if isinstance(node, ast.Import): + return [alias.name for alias in node.names] + if isinstance(node, ast.ImportFrom): + return [node.module or "", *[alias.name for alias in node.names]] + if isinstance(node, ast.Constant) and isinstance(node.value, str): + return [node.value] + return [] + + +def _offenders(path: Path, tree: ast.AST) -> list[str]: + found: list[str] = [] + for node in ast.walk(tree): + for name in _imported_names(node): + why = _consumer_ref(name) + if why is not None: + found.append(f"{path}:{getattr(node, 'lineno', 0)} {why}") + return found + + +def test_src_docgen_does_not_import_consumer_repository() -> None: + offenders: list[str] = [] + for path in sorted(SRC.rglob("*.py")): + tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path)) + offenders.extend(_offenders(path, tree)) + assert offenders == [] From f65722a4156009ca5a52a5910d3f2f88f12f7c0b Mon Sep 17 00:00:00 2001 From: John Menke Date: Wed, 7 Oct 2026 18:57:16 -0400 Subject: [PATCH 05/17] Prove timestamps --engine whisper exits 1 when the provider has no speech-to-text. (#178) Co-authored-by: Cursor --- tests/test_timestamps_local.py | 28 ++++++++++++++++++++++++++++ 1 file changed, 28 insertions(+) diff --git a/tests/test_timestamps_local.py b/tests/test_timestamps_local.py index 8279d73..e97ee3a 100644 --- a/tests/test_timestamps_local.py +++ b/tests/test_timestamps_local.py @@ -403,3 +403,31 @@ def test_load_bundle_timing_accepts_numeric_start_end(self, cfg) -> None: out.write_text(json.dumps(payload), encoding="utf-8") assert load_bundle_timing(cfg) == payload + +def test_cli_timestamps_whisper_without_stt_exits_1(tmp_path) -> None: + """Claude chat has no speech-to-text; whisper must fail closed and leave timing.json.""" + from click.testing import CliRunner + + from docgen.cli import main + + raw = { + "ai": {"provider": "anthropic"}, + "segments": {"all": ["01"]}, + "segment_names": {"01": "01-x"}, + } + (tmp_path / "docgen.yaml").write_text(yaml.dump(raw), encoding="utf-8") + timing = tmp_path / "animations" / "timing.json" + timing.parent.mkdir(parents=True) + stale = '{"keep": true}\n' + timing.write_text(stale, encoding="utf-8") + + result = CliRunner().invoke( + main, + ["--config", str(tmp_path / "docgen.yaml"), "timestamps", "--engine", "whisper"], + ) + combined = result.output + result.stderr + assert result.exit_code == 1, combined + assert "speech-to-text" in combined + assert "Traceback" not in combined + assert timing.read_text(encoding="utf-8") == stale + From c697cfe5d3174663a03903b23cf9a26dd3313249 Mon Sep 17 00:00:00 2001 From: John Menke Date: Wed, 7 Oct 2026 18:57:19 -0400 Subject: [PATCH 06/17] Prove docgen tts exits 1 when a listed segment has no narration. (#177) Co-authored-by: Cursor --- tests/test_tts.py | 30 ++++++++++++++++++++++++++++++ 1 file changed, 30 insertions(+) diff --git a/tests/test_tts.py b/tests/test_tts.py index 401ff4b..de234a9 100644 --- a/tests/test_tts.py +++ b/tests/test_tts.py @@ -155,6 +155,36 @@ def _write_empty(*, output_path: Path, **_kwargs: object) -> None: assert not (tmp_path / "audio" / "01-intro.mp3").exists() +def test_cli_tts_missing_narration_exits_1(tmp_path: Path) -> None: + """A listed segment with no narration file must fail closed and leave audio.""" + from click.testing import CliRunner + + from docgen.cli import main + + raw = { + "dirs": {"narration": "narration", "audio": "audio"}, + "segments": {"all": ["01"], "default": ["01"]}, + "segment_names": {"01": "01-intro"}, + } + (tmp_path / "docgen.yaml").write_text(yaml.dump(raw), encoding="utf-8") + (tmp_path / "narration").mkdir() + audio = tmp_path / "audio" / "01-intro.mp3" + audio.parent.mkdir() + stale = b"keep-me" + audio.write_bytes(stale) + + result = CliRunner().invoke( + main, + ["--config", str(tmp_path / "docgen.yaml"), "tts", "--segment", "01"], + ) + combined = result.output + result.stderr + assert result.exit_code == 1, combined + assert "No narration file" in combined + assert "01-intro.md" in combined + assert "Traceback" not in combined + assert audio.read_bytes() == stale + + def test_probe_duration_returns_none_for_missing_file(tmp_path): result = _probe_duration(tmp_path / "nonexistent.mp3") assert result is None From 672f6e5e7466bf83ed160fb1009296f9bc5bc74f Mon Sep 17 00:00:00 2001 From: John Menke Date: Wed, 7 Oct 2026 18:57:23 -0400 Subject: [PATCH 07/17] Prove docgen concat exits 1 when a segment recording is missing. (#176) Co-authored-by: Cursor --- tests/test_concat.py | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/tests/test_concat.py b/tests/test_concat.py index 6e41920..906eb29 100644 --- a/tests/test_concat.py +++ b/tests/test_concat.py @@ -38,6 +38,31 @@ def test_concat_missing_recording_raises_before_ffmpeg(tmp_path: Path) -> None: ConcatBuilder(cfg).build(name="full") +def test_cli_concat_missing_recording_exits_1(tmp_path: Path) -> None: + """A named concat target with a missing segment recording exits 1. + + The stitched file is not created, and the segment that is present stays + byte-for-byte unchanged. + """ + from click.testing import CliRunner + + from docgen.cli import main + + cfg = _cfg(tmp_path, {"full": ["01", "02"]}) + recordings = tmp_path / "recordings" + recordings.mkdir() + present = recordings / "01-a.mp4" + present.write_bytes(b"segment-a") + stitched = recordings / "full.mp4" + runner = CliRunner() + result = runner.invoke(main, ["--config", str(cfg.yaml_path), "concat", "full"]) + assert result.exit_code == 1 + assert "missing recording" in result.output + assert "02" in result.output + assert not stitched.exists() + assert present.read_bytes() == b"segment-a" + + def test_concat_empty_map_is_noop(tmp_path: Path) -> None: cfg = _cfg(tmp_path, {}) ConcatBuilder(cfg).build() From 6610de4d64a80ece1cb3d34bc6f3056c715be2f9 Mon Sep 17 00:00:00 2001 From: John Menke Date: Wed, 7 Oct 2026 18:57:26 -0400 Subject: [PATCH 08/17] Prove docgen benchmark exits 1 when a quality score falls below baseline. (#175) Co-authored-by: Cursor --- tests/test_scene_benchmark.py | 30 ++++++++++++++++++++++++++++++ 1 file changed, 30 insertions(+) diff --git a/tests/test_scene_benchmark.py b/tests/test_scene_benchmark.py index b7fe148..6377b02 100644 --- a/tests/test_scene_benchmark.py +++ b/tests/test_scene_benchmark.py @@ -168,6 +168,36 @@ def test_cli_benchmark_text_and_json(tmp_path: Path) -> None: assert payload["cases"][0]["case_id"] == "early_title" +def test_cli_benchmark_score_regression_exits_1(tmp_path: Path) -> None: + """Consumers and CI depend on exit 1 when a quality case falls below baseline.""" + baseline = tmp_path / "baseline.json" + baseline.write_text( + json.dumps( + { + "version": 1, + "cases": { + "early_title": { + "role": "quality", + "wait_skips": 0, + "defect_points": 0, + "quality_points": 16, + "mid_hold_pulses": 4, + "score": 101, + } + }, + } + ), + encoding="utf-8", + ) + result = CliRunner().invoke( + main, + ["benchmark", "--case", "early_title", "--baseline", str(baseline)], + ) + assert result.exit_code == 1, result.output + assert "regressions vs baseline:" in result.output + assert "early_title: score 101 → 100" in result.output + + def test_packaged_baseline_exists() -> None: path = default_baseline_path() assert path.is_file() From 94177d5afebed63d710fca221485e80fe0fdc369 Mon Sep 17 00:00:00 2001 From: John Menke Date: Wed, 7 Oct 2026 18:57:31 -0400 Subject: [PATCH 09/17] Fail stale-helper checks when helpers are inlined (#172) * Fail closed when scenes.py inlines or renames _TimedScene helpers. The stale-helper check returned no issues when none of the canonical helpers were top-level defs, so inlined or renamed helpers skipped staleness. Co-authored-by: Cursor * Keep inlined-helper detection without raising cyclomatic complexity. --------- Co-authored-by: Cursor --- src/docgen/scene_asset_validate.py | 81 ++++++++++++++++++++++++++++-- tests/test_scene_asset_validate.py | 34 +++++++++++++ 2 files changed, 112 insertions(+), 3 deletions(-) diff --git a/src/docgen/scene_asset_validate.py b/src/docgen/scene_asset_validate.py index e228fe5..5de31a1 100644 --- a/src/docgen/scene_asset_validate.py +++ b/src/docgen/scene_asset_validate.py @@ -250,8 +250,83 @@ def extract_class_source(scenes_text: str, class_name: str) -> str | None: return None +_CANONICAL_HELPERS = ("_box", "_arrow", "_TimedScene", "_load_timing", "_load_timing_words") +_TIMED_SCENE_METHODS = frozenset({"timed_play", "wait_until_word"}) + + +def _unique(names: list[str]) -> list[str]: + return list(dict.fromkeys(names)) + + +def _note_renamed_timed_scene(node: ast.AST, renamed: list[str]) -> None: + if not isinstance(node, ast.ClassDef) or node.name == "_TimedScene": + return + methods = { + child.name + for child in node.body + if isinstance(child, (ast.FunctionDef, ast.AsyncFunctionDef)) + } + if _TIMED_SCENE_METHODS.intersection(methods): + renamed.append(node.name) + + +def _note_helper_node(node: ast.AST, inlined: list[str], referenced: list[str], renamed: list[str]) -> None: + if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)): + if node.name in _CANONICAL_HELPERS: + inlined.append(node.name) + _note_renamed_timed_scene(node, renamed) + elif isinstance(node, ast.Name) and node.id in _CANONICAL_HELPERS: + referenced.append(node.id) + + +def _helper_issue_parts(inlined: list[str], referenced: list[str], renamed: list[str]) -> list[str]: + parts: list[str] = [] + if inlined: + parts.append("inlined " + ", ".join(_unique(inlined))) + if renamed: + parts.append("renamed _TimedScene-style " + ", ".join(_unique(renamed))) + missing = [name for name in _unique(referenced) if name not in inlined] + if missing: + parts.append("referenced but not top-level " + ", ".join(missing)) + return parts + + +def _inlined_or_renamed_helper_issue(tree: ast.AST) -> str | None: + """Fail closed when canonical helpers are not top-level defs. + + Nested ``_box`` / ``_TimedScene`` bodies, calls or bases that name them, and + classes that carry ``timed_play`` / ``wait_until_word`` under another name + used to skip :func:`helper_needs_refresh` entirely. + """ + inlined: list[str] = [] + referenced: list[str] = [] + renamed: list[str] = [] + for node in ast.walk(tree): + _note_helper_node(node, inlined, referenced, renamed) + if not inlined and not referenced and not renamed: + return None + detail = "; ".join(_helper_issue_parts(inlined, referenced, renamed)) + return ( + "helpers: canonical _box / _arrow / _TimedScene / _load_timing / " + "_load_timing_words are missing or inlined " + f"({detail}) — stale-helper check cannot pass silently; " + "run `docgen scene-compile` to restore top-level helper bodies" + ) + + +def _missing_top_level_helper_issues(tree: ast.AST) -> list[str]: + issue = _inlined_or_renamed_helper_issue(tree) + if issue: + return [issue] + return [] + + def helper_api_violations(scenes_text: str) -> list[str]: - """Stale ``_box`` / ``_arrow`` / ``_TimedScene`` that will mis-render new specs.""" + """Stale ``_box`` / ``_arrow`` / ``_TimedScene`` that will mis-render new specs. + + Helpers must be top-level. Inlined or renamed ``_TimedScene``-style helpers + fail closed with an explicit missing/inlined message. + """ from docgen.manim_scene_support import helper_needs_refresh try: @@ -262,8 +337,8 @@ def helper_api_violations(scenes_text: str) -> list[str]: for node in tree.body: if isinstance(node, (ast.FunctionDef, ast.ClassDef)): defined.add(node.name) - if not defined.intersection({"_box", "_arrow", "_TimedScene", "_load_timing", "_load_timing_words"}): - return [] + if not defined.intersection(set(_CANONICAL_HELPERS)): + return _missing_top_level_helper_issues(tree) issues: list[str] = [] if "MANIM_FONT" not in scenes_text: issues.append( diff --git a/tests/test_scene_asset_validate.py b/tests/test_scene_asset_validate.py index 9f2ef6d..9148642 100644 --- a/tests/test_scene_asset_validate.py +++ b/tests/test_scene_asset_validate.py @@ -263,6 +263,40 @@ def test_helper_api_clean_for_current_bootstrap() -> None: assert helper_api_violations(BOOTSTRAP_HEADER) == [] +def test_helper_api_flags_inlined_or_renamed_helpers() -> None: + inlined = """ +class OverviewScene: + class _TimedScene: + def timed_play(self, *a, run_time=1.0): + pass + def wait_until_word(self, words, index): + return + def construct(self): + _box("Alpha", "#fff") +""" + inlined_issues = helper_api_violations(inlined) + assert any( + "missing or inlined" in issue and "inlined _TimedScene" in issue + for issue in inlined_issues + ) + assert any("_box" in issue for issue in inlined_issues) + + renamed = """ +class SceneClock: + def timed_play(self, *a, run_time=1.0): + pass + def wait_until_word(self, words, index): + return +""" + renamed_issues = helper_api_violations(renamed) + assert any( + "missing or inlined" in issue and "renamed _TimedScene-style SceneClock" in issue + for issue in renamed_issues + ) + + assert helper_api_violations("def render():\n return 1\n") == [] + + def test_compiled_sync_passes_when_scenes_match_compile() -> None: spec = _spec([_box("Alpha", wait_word=0), _box("Beta", wait_word=1)]) words = _wide_words() From d9422bf9c25b70e1dad38d05059fd0f5bc609f02 Mon Sep 17 00:00:00 2001 From: John Menke Date: Wed, 7 Oct 2026 18:57:34 -0400 Subject: [PATCH 10/17] Fail timing checks when ffprobe cannot read a duration (#170) * Fail closed when an ffprobe duration probe fails or returns an unusable duration. Co-authored-by: Cursor * Keep the ffprobe failure path without raising cyclomatic complexity. --------- Co-authored-by: Cursor --- src/docgen/validate.py | 102 +++++++++++++++++++++-------- tests/test_validate_timing_sync.py | 71 ++++++++++++++++++++ 2 files changed, 144 insertions(+), 29 deletions(-) diff --git a/src/docgen/validate.py b/src/docgen/validate.py index e79031b..0046bda 100644 --- a/src/docgen/validate.py +++ b/src/docgen/validate.py @@ -24,6 +24,38 @@ from docgen.config import Config +def _unusable_duration(duration: object) -> bool: + if duration is None or isinstance(duration, bool) or not isinstance(duration, (int, float)): + return True + value = float(duration) + return not math.isfinite(value) or value <= 0 + + +def _ffprobe_launch_error(path: Path, exc: BaseException) -> str: + if isinstance(exc, subprocess.TimeoutExpired): + return f"ffprobe timed out on {path.name}" + return "ffprobe not found in PATH" + + +def _positive_ffprobe_duration(out: subprocess.CompletedProcess[str]) -> float: + if out.returncode != 0: + extra = (out.stderr or "").strip() + msg = f"ffprobe failed (exit {out.returncode})" + if extra: + msg = f"{msg}: {extra[:200]}" + raise RuntimeError(msg) + raw = (out.stdout or "").strip() + try: + duration = float(raw) + except (TypeError, ValueError) as exc: + raise RuntimeError(f"ffprobe duration is not a number: {raw[:80]!r}") from exc + if not math.isfinite(duration) or duration <= 0: + raise RuntimeError( + f"ffprobe duration is not a finite positive number: {raw[:80]!r}" + ) + return duration + + @dataclass class CheckResult: name: str @@ -833,16 +865,9 @@ def _check_timing_sync(self, seg_id: str) -> CheckResult: ) return CheckResult("timing_sync", True, ["Empty timing entry (skipped, non-manim)"]) - audio_dur = self._probe_media_duration(audio) - if audio_dur is None: - return CheckResult( - "timing_sync", - False, - [ - f"cannot probe audio duration for {audio.name} — " - "ffprobe failed; timing_sync cannot compare the mp3 to timing.json" - ], - ) + audio_dur, probe_fail = self._duration_or_fail(audio, "timing_sync") + if probe_fail is not None: + return probe_fail max_tail = float(ts_cfg.get("max_tail_gap_sec", 3.0)) max_overrun = float(ts_cfg.get("max_end_overrun_sec", 1.0)) @@ -947,16 +972,9 @@ def _check_story_end(self, seg_id: str) -> CheckResult: audio = self._find_audio(seg_id) if audio and not _is_lfs_pointer(audio): - audio_end = self._probe_media_duration(audio) - if audio_end is None or audio_end <= 0: - return CheckResult( - "story_end", - False, - [ - f"cannot probe audio duration for {audio.name} — " - "story_end cannot compare last paced reveal to the mp3" - ], - ) + audio_end, probe_fail = self._duration_or_fail(audio, "story_end") + if probe_fail is not None: + return probe_fail end_t = audio_end else: # No local mp3 (or LFS pointer): compare against transcript end only. @@ -1073,22 +1091,48 @@ def _timing_last_end(block: dict[str, Any]) -> float | None: return max(ends) return None + def _duration_or_fail(self, path: Path, check_name: str) -> tuple[float, None] | tuple[None, CheckResult]: + """Duration for *check_name*, or a failed check when the probe cannot be used. + + A non-zero ffprobe exit, timeout, missing binary, or unusable duration + fails the check that asked (``passed=False``). ``None`` is not a skip. + """ + try: + duration = self._probe_media_duration(path) + except RuntimeError as exc: + return None, CheckResult( + check_name, + False, + [f"cannot probe audio duration for {path.name} — {exc}"], + ) + if _unusable_duration(duration): + return None, CheckResult( + check_name, + False, + [ + f"cannot probe audio duration for {path.name} — " + f"ffprobe duration is not a finite positive number: {duration!r}" + ], + ) + return float(duration), None + @staticmethod - def _probe_media_duration(path: Path) -> float | None: + def _probe_media_duration(path: Path) -> float: + """Finite positive duration from ffprobe. + + Raises ``RuntimeError`` on a non-zero exit, timeout, missing ffprobe, + or a duration that is not a finite positive number. Does not return + ``None`` for those failures (callers must fail the check). + """ try: out = subprocess.run( ["ffprobe", "-v", "error", "-show_entries", "format=duration", "-of", "csv=p=0", str(path)], capture_output=True, text=True, timeout=30, ) - except (subprocess.TimeoutExpired, FileNotFoundError): - return None - if out.returncode != 0: - return None - try: - return float(out.stdout.strip()) - except ValueError: - return None + except (subprocess.TimeoutExpired, FileNotFoundError) as exc: + raise RuntimeError(_ffprobe_launch_error(path, exc)) from exc + return _positive_ffprobe_duration(out) # ── Helpers ──────────────────────────────────────────────────────── diff --git a/tests/test_validate_timing_sync.py b/tests/test_validate_timing_sync.py index 2775ca4..835a797 100644 --- a/tests/test_validate_timing_sync.py +++ b/tests/test_validate_timing_sync.py @@ -3,6 +3,7 @@ from __future__ import annotations import json +import subprocess import sys import types from pathlib import Path @@ -212,6 +213,76 @@ def _write_scene_spec(cfg: Config, *, labels: list[str]) -> None: (specs / "01-x.scene.yaml").write_text(yaml.dump(raw), encoding="utf-8") +class _FakeDurationProbe: + def __init__(self, returncode: int, stdout: str, stderr: str = "") -> None: + self.returncode = returncode + self.stdout = stdout + self.stderr = stderr + + +def _install_duration_probe(monkeypatch: pytest.MonkeyPatch, mode: str) -> None: + """Drive the real ``_probe_media_duration`` (do not replace the method).""" + + def _run(*_args: object, **_kwargs: object) -> _FakeDurationProbe: + if mode == "timeout": + raise subprocess.TimeoutExpired(cmd=["ffprobe"], timeout=30) + if mode == "nonzero": + # Leftover stdout looks like a duration that would pass the check. + return _FakeDurationProbe(1, "10.4\n", stderr="Invalid data") + if mode == "nan": + return _FakeDurationProbe(0, "nan\n") + if mode == "inf": + return _FakeDurationProbe(0, "inf\n") + return _FakeDurationProbe(0, "not-a-duration\n") + + monkeypatch.setattr(subprocess, "run", _run) + + +class TestMediaDurationProbeFailClosed: + """ffprobe duration failures fail the check that asked; they are not a skip.""" + + @pytest.mark.parametrize("mode", ("nonzero", "timeout", "invalid", "nan", "inf")) + def test_timing_sync_fails_closed(self, cfg: Config, monkeypatch: pytest.MonkeyPatch, mode: str) -> None: + _write_timing(cfg, last_end=10.0) + _install_duration_probe(monkeypatch, mode) + check = Validator(cfg)._check_timing_sync("01") + assert check.passed is False + assert any("cannot probe audio duration" in detail for detail in check.details) + if mode == "nonzero": + assert any("exit 1" in detail for detail in check.details) + elif mode == "timeout": + assert any("timed out" in detail for detail in check.details) + else: + assert any("ffprobe duration" in detail for detail in check.details) + + @pytest.mark.parametrize("mode", ("nonzero", "timeout", "invalid", "nan", "inf")) + def test_story_end_fails_closed(self, cfg: Config, monkeypatch: pytest.MonkeyPatch, mode: str) -> None: + words = [ + {"word": "Alpha", "start": 2.0, "end": 2.4}, + {"word": "Omega", "start": 80.0, "end": 80.5}, + ] + (cfg.animations_dir / "timing.json").write_text( + json.dumps( + { + "01-x": { + "text": "Alpha Omega", + "words": words, + "segments": [{"start": 0.0, "end": 85.0, "text": "x"}], + } + } + ), + encoding="utf-8", + ) + _write_scene_spec(cfg, labels=["Alpha", "Omega"]) + _install_duration_probe(monkeypatch, mode) + check = Validator(cfg)._check_story_end("01") + assert check.passed is False + assert any("cannot probe audio duration" in detail for detail in check.details) + if mode == "nonzero": + assert any("exit 1" in detail for detail in check.details) + assert not any("skipped" in detail.lower() for detail in check.details) + + class TestStoryEnd: def test_story_finishes_early_fails(self, cfg, monkeypatch) -> None: """Board done at ~10s while audio runs ~100s → story_end hard fail.""" From e7c0a3c16e1e7825e4ba589792ba04f77ebc495e Mon Sep 17 00:00:00 2001 From: John Menke Date: Sat, 26 Sep 2026 20:48:16 -0400 Subject: [PATCH 11/17] Fail closed when a benchmark baseline case is missing from the current scores. Co-authored-by: Cursor --- src/docgen/cli.py | 3 +++ src/docgen/scene_benchmark.py | 20 +++++++++++++++++++- tests/test_scene_benchmark.py | 31 +++++++++++++++++++++++++++++++ 3 files changed, 53 insertions(+), 1 deletion(-) diff --git a/src/docgen/cli.py b/src/docgen/cli.py index 9501b4c..4a4426d 100644 --- a/src/docgen/cli.py +++ b/src/docgen/cli.py @@ -1465,6 +1465,7 @@ def benchmark( holds, title skip, page transitions). """ from docgen.scene_benchmark import ( + baseline_scoped_to_case, compare_to_baseline, default_baseline_path, format_table, @@ -1495,6 +1496,8 @@ def benchmark( written = write_baseline(scores, base_path) click.echo(f"wrote baseline {written}") baseline = load_baseline(base_path) + if case_id: + baseline = baseline_scoped_to_case(baseline, case_id) regressions = compare_to_baseline(scores, baseline) report = scores_as_json(scores, regressions=regressions) if output_path: diff --git a/src/docgen/scene_benchmark.py b/src/docgen/scene_benchmark.py index 7315778..7204d7e 100644 --- a/src/docgen/scene_benchmark.py +++ b/src/docgen/scene_benchmark.py @@ -391,13 +391,29 @@ def write_baseline(scores: list[CaseScore], path: Path | None = None) -> Path: return p +def baseline_scoped_to_case(baseline: dict[str, Any], case_id: str) -> dict[str, Any]: + """Limit a baseline to one case so a filtered run is not a deleted-corpus failure.""" + stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} + scoped = dict(baseline) + scoped["cases"] = {key: value for key, value in stored.items() if key == case_id} + return scoped + + def compare_to_baseline( scores: list[CaseScore], baseline: dict[str, Any], ) -> list[str]: - """Return regression notes. Empty means the run meets or beats the baseline.""" + """Return regression notes. Empty means the run meets or beats the baseline. + + A baseline case id absent from ``scores`` is a failure. Deleting a corpus + case must not leave that committed id unchecked. + """ notes: list[str] = [] stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} + scored_ids = {score.case_id for score in scores} + for case_id in stored: + if case_id not in scored_ids: + notes.append(f"{case_id}: missing from current scores") for score in scores: prev = stored.get(score.case_id) if not isinstance(prev, dict): @@ -482,6 +498,8 @@ def build_benchmark_report( scores = run_benchmark(case_id=case_id) path = baseline_path or default_baseline_path() baseline = load_baseline(path) + if case_id: + baseline = baseline_scoped_to_case(baseline, case_id) regressions = compare_to_baseline(scores, baseline) report = scores_as_json(scores, regressions=regressions) stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} diff --git a/tests/test_scene_benchmark.py b/tests/test_scene_benchmark.py index 6377b02..7f09548 100644 --- a/tests/test_scene_benchmark.py +++ b/tests/test_scene_benchmark.py @@ -11,6 +11,7 @@ from docgen.cli import main from docgen.scene_benchmark import ( BenchmarkCase, + CaseScore, compare_to_baseline, default_baseline_path, format_table, @@ -133,6 +134,36 @@ def test_full_corpus_meets_committed_baseline() -> None: assert "quality average" in table +def _tiny_score(case_id: str) -> CaseScore: + return CaseScore( + case_id=case_id, + title=case_id, + role="quality", + wait_skips=0, + overshoots=0, + hold_idle_violations=0, + cadence_violations=0, + sim_drift=0, + mid_hold_pulses=1, + box_reveals=1, + last_motion_frac=1.0, + audio_end=1.0, + defect_points=0, + quality_points=10, + score=100, + ) + + +def test_compare_flags_baseline_id_missing_from_current_scores() -> None: + """A baseline id dropped from the score list must fail (leftover #17).""" + scores = [_tiny_score("alpha"), _tiny_score("beta")] + baseline = {"version": 1, "cases": {score.case_id: score.snapshot() for score in scores}} + assert compare_to_baseline(scores, baseline) == [] + reduced = [score for score in scores if score.case_id != "beta"] + notes = compare_to_baseline(reduced, baseline) + assert notes == ["beta: missing from current scores"] + + def test_compare_flags_skip_regression() -> None: scores = run_benchmark() dump = load_baseline() From 3460370be4cb1283e2dc68f1df9374a3f293aa29 Mon Sep 17 00:00:00 2001 From: John Menke Date: Sat, 26 Sep 2026 21:26:19 -0400 Subject: [PATCH 12/17] Keep baseline-id checks without raising cyclomatic complexity. --- src/docgen/cli.py | 4 +--- src/docgen/scene_benchmark.py | 24 +++++++++++++++--------- 2 files changed, 16 insertions(+), 12 deletions(-) diff --git a/src/docgen/cli.py b/src/docgen/cli.py index 4a4426d..1f96752 100644 --- a/src/docgen/cli.py +++ b/src/docgen/cli.py @@ -1495,9 +1495,7 @@ def benchmark( raise click.ClickException("--update-baseline requires the full corpus (omit --case)") written = write_baseline(scores, base_path) click.echo(f"wrote baseline {written}") - baseline = load_baseline(base_path) - if case_id: - baseline = baseline_scoped_to_case(baseline, case_id) + baseline = baseline_scoped_to_case(load_baseline(base_path), case_id) regressions = compare_to_baseline(scores, baseline) report = scores_as_json(scores, regressions=regressions) if output_path: diff --git a/src/docgen/scene_benchmark.py b/src/docgen/scene_benchmark.py index 7204d7e..7c545c9 100644 --- a/src/docgen/scene_benchmark.py +++ b/src/docgen/scene_benchmark.py @@ -391,14 +391,26 @@ def write_baseline(scores: list[CaseScore], path: Path | None = None) -> Path: return p -def baseline_scoped_to_case(baseline: dict[str, Any], case_id: str) -> dict[str, Any]: +def baseline_scoped_to_case(baseline: dict[str, Any], case_id: str | None) -> dict[str, Any]: """Limit a baseline to one case so a filtered run is not a deleted-corpus failure.""" + if not case_id: + return baseline stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} scoped = dict(baseline) scoped["cases"] = {key: value for key, value in stored.items() if key == case_id} return scoped +def missing_baseline_case_notes(scores: list[CaseScore], stored: dict[str, Any]) -> list[str]: + """Baseline ids that this run did not score. Deleting a case must fail.""" + scored_ids = {score.case_id for score in scores} + return [ + f"{case_id}: missing from current scores" + for case_id in stored + if case_id not in scored_ids + ] + + def compare_to_baseline( scores: list[CaseScore], baseline: dict[str, Any], @@ -408,12 +420,8 @@ def compare_to_baseline( A baseline case id absent from ``scores`` is a failure. Deleting a corpus case must not leave that committed id unchecked. """ - notes: list[str] = [] stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} - scored_ids = {score.case_id for score in scores} - for case_id in stored: - if case_id not in scored_ids: - notes.append(f"{case_id}: missing from current scores") + notes: list[str] = missing_baseline_case_notes(scores, stored) for score in scores: prev = stored.get(score.case_id) if not isinstance(prev, dict): @@ -497,9 +505,7 @@ def build_benchmark_report( """JSON payload for the CLI, wizard Vue view, and desktop GUI.""" scores = run_benchmark(case_id=case_id) path = baseline_path or default_baseline_path() - baseline = load_baseline(path) - if case_id: - baseline = baseline_scoped_to_case(baseline, case_id) + baseline = baseline_scoped_to_case(load_baseline(path), case_id) regressions = compare_to_baseline(scores, baseline) report = scores_as_json(scores, regressions=regressions) stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} From 3d159859ce1b748a7c2be7f3c8e6b62fa78d52a2 Mon Sep 17 00:00:00 2001 From: John Menke Date: Wed, 7 Oct 2026 19:16:15 -0400 Subject: [PATCH 13/17] Keep the benchmark case filter out of cli.py. The hotspot gate fails any edit to that file, so a filtered run marks its scores and compare_to_baseline scopes the missing-id check. Co-authored-by: Cursor --- src/docgen/cli.py | 3 +-- src/docgen/scene_benchmark.py | 38 ++++++++++++++++++++++++++++++++--- tests/test_scene_benchmark.py | 3 +++ 3 files changed, 39 insertions(+), 5 deletions(-) diff --git a/src/docgen/cli.py b/src/docgen/cli.py index 1f96752..9501b4c 100644 --- a/src/docgen/cli.py +++ b/src/docgen/cli.py @@ -1465,7 +1465,6 @@ def benchmark( holds, title skip, page transitions). """ from docgen.scene_benchmark import ( - baseline_scoped_to_case, compare_to_baseline, default_baseline_path, format_table, @@ -1495,7 +1494,7 @@ def benchmark( raise click.ClickException("--update-baseline requires the full corpus (omit --case)") written = write_baseline(scores, base_path) click.echo(f"wrote baseline {written}") - baseline = baseline_scoped_to_case(load_baseline(base_path), case_id) + baseline = load_baseline(base_path) regressions = compare_to_baseline(scores, baseline) report = scores_as_json(scores, regressions=regressions) if output_path: diff --git a/src/docgen/scene_benchmark.py b/src/docgen/scene_benchmark.py index 7c545c9..cad08ab 100644 --- a/src/docgen/scene_benchmark.py +++ b/src/docgen/scene_benchmark.py @@ -367,7 +367,38 @@ def run_benchmark( if not cases: known = ", ".join(c.id for c in standard_cases()) raise ValueError(f"unknown benchmark case {case_id!r}; known: {known}") - return [score_case(c) for c in cases] + return _mark_case_filter([score_case(c) for c in cases], case_id) + + +def _mark_case_filter(scores: list[CaseScore], case_id: str | None) -> list[CaseScore]: + """Remember an explicit ``--case`` filter so other baseline ids are not deletions.""" + if not case_id: + return scores + for score in scores: + score.filtered_case_id = case_id + return scores + + +def _filtered_case_id(scores: list[CaseScore]) -> str | None: + if not scores: + return None + marked = [getattr(score, "filtered_case_id", None) for score in scores] + if any(item is None for item in marked): + return None + if len(set(marked)) != 1: + return None + return marked[0] + + +def _stored_cases_for_scores( + baseline: dict[str, Any], + scores: list[CaseScore], +) -> dict[str, Any]: + stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} + case_id = _filtered_case_id(scores) + if not case_id: + return stored + return {key: value for key, value in stored.items() if key == case_id} def load_baseline(path: Path | None = None) -> dict[str, Any]: @@ -418,9 +449,10 @@ def compare_to_baseline( """Return regression notes. Empty means the run meets or beats the baseline. A baseline case id absent from ``scores`` is a failure. Deleting a corpus - case must not leave that committed id unchecked. + case must not leave that committed id unchecked. Scores from + ``run_benchmark(case_id=...)`` only check that one id. """ - stored = baseline.get("cases") if isinstance(baseline.get("cases"), dict) else {} + stored = _stored_cases_for_scores(baseline, scores) notes: list[str] = missing_baseline_case_notes(scores, stored) for score in scores: prev = stored.get(score.case_id) diff --git a/tests/test_scene_benchmark.py b/tests/test_scene_benchmark.py index 7f09548..65c271c 100644 --- a/tests/test_scene_benchmark.py +++ b/tests/test_scene_benchmark.py @@ -162,6 +162,9 @@ def test_compare_flags_baseline_id_missing_from_current_scores() -> None: reduced = [score for score in scores if score.case_id != "beta"] notes = compare_to_baseline(reduced, baseline) assert notes == ["beta: missing from current scores"] + kept = next(score for score in scores if score.case_id == "alpha") + kept.filtered_case_id = "alpha" + assert compare_to_baseline([kept], baseline) == [] def test_compare_flags_skip_regression() -> None: From 451764ac0c8b2a0aa01a5306140fc685e6351cd0 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Sat, 26 Sep 2026 23:11:06 +0000 Subject: [PATCH 14/17] Ground image-generate prompts in documentation and fail unaligned artwork. Image elements now share the same fail-closed contract as subject-beat coverage: scene-spec-generate feeds source snippets into the LLM, image prompts must use documented terms, and image-generate wraps the Images API call with narration/source plus a validate gate. Co-authored-by: jmjava --- AGENTS.md | 7 +- README.md | 27 +++++-- src/docgen/cli.py | 3 + src/docgen/config.py | 44 +++++++++++ src/docgen/image_generate.py | 115 +++++++++++++++++++++++++++-- src/docgen/init.py | 1 + src/docgen/scene_spec.py | 58 +++++++++++++++ src/docgen/scene_spec_generate.py | 85 ++++++++++++++++++--- src/docgen/validate.py | 63 +++++++++++++++- src/docgen/yaml_generate.py | 1 + tests/test_config.py | 27 ++++++- tests/test_image_generate.py | 102 +++++++++++++++++++++++++ tests/test_scene_spec.py | 34 +++++++++ tests/test_scene_spec_generate.py | 54 ++++++++++++++ tests/test_validate.py | 97 +++++++++++++++++++++++- tests/test_validate_timing_sync.py | 7 +- tests/test_yaml_generate.py | 1 + 17 files changed, 691 insertions(+), 35 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 4fa24aa..e14875f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -70,13 +70,13 @@ Commands registered on the **`docgen`** CLI include: - **`freeze`** — ``docgen freeze`` builds the **`docgen-gui`** onedir (`pip install 'docgen[packaging]'`). Optional ``--smoke`` runs the binary headless. Do not run a full freeze in routine pytest; set ``DOCGEN_FREEZE_SMOKE=1`` for the optional test. - **`tts`** — text-to-speech for segment files (OpenAI or xAI `/v1/tts`). - **`timestamps`** — word/segment timing (`timing.json`). Default engine **`local`** aligns the known narration text against the mp3 offline (ffmpeg silencedetect, no API); **`--engine whisper`** uses OpenAI whisper-1 or xAI `/v1/stt` when `ai.provider` is grok. Both emit the same Whisper-shaped blocks. Failed ffmpeg silencedetect raises `AlignmentError` (empty stderr is not treated as full-span speech). OpenAI whisper-1 word/segment `start`/`end` must be finite JSON numbers (bool/NaN raise `AIError`). Grok `/v1/stt` rejects empty word tokens and inverted `end < start` intervals (`AIError`). Empty ``segments.all`` raises ``TimestampError`` (same as TTS) and does not leave a stale ``timing.json`` as success. -- **`image-generate`** — render scene-spec **image elements** (`image:` + `prompt:` boxes) via OpenAI Images or xAI Imagine into the bundle (also runs for missing assets inside `generate-all`). +- **`image-generate`** — render scene-spec **image elements** (`image:` + `prompt:` boxes) via OpenAI Images or xAI Imagine into the bundle (also runs for missing assets inside `generate-all`). Default **`align_with_docs`** wraps each prompt with narration + source snippets and fails when the authored prompt shares no documented terms. - **`manim`** — render Manim scenes declared in config. - **`compose`** — mux narration audio with visual sources via ffmpeg. With no segment ids, uses ``segments.all`` (same as ``generate-all``), not ``segments.default``. A ``type: mixed`` row raises ``ComposeError`` if any listed source is missing (no silent subset mux). -- **`validate`** / **`validate --pre-push`** — drift, narration lint, Manim hints, **`timing_sync`**, **`story_end`** (last paced reveal vs audio end; hard fail), **`scene_assets`** (pre-render: stuck-board cadence, frame-budget overlaps, `MANIM_FONT` consistency, stale helpers / stale compiled class including hand-edited generated-region labels and `run_time` — hard fail; also a `generate-all` gate before Manim), **`av_sync`** (hard fail on `--pre-push` / `generate-all`; prefers scene-spec labels as OCR anchors), **`subject_beat_coverage`** (declarative specs vs narration topic beats; hard fail when enabled), and related visual-sync checks (`ocr_scan`, `layout`, `freeze_ratio` — hard fail on `--pre-push` / `generate-all`). Missing tesseract fails `ocr_scan` / `av_sync` / `layout` (not skip-PASS). Missing audio or an LFS pointer fails `timing_sync` and recording media gates (`stream_presence`, `av_drift`, `ocr_scan`, `av_sync`) (not skip-PASS). Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` for `type: manim` (not skip-PASS). +- **`validate`** / **`validate --pre-push`** — drift, narration lint, Manim hints, **`timing_sync`**, **`story_end`** (last paced reveal vs audio end; hard fail), **`scene_assets`** (pre-render: stuck-board cadence, frame-budget overlaps, `MANIM_FONT` consistency, stale helpers / stale compiled class including hand-edited generated-region labels and `run_time` — hard fail; also a `generate-all` gate before Manim), **`av_sync`** (hard fail on `--pre-push` / `generate-all`; prefers scene-spec labels as OCR anchors), **`subject_beat_coverage`** (declarative specs vs narration topic beats; hard fail when enabled), **`image_prompt_alignment`** (image-element prompts must share documented terms with narration/source; hard fail when enabled), and related visual-sync checks (`ocr_scan`, `layout`, `freeze_ratio` — hard fail on `--pre-push` / `generate-all`). Missing tesseract fails `ocr_scan` / `av_sync` / `layout` (not skip-PASS). Missing audio or an LFS pointer fails `timing_sync` and recording media gates (`stream_presence`, `av_drift`, `ocr_scan`, `av_sync`) (not skip-PASS). Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` / `image_prompt_alignment` for `type: manim` (not skip-PASS). - **`lint`** — narration lint helper. - **`narration-generate`** — LLM-assisted narration from hints and repo context; optional **`--revise --revision-notes`** for in-place edits (same contract as the wizard Revise button). -- **`scene-spec-generate`** — LLM emits declarative **`*.scene.yaml`**; enforces frame budget + **subject-beat coverage** (dwell OK; cover topic shifts; reject invented labels). +- **`scene-spec-generate`** — LLM emits declarative **`*.scene.yaml`**; enforces frame budget + **subject-beat coverage** (dwell OK; cover topic shifts; reject invented labels) and **image-prompt alignment** (image `prompt:` must use documented terms from narration/source). - **`scene-compile`** — compile specs into **`scenes.py`** (generated regions only). - **`yaml-generate`** — merge defaults and hint wiring into **`docgen.yaml`**. - **`clean-bundle`** — remove regenerable outputs per policy. @@ -91,6 +91,7 @@ Commands registered on the **`docgen`** CLI include: - **Manim / `scenes.py` (marker blocks):** Fix generators under `src/docgen/**` (`manim_scene_support.py`, `scene_spec.py`, `scene_spec_generate.py`, `validate`, `yaml_generate`, tests). **Do not** patch generated classes inside a consumer's **`animations/scenes.py`** between **`BEGIN/END GENERATED SCENE`** markers; re-run **`scene-spec-generate`** / **`scene-compile --retime`** and **`manim`** instead. Preferred consumer order: narration → TTS → timestamps → scene-spec/compile → Manim → compose. - **Beat sync (fail-closed):** when `timing.json` has words, every story box label must match a spoken phrase (`wait_word`); unmatched labels and leftover LLM indices are rejected. Opt out with ``pace: none``. Legacy row-level ``wait_segment`` is upgraded to ``wait_word`` and written back on ``scene-compile`` (direct ``compile_scene_class`` still rejects leftover ``wait_segment``). Fuzzy containment matching is not used. **`scene-compile` clamps FadeIn / page-fade `run_time` against the next word start** so `_TimedScene._clock` cannot race past waits (issue #66 — do not emit cascading first-board dumps). After a reveal, a **dwell** slot may play `Indicate` / `Circumscribe` when the gap to the next `wait_word` is long enough (also clamped). Long holds emit additional mid-hold pulses (`timed_wait` + emphasis) so the board does not freeze after the first Indicate. Optional box fields: `shape` (rounded/pill/diamond), `reveal` (fade/grow/slide), `emphasis` (none/pulse/ring). Page transitions FadeOut revealed boxes, not the parent `VGroup`. `scene-compile` refreshes stale `_box` / `_arrow` / `_TimedScene` helpers in `scenes.py`. - **Subject-beat coverage:** implemented in `scene_spec.layout_density_violations` / `cluster_subject_beats`; enforced by **`scene-spec-generate`** and **`validate`** (`validation.subject_beat_coverage.enabled`, default true). Not a blind label count. +- **Image-prompt alignment:** implemented in `scene_spec.image_prompt_alignment_violations`; **`scene-spec-generate`** and **`image-generate`** reject prompts that share no documented terms; **`validate`** (`validation.image_prompt_alignment.enabled`, default true) is the same gate. `image-generate` also wraps the Images API prompt with narration/source (`image_generation.align_with_docs`, default true). - **`docgen benchmark` is required** after clock / compile / `_TimedScene` / dwell changes. Pytest string assertions are not a substitute. Do not remove the CI `benchmark` job. See **Required gate** below. - Prefer **stable CLI / library contracts** and **documented exit codes** so CI can depend on them. - **`narration_from_source`:** hints in config + **`docgen narration-generate`** — owner-supplied context paths, not opaque bulk edits to outputs. diff --git a/README.md b/README.md index aec8a80..e6802e1 100644 --- a/README.md +++ b/README.md @@ -55,21 +55,26 @@ If you still need the legacy behaviour, pin a pre-removal commit - **Image assets in Manim scenes** — a scene-spec box may be an **image element** (`image: images/.png` + `prompt:`); `docgen image-generate` renders the prompt via OpenAI Images (default `gpt-image-1`) or xAI Imagine - (`grok-imagine-image-2.0` when `ai.provider` is `grok`). The compiled scene - shows it with the `_image` helper. `generate-all` fills in missing assets - automatically. + (`grok-imagine-image-2.0` when `ai.provider` is `grok`). By default the + authored prompt is **grounded** in the segment narration plus + `manim_scene_generation` source snippets, and prompts that share no + documented terms fail closed (`image_prompt_alignment`). The compiled scene + shows the asset with the `_image` helper. `generate-all` fills in missing + assets automatically. - **ffmpeg composition** — combine narration audio and Manim video into final segments, with a freeze-tail guard. - **Validation** — A/V drift, freeze ratio, OCR error scan, layout, narration lint, Manim scene lint, **timing_sync** (stale `timing.json` vs regenerated mp3 — hard fail), **story_end** (paced visual story finishes long before narration — - hard fail), and **av_sync** (OCR check that scene-spec label anchors appear on - screen near their spoken time — hard fail on `--pre-push` / `generate-all`). + hard fail), **image_prompt_alignment** (image-element `prompt:` must share + documented terms with narration/source — hard fail), and **av_sync** (OCR + check that scene-spec label anchors appear on screen near their spoken time + — hard fail on `--pre-push` / `generate-all`). Missing tesseract fails `ocr_scan` / `av_sync` / `layout` instead of skip-PASS. Missing audio or an LFS pointer fails `timing_sync` and recording media gates (`stream_presence`, `av_drift`, `ocr_scan`, `av_sync`) instead of skip-PASS. - Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` for - `type: manim` instead of skip-PASS. A hand-edited generated-region label + Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` / + `image_prompt_alignment` for `type: manim` instead of skip-PASS. A hand-edited generated-region label or `run_time` in `scenes.py` fails `scene_assets` compile_sync (clock / benchmark still execute a fresh `compile_scene_class`, not the on-disk file). - **GitHub Pages** — auto-generate `index.html`, deploy workflow, LFS rules, @@ -128,7 +133,9 @@ using the Cursor key: image_generation: model: gpt-image-1 # or dall-e-3, gpt-image-1-mini, … size: 1536x1024 + align_with_docs: true # wrap prompts with narration/source; fail invented terms # quality: high + # style: "…" # optional override of the educational-diagram prefix ``` ```bash @@ -241,7 +248,7 @@ docgen --repo /path/to/your-project generate-all | `docgen freeze [--dist DIR] [--smoke]` | PyInstaller onedir for **`docgen-gui` only** (`pip install 'docgen[packaging]'`). Not the full Manim CLI | | `docgen tts [--segment 01] [--dry-run]` | Generate TTS audio | | `docgen timestamps [--engine local\|whisper]` | Extract word/segment timestamps from TTS audio → `timing.json` (default `local`: offline narration-text alignment; `whisper`: OpenAI transcription). Empty `segments.all` is `TimestampError` (stale `timing.json` is not success) | -| `docgen image-generate [--segment 01 \| --all \| --spec PATH] [--force] [--dry-run] [--model …] [--size …]` | Generate scene-spec image assets (`image:` + `prompt:` boxes) via the OpenAI Images API into the bundle | +| `docgen image-generate [--segment 01 \| --all \| --spec PATH] [--force] [--dry-run] [--model …] [--size …]` | Generate scene-spec image assets (`image:` + `prompt:` boxes) via the OpenAI Images API; grounds prompts in narration/source and fails unaligned prompts unless `image_generation.align_with_docs` is false | | `docgen manim [--scene StackDAGScene]` | Render Manim animations | | `docgen compose [01 02 03] [--ffmpeg-timeout 900]` | Compose segments (audio + video). Omit ids to walk `segments.all` (same as `generate-all`); a mapped segment with missing audio/visuals is a hard fail | | `docgen validate [--max-drift 2.75] [--pre-push]` | Run all validation checks | @@ -344,7 +351,9 @@ timestamps: image_generation: # scene-spec image elements (docgen image-generate) model: gpt-image-1 # Cursor/OpenAI Images; Grok remaps gpt-image-* to Imagine size: 1536x1024 + align_with_docs: true # ground prompts in narration/source (default) # quality: high # optional, model-specific + # style: "…" # optional Images-API prefix override manim: quality: 1080p30 # supports 480p15, 720p30, 1080p30, 1080p60, 1440p30, 1440p60, 2160p60 @@ -355,6 +364,8 @@ manim: validation: subject_beat_coverage: enabled: true # scene-spec-generate + validate: cover narration topic beats + image_prompt_alignment: + enabled: true # image-element prompts must use documented terms compose: ffmpeg_timeout_sec: 300 # can also be overridden with: docgen compose --ffmpeg-timeout N diff --git a/src/docgen/cli.py b/src/docgen/cli.py index 9501b4c..49c075b 100644 --- a/src/docgen/cli.py +++ b/src/docgen/cli.py @@ -1142,6 +1142,9 @@ def image_generate_cmd( if res.status == "dry-run": click.echo(f"[image-generate] {target.name}: would generate {res.relpath}") click.echo(f" prompt: {res.prompt}") + if res.effective_prompt and res.effective_prompt != res.prompt: + click.echo(" aligned prompt:") + click.echo(res.effective_prompt) elif res.status == "exists": click.echo(f"[image-generate] {target.name}: {res.relpath} exists (skip; use --force)") else: diff --git a/src/docgen/config.py b/src/docgen/config.py index 9f37b47..184cd44 100644 --- a/src/docgen/config.py +++ b/src/docgen/config.py @@ -331,6 +331,7 @@ def __post_init__(self) -> None: "story_end", "narration_lint", "subject_beat_coverage", + "image_prompt_alignment", ): self._sub_block(validation, nested, label=f"validation.{nested}") # List-valued keys: a string must not be iterated as characters. @@ -485,6 +486,14 @@ def __post_init__(self) -> None: require_yaml_string(ig["size"], label="image_generation.size", source=src) if ig.get("quality") is not None: require_yaml_string(ig["quality"], label="image_generation.quality", source=src) + if ig.get("align_with_docs") is not None: + require_yaml_bool( + ig["align_with_docs"], + label="image_generation.align_with_docs", + source=src, + ) + if ig.get("style") is not None: + require_yaml_string(ig["style"], label="image_generation.style", source=src) ai = self._block("ai") if ai.get("provider") is not None: require_yaml_string(ai["provider"], label="ai.provider", source=src) @@ -766,6 +775,17 @@ def __post_init__(self) -> None: label="validation.subject_beat_coverage.enabled", source=src, ) + ipa = self._sub_block( + validation, + "image_prompt_alignment", + label="validation.image_prompt_alignment", + ) + if ipa.get("enabled") is not None: + require_yaml_bool( + ipa["enabled"], + label="validation.image_prompt_alignment.enabled", + source=src, + ) def _source_label(self) -> str: return self.yaml_path.name if self.yaml_path else "docgen.yaml" @@ -950,10 +970,34 @@ def image_generation_config(self) -> dict[str, Any]: defaults: dict[str, Any] = { "model": "gpt-image-1", "size": "1536x1024", + "align_with_docs": True, } defaults.update(self._block("image_generation")) return defaults + @property + def image_align_with_docs(self) -> bool: + """When true (default), ground image prompts in narration/source and fail unaligned ones.""" + block = self._block("image_generation") + if "align_with_docs" in block: + return bool(block.get("align_with_docs")) + return True + + @property + def image_prompt_alignment_enabled(self) -> bool: + """When true (default), validate + scene-spec-generate enforce image-prompt grounding. + + Config: ``validation.image_prompt_alignment.enabled`` (bool). + """ + block = self._sub_block( + self._block("validation"), + "image_prompt_alignment", + label="validation.image_prompt_alignment", + ) + if "enabled" in block: + return bool(block.get("enabled")) + return True + # -- Manim ----------------------------------------------------------------- @property diff --git a/src/docgen/image_generate.py b/src/docgen/image_generate.py index b41722b..b78a627 100644 --- a/src/docgen/image_generate.py +++ b/src/docgen/image_generate.py @@ -11,7 +11,11 @@ ``docgen image-generate`` scans specs, calls the Images API (OpenAI or xAI Imagine) for elements whose asset is missing (or ``--force``), and writes PNG bytes to -``/``. ``docgen manim`` then loads the asset via the +``/``. By default the authored ``prompt`` is **grounded** in +the segment narration plus ``manim_scene_generation`` source snippets so the +image model sees the documentation, not only a short caption. Prompts that +share no documented terms fail closed (same check as ``validate`` / +``image_prompt_alignment``). ``docgen manim`` then loads the asset via the ``_image`` helper in ``scenes.py``. """ @@ -20,16 +24,28 @@ import base64 from dataclasses import dataclass from pathlib import Path -from typing import TYPE_CHECKING, Callable +from typing import TYPE_CHECKING, Any, Callable from docgen.openai_retry import call_with_rate_limit_retries -from docgen.scene_spec import iter_image_elements, load_scene_spec +from docgen.scene_spec import ( + image_prompt_alignment_violations, + iter_image_elements, + load_scene_spec, +) if TYPE_CHECKING: from docgen.config import Config DEFAULT_IMAGE_MODEL = "gpt-image-1" DEFAULT_IMAGE_SIZE = "1536x1024" +DEFAULT_IMAGE_STYLE = ( + "Clean educational diagram for a documentation video. Flat vector " + "illustration, high contrast, no watermark, no signature, no decorative " + "fake UI or invented product logos. Any readable text must be short ASCII " + "copied from the documented subject. Do not add components that are not " + "named in the documentation." +) +_ALIGN_CORPUS_EXCERPT = 2200 class ImageGenerationError(RuntimeError): @@ -42,6 +58,7 @@ class ImageAssetResult: path: Path status: str # "generated" | "exists" | "dry-run" prompt: str + effective_prompt: str = "" def generate_image_bytes( @@ -128,6 +145,69 @@ def generate_image_bytes( ) +def _excerpt(text: str, limit: int = _ALIGN_CORPUS_EXCERPT) -> str: + collapsed = " ".join((text or "").split()) + if len(collapsed) <= limit: + return collapsed + cut = collapsed[: limit - 1] + if " " in cut: + cut = cut.rsplit(" ", 1)[0] + return cut + "…" + + +def build_aligned_image_prompt( + authored: str, + *, + corpus_text: str = "", + label: str = "", + style: str = DEFAULT_IMAGE_STYLE, +) -> str: + """Wrap an authored scene-spec prompt with documentation + style constraints.""" + parts: list[str] = [(style or "").strip() or DEFAULT_IMAGE_STYLE, ""] + corpus = (corpus_text or "").strip() + if corpus: + parts.append( + "Documented subject (use these terms; do not invent names, logos, " + "or extra components):" + ) + parts.append(_excerpt(corpus)) + parts.append("") + lab = (label or "").strip() + if lab: + parts.append(f"On-screen timing label (must remain accurate): {lab}") + parts.append("") + parts.append("Illustration request:") + parts.append((authored or "").strip()) + return "\n".join(parts).strip() + "\n" + + +def collect_alignment_corpus(cfg: "Config", spec: dict[str, Any]) -> str: + """Narration + scene-generation hints + source snippets for one spec.""" + parts: list[str] = [] + seg_id = str(spec.get("segment_id") or "").strip() + if seg_id: + found = cfg.find_segment_asset(cfg.narration_dir, seg_id, ".md") + if found is not None and found.is_file(): + try: + parts.append(found.read_text(encoding="utf-8")) + except OSError: + pass + from docgen.manim_scene_support import ( + collect_source_snippets, + merged_scene_generation_settings, + ) + + settings = merged_scene_generation_settings(cfg, seg_id) + for h in settings.hints: + if str(h).strip(): + parts.append(str(h).strip()) + for label, text in collect_source_snippets(cfg, settings, extra_paths=[]): + body = str(text or "").strip() + if body: + parts.append(f"{label}\n{body}") + return "\n\n".join(p for p in parts if str(p).strip()) + + def _resolve_asset_path(cfg: "Config", relpath: str) -> Path: p = Path(relpath) if p.is_absolute() or ".." in p.parts: @@ -163,15 +243,36 @@ def generate_images_for_spec( size = (size_override or "").strip() or str(icfg.get("size") or DEFAULT_IMAGE_SIZE) quality = icfg.get("quality") quality = str(quality).strip() if quality else None + align = bool(icfg.get("align_with_docs", True)) + style = str(icfg.get("style") or "").strip() or DEFAULT_IMAGE_STYLE + corpus = collect_alignment_corpus(cfg, spec) if align else "" + + if align: + issues = image_prompt_alignment_violations(spec, corpus_text=corpus) + if issues: + joined = "\n ".join(issues) + raise ImageGenerationError( + f"{spec_path}: image prompt alignment failed — rewrite each " + f"`prompt` so it uses documented terms from narration/source " + f"(or set image_generation.align_with_docs: false):\n {joined}" + ) results: list[ImageAssetResult] = [] for el in elements: rel = str(el["image"]).strip() prompt = str(el.get("prompt") or "").strip() out = _resolve_asset_path(cfg, rel) + label = str(el.get("label") or "").strip() + effective = ( + build_aligned_image_prompt( + prompt, corpus_text=corpus, label=label, style=style + ) + if align and prompt + else prompt + ) if out.is_file() and not force: - results.append(ImageAssetResult(rel, out, "exists", prompt)) + results.append(ImageAssetResult(rel, out, "exists", prompt, effective)) continue if not prompt: raise ImageGenerationError( @@ -179,7 +280,7 @@ def generate_images_for_spec( f"({out}); add the file to the bundle or set a prompt in the spec." ) if dry_run: - results.append(ImageAssetResult(rel, out, "dry-run", prompt)) + results.append(ImageAssetResult(rel, out, "dry-run", prompt, effective)) continue fn = image_fn or ( @@ -187,14 +288,14 @@ def generate_images_for_spec( prompt=p, model=model, size=size, quality=quality, cfg=cfg ) ) - data = fn(prompt) + data = fn(effective) if not data: raise ImageGenerationError( f"{spec_path}: image element {rel!r} — provider returned empty bytes" ) out.parent.mkdir(parents=True, exist_ok=True) out.write_bytes(data) - results.append(ImageAssetResult(rel, out, "generated", prompt)) + results.append(ImageAssetResult(rel, out, "generated", prompt, effective)) return results diff --git a/src/docgen/init.py b/src/docgen/init.py index 8f16208..2a9d583 100644 --- a/src/docgen/init.py +++ b/src/docgen/init.py @@ -338,6 +338,7 @@ def _write_config(plan: InitPlan) -> str: "image_generation": { "model": "gpt-image-1", # Cursor/OpenAI Images; override with --model "size": "1536x1024", + "align_with_docs": True, # ground prompts in narration/source; fail invented terms }, "compose": { "ffmpeg_timeout_sec": 300, diff --git a/src/docgen/scene_spec.py b/src/docgen/scene_spec.py index ed7d94c..4af804d 100644 --- a/src/docgen/scene_spec.py +++ b/src/docgen/scene_spec.py @@ -1339,6 +1339,64 @@ def content_tokens(text: str) -> set[str]: return {w for w in words if len(w) > 2 and w not in _CONTENT_STOPWORDS} +# Visual-style words an image prompt may use without being documented terms. +_IMAGE_STYLE_TOKENS = frozenset( + """ + diagram diagrams illustration illustrations icon icons flat clean vector + isometric cartoon watercolor sketch photo photographic realistic abstract + artwork image images picture pictures visual background foreground style + colored colour color palette high contrast simple minimal educational + documentary video watermark signature logo logos chrome banner poster + scene board rounded pill diamond arrow arrows flow flowchart infographic + thumbnail render rendering drawing draw depict showing shows show + depicting depicts labeled labelled label labels text white black blue + green orange red gray grey dark light bright soft hard wide tall small + large tiny big thin thick line lines box boxes node nodes panel panels + layout grid row rows column columns + """.split() +) + + +def image_prompt_alignment_violations( + spec: dict[str, Any], + *, + corpus_text: str, +) -> list[str]: + """Reject image prompts that share no documented terms with narration/source. + + Style adjectives (``flat``, ``diagram``, ``illustration``, …) are ignored. + A missing corpus (no narration and no source) is a no-op — there is nothing + to align against. Specs with no image elements, or image elements that omit + ``prompt`` (committed assets), also pass. + """ + elements = iter_image_elements(spec) + if not elements: + return [] + corpus = content_tokens(corpus_text) + if not corpus: + return [] + issues: list[str] = [] + for el in elements: + rel = str(el.get("image") or "").strip() or "(unnamed image)" + prompt = str(el.get("prompt") or "").strip() + if not prompt: + continue + toks = content_tokens(prompt) + substance = toks - _IMAGE_STYLE_TOKENS + if not substance: + issues.append( + f"{rel}: image prompt is only visual style (no documented subject terms)" + ) + continue + if not (substance & corpus): + preview = prompt if len(prompt) <= 80 else prompt[:77] + "..." + issues.append( + f"{rel}: image prompt shares no documented terms with " + f"narration/source: {preview!r}" + ) + return issues + + def cluster_subject_beats( sentences: list[str], *, diff --git a/src/docgen/scene_spec_generate.py b/src/docgen/scene_spec_generate.py index 693d07e..73b9410 100644 --- a/src/docgen/scene_spec_generate.py +++ b/src/docgen/scene_spec_generate.py @@ -36,6 +36,7 @@ sanitize_pacing_conflicts, compile_scene_class, cluster_subject_beats, + image_prompt_alignment_violations, layout_budget_violations, layout_density_violations, layout_stack_budget, @@ -90,10 +91,14 @@ may instead be an image element with: - image: bundle-relative asset path, e.g. ``images/.png`` (no absolute paths, no "..") - width / height: positive numbers (frame budget rules above apply; images count like boxes) - - prompt: string — a clear visual description; ``docgen image-generate`` renders it via the OpenAI Images API - - label: optional single word from the narration used as the timing anchor for the reveal + - prompt: string — a clear visual description grounded in the narration **and** SOURCE + DOCUMENTATION; ``docgen image-generate`` renders it and rejects prompts that invent + undocumented product names or share no documented terms + - label: optional spoken phrase from the narration used as the timing anchor for the reveal Image elements must NOT carry ``color`` or ``font_size``. Prefer labeled boxes for diagrams; use images -only for illustrative artwork the hints explicitly request. +only for illustrative artwork the hints explicitly request. Image ``prompt`` text must name +the same concepts as the narration/source (not generic "a diagram" / invented architecture). +Any words you expect to appear *inside* the artwork must be short ASCII copied from the docs. Optional per-box (**Whisper ``words`` only**); omit if unsure — compile fills from each box ``label`` → first transcript match: - wait_word: non-negative int — index into ``timing.json`` → ``words``; that box waits until that token's **start**, then fades in (**one box at a time** within each row). @@ -251,6 +256,22 @@ def build_scene_spec_user_message( if str(h).strip(): parts.append(f"- {str(h).strip()}") + if source_snippets: + parts.append("") + parts.append("--- SOURCE DOCUMENTATION ---") + parts.append( + "Use these excerpts when authoring box labels **and** image prompts. " + "Do not invent product names, APIs, or architecture that is not in this " + "source or the narration." + ) + for label, text in source_snippets: + body = str(text or "").strip() + if not body: + continue + parts.append("") + parts.append(f"### {label}") + parts.append(body) + if reference_scenes: parts.append("") parts.append( @@ -449,6 +470,7 @@ def _parse_and_harden_llm_spec( raw: str, enforce_density: bool, density_slack: int = 0, + corpus_text: str = "", ) -> dict[str, Any]: """Parse YAML, auto-layout, validate schema/budget/(optional) density, compile-lint.""" body = strip_yaml_fences(raw) @@ -499,6 +521,17 @@ def _parse_and_harden_llm_spec( f"segment {seg_id}: scene spec failed subject-beat coverage:\n {joined}\nDraft: {draft}" ) + if getattr(cfg, "image_prompt_alignment_enabled", True): + align_issues = image_prompt_alignment_violations( + merged_spec, corpus_text=corpus_text or narration_text + ) + if align_issues: + draft = _save_draft(cfg, seg_id, body) + joined = "\n ".join(align_issues) + raise SceneGenerationError( + f"segment {seg_id}: image prompt alignment failed:\n {joined}\nDraft: {draft}" + ) + try: _, _ = linted_class_block_from_spec(cfg, merged_spec, timing_key=seg_name) except SceneGenerationError as exc: @@ -536,6 +569,15 @@ def generate_scene_spec( existing = scenes_path.read_text(encoding="utf-8") if scenes_path.exists() else "" reference_scenes = extract_reference_classes(existing) snippets = collect_source_snippets(cfg, settings, extra_paths=extra_paths) + corpus_parts = [narration_text] + for h in list(settings.hints) + list(extra_hints): + if str(h).strip(): + corpus_parts.append(str(h).strip()) + for label, text in snippets: + body = str(text or "").strip() + if body: + corpus_parts.append(f"{label}\n{body}") + corpus_text = "\n\n".join(p for p in corpus_parts if str(p).strip()) system_prompt = scene_spec_system_prompt(cfg, seg_id) user_message = build_scene_spec_user_message( @@ -578,13 +620,23 @@ def generate_scene_spec( for attempt in range(3): msg = user_message if attempt > 0 and last_sparse is not None: - msg = ( - f"{user_message}\n\n--- RETRY: SUBJECT-BEAT COVERAGE FAILED ---\n" - f"{last_sparse}\n" - f"Cover each of the {n_beats} subject beats with a spoken-phrase label. " - "Hold the board across sentences in the same beat; add a new label only " - "when the topic shifts. Do not invent unspoken diagram terms." - ) + err = str(last_sparse) + if "image prompt alignment" in err: + msg = ( + f"{user_message}\n\n--- RETRY: IMAGE PROMPT ALIGNMENT FAILED ---\n" + f"{last_sparse}\n" + "Rewrite each image prompt so it uses documented terms from the " + "narration and SOURCE DOCUMENTATION. Do not invent product names " + "or generic 'a diagram' artwork with no subject." + ) + else: + msg = ( + f"{user_message}\n\n--- RETRY: SUBJECT-BEAT COVERAGE FAILED ---\n" + f"{last_sparse}\n" + f"Cover each of the {n_beats} subject beats with a spoken-phrase label. " + "Hold the board across sentences in the same beat; add a new label only " + "when the topic shifts. Do not invent unspoken diagram terms." + ) try: raw = invoke( system_prompt=system_prompt, @@ -612,12 +664,22 @@ def generate_scene_spec( raw=raw, enforce_density=True, density_slack=0, + corpus_text=corpus_text, ) last_sparse = None break except SceneGenerationError as exc: - if "subject-beat coverage" not in str(exc): + err = str(exc) + retryable = ( + "subject-beat coverage" in err or "image prompt alignment" in err + ) + if not retryable: raise + if "image prompt alignment" in err: + if attempt >= 2: + raise + last_sparse = exc + continue # Near-miss: accept without another LLM call when close enough. try: merged_spec = _parse_and_harden_llm_spec( @@ -630,6 +692,7 @@ def generate_scene_spec( raw=raw, enforce_density=True, density_slack=near_miss_slack, + corpus_text=corpus_text, ) last_sparse = None break diff --git a/src/docgen/validate.py b/src/docgen/validate.py index 0046bda..30ebc83 100644 --- a/src/docgen/validate.py +++ b/src/docgen/validate.py @@ -394,6 +394,7 @@ def validate_segment( if self.config.visual_map.get(seg_id, {}).get("type") == "manim": report.checks.append(self._check_manim_scene_lint()) report.checks.append(self._check_subject_beat_coverage(seg_id)) + report.checks.append(self._check_image_prompt_alignment(seg_id)) report.checks.append(self._check_scene_assets(seg_id)) return report.to_dict() @@ -404,9 +405,9 @@ def run_pre_push(self) -> None: Missing recordings are reported as warnings, not failures — a project that hasn't generated videos yet should still be pushable. Quality checks on *existing* recordings — including visual-sync (``av_sync``, - ``subject_beat_coverage``, ``ocr_scan``, ``layout``, ``freeze_ratio``) - — and narration lint are hard failures so ``generate-all`` cannot - print ``Pipeline complete`` after a desynced mux. + ``subject_beat_coverage``, ``image_prompt_alignment``, ``ocr_scan``, + ``layout``, ``freeze_ratio``) — and narration lint are hard failures so + ``generate-all`` cannot print ``Pipeline complete`` after a desynced mux. """ reports = self.run_all() hard_fail = False @@ -678,6 +679,62 @@ def _check_subject_beat_coverage(self, seg_id: str) -> CheckResult: ["Subject beats covered; no invented unspoken labels"], ) + def _check_image_prompt_alignment(self, seg_id: str) -> CheckResult: + """Ensure scene-spec image prompts share documented terms (not generic art).""" + if not self.config.image_prompt_alignment_enabled: + return CheckResult( + "image_prompt_alignment", + True, + ["validation.image_prompt_alignment disabled in config (skipped)"], + ) + + seg_name = self.config.resolve_segment_name(seg_id) + spec_path = self.config.animations_dir / "specs" / f"{seg_name}.scene.yaml" + if not spec_path.is_file(): + return _missing_manim_spec_result("image_prompt_alignment", spec_path.name) + + import yaml + + from docgen.image_generate import collect_alignment_corpus + from docgen.scene_spec import image_prompt_alignment_violations, iter_image_elements + + try: + raw = yaml.safe_load(spec_path.read_text(encoding="utf-8")) + except (OSError, yaml.YAMLError) as exc: + return CheckResult( + "image_prompt_alignment", + False, + [f"could not load {spec_path.name}: {exc}"], + ) + if not isinstance(raw, dict): + return CheckResult( + "image_prompt_alignment", + False, + [f"{spec_path.name}: root must be a mapping"], + ) + if not iter_image_elements(raw): + return CheckResult( + "image_prompt_alignment", + True, + ["No image elements (skipped)"], + ) + + corpus = collect_alignment_corpus(self.config, raw) + issues = image_prompt_alignment_violations(raw, corpus_text=corpus) + if issues: + return CheckResult("image_prompt_alignment", False, issues) + if not corpus.strip(): + return CheckResult( + "image_prompt_alignment", + True, + ["No narration/source corpus yet — image prompts not checked"], + ) + return CheckResult( + "image_prompt_alignment", + True, + ["Image prompts share documented terms with narration/source"], + ) + def _check_scene_assets(self, seg_id: str) -> CheckResult: """Pre-render stuck / overlap / font / compile-sync gate (no video required).""" sa_cfg = self.config.scene_assets_config diff --git a/src/docgen/yaml_generate.py b/src/docgen/yaml_generate.py index c8163ac..0e87256 100644 --- a/src/docgen/yaml_generate.py +++ b/src/docgen/yaml_generate.py @@ -220,6 +220,7 @@ def merge_defaults( raw["image_generation"] = { "model": "gpt-image-1", "size": "1536x1024", + "align_with_docs": True, } changes.append( "image_generation: added model gpt-image-1 (Cursor/OpenAI Images; override with --model)" diff --git a/tests/test_config.py b/tests/test_config.py index cb79082..217355a 100644 --- a/tests/test_config.py +++ b/tests/test_config.py @@ -1136,7 +1136,8 @@ def test_from_yaml_validation_enable_bools_allowed(tmp_path: Path) -> None: " scene_assets:\n enabled: true\n" " story_end:\n enabled: false\n" " layout:\n check_overlap: false\n" - " subject_beat_coverage:\n enabled: false\n", + " subject_beat_coverage:\n enabled: false\n" + " image_prompt_alignment:\n enabled: false\n", encoding="utf-8", ) c = Config.from_yaml(p) @@ -1148,3 +1149,27 @@ def test_from_yaml_validation_enable_bools_allowed(tmp_path: Path) -> None: assert c.story_end_config["enabled"] is False assert c.layout_config["check_overlap"] is False assert c.subject_beat_coverage_enabled is False + assert c.image_prompt_alignment_enabled is False + + +def test_from_yaml_int_image_prompt_alignment_enabled_raises(tmp_path: Path) -> None: + p = tmp_path / "docgen.yaml" + p.write_text( + "validation:\n image_prompt_alignment:\n enabled: 0\n", + encoding="utf-8", + ) + with pytest.raises( + ConfigError, + match="validation.image_prompt_alignment.enabled must be a YAML boolean", + ): + Config.from_yaml(p) + + +def test_from_yaml_int_image_align_with_docs_raises(tmp_path: Path) -> None: + p = tmp_path / "docgen.yaml" + p.write_text("image_generation:\n align_with_docs: 1\n", encoding="utf-8") + with pytest.raises( + ConfigError, + match="image_generation.align_with_docs must be a YAML boolean", + ): + Config.from_yaml(p) diff --git a/tests/test_image_generate.py b/tests/test_image_generate.py index ea0df85..9f1b916 100644 --- a/tests/test_image_generate.py +++ b/tests/test_image_generate.py @@ -9,7 +9,9 @@ from docgen.config import Config from docgen.image_generate import ( + DEFAULT_IMAGE_STYLE, ImageGenerationError, + build_aligned_image_prompt, generate_images_for_spec, generate_missing_images_for_bundle, spec_files_for_bundle, @@ -82,6 +84,8 @@ def test_dry_run_reports_without_writing(cfg: Config) -> None: results = generate_images_for_spec(cfg, spec, dry_run=True, image_fn=lambda p: _PNG_BYTES) assert [r.status for r in results] == ["dry-run"] assert results[0].prompt == "a diagram" + assert DEFAULT_IMAGE_STYLE.split(".")[0] in results[0].effective_prompt + assert "Illustration request:" in results[0].effective_prompt assert not (cfg.base_dir / "images" / "arch.png").exists() @@ -125,6 +129,104 @@ def test_no_specs_dir_is_noop(cfg: Config) -> None: assert generate_missing_images_for_bundle(cfg, image_fn=lambda p: _PNG_BYTES) == [] +def test_aligned_prompt_is_what_the_provider_sees(cfg: Config) -> None: + spec = _write_spec( + cfg.animations_dir / "specs" / "01-x.scene.yaml", + prompt="clean diagram of the bootstrap pipeline", + ) + (cfg.narration_dir).mkdir(parents=True, exist_ok=True) + (cfg.narration_dir / "1.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", encoding="utf-8" + ) + seen: list[str] = [] + + def _capture(prompt: str) -> bytes: + seen.append(prompt) + return _PNG_BYTES + + generate_images_for_spec(cfg, spec, image_fn=_capture) + assert seen + assert "bootstrap pipeline" in seen[0] + assert "Documented subject" in seen[0] + assert seen[0] != "clean diagram of the bootstrap pipeline" + + +def test_unaligned_prompt_fails_before_provider(cfg: Config) -> None: + spec = _write_spec( + cfg.animations_dir / "specs" / "01-x.scene.yaml", + prompt="isometric render of the WidgetX orchestrator", + ) + cfg.narration_dir.mkdir(parents=True, exist_ok=True) + (cfg.narration_dir / "1.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", encoding="utf-8" + ) + with pytest.raises(ImageGenerationError, match="image prompt alignment"): + generate_images_for_spec(cfg, spec, image_fn=lambda p: _PNG_BYTES) + assert not (cfg.base_dir / "images" / "arch.png").exists() + + +def test_align_with_docs_false_sends_authored_prompt(tmp_path: Path) -> None: + (tmp_path / "docgen.yaml").write_text( + yaml.dump( + { + "segments": {"all": ["1"]}, + "image_generation": {"align_with_docs": False}, + } + ), + encoding="utf-8", + ) + cfg = Config.from_yaml(tmp_path / "docgen.yaml") + spec = _write_spec( + cfg.animations_dir / "specs" / "01-x.scene.yaml", + prompt="isometric render of the WidgetX orchestrator", + ) + cfg.narration_dir.mkdir(parents=True, exist_ok=True) + (cfg.narration_dir / "1.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", encoding="utf-8" + ) + seen: list[str] = [] + generate_images_for_spec(cfg, spec, image_fn=lambda p: seen.append(p) or _PNG_BYTES) + assert seen == ["isometric render of the WidgetX orchestrator"] + + +def test_source_docs_ground_prompt_without_narration(tmp_path: Path) -> None: + (tmp_path / "docs").mkdir() + (tmp_path / "docs" / "arch.md").write_text( + "The checkout service owns the cart lock.\n", encoding="utf-8" + ) + (tmp_path / "docgen.yaml").write_text( + yaml.dump( + { + "repo_root": ".", + "segments": {"all": ["1"]}, + "manim_scene_generation": {"context": {"paths": ["docs/arch.md"]}}, + } + ), + encoding="utf-8", + ) + cfg = Config.from_yaml(tmp_path / "docgen.yaml") + spec = _write_spec( + cfg.animations_dir / "specs" / "01-x.scene.yaml", + prompt="clean diagram of the checkout service", + ) + seen: list[str] = [] + generate_images_for_spec(cfg, spec, image_fn=lambda p: seen.append(p) or _PNG_BYTES) + assert seen + assert "checkout service" in seen[0] + assert "cart lock" in seen[0] + + +def test_build_aligned_image_prompt_includes_corpus_and_label() -> None: + out = build_aligned_image_prompt( + "clean diagram of the bootstrap pipeline", + corpus_text="The bootstrap pipeline seeds the cluster.", + label="bootstrap", + ) + assert "bootstrap pipeline" in out + assert "On-screen timing label" in out + assert "Illustration request:" in out + + def test_cli_image_generate_all_fails_when_manim_has_no_specs(tmp_path: Path) -> None: from click.testing import CliRunner diff --git a/tests/test_scene_spec.py b/tests/test_scene_spec.py index 6471196..16cf2a0 100644 --- a/tests/test_scene_spec.py +++ b/tests/test_scene_spec.py @@ -17,6 +17,7 @@ disk_spec_with_merged_wait_words, cluster_subject_beats, count_spec_labels, + image_prompt_alignment_violations, iter_paced_label_anchors, last_paced_reveal_time, layout_budget_violations, @@ -569,6 +570,39 @@ def test_layout_stack_budget_decreases_with_larger_title_font() -> None: assert b_small > b_large +def test_image_prompt_alignment_requires_documented_terms() -> None: + spec = { + "title": {"text": "T", "font_size": 36, "color": "C_WHITE"}, + "rows": [ + { + "run_time": 1.0, + "boxes": [ + { + "image": "images/arch.png", + "width": 4.0, + "height": 2.5, + "prompt": "clean flat diagram of the bootstrap pipeline", + } + ], + } + ], + } + narr = "The bootstrap pipeline seeds the cluster." + assert image_prompt_alignment_violations(spec, corpus_text=narr) == [] + + spec["rows"][0]["boxes"][0]["prompt"] = "a clean flat illustration" + issues = image_prompt_alignment_violations(spec, corpus_text=narr) + assert issues + assert any("only visual style" in msg for msg in issues) + + spec["rows"][0]["boxes"][0]["prompt"] = "isometric render of the WidgetX orchestrator" + issues = image_prompt_alignment_violations(spec, corpus_text=narr) + assert issues + assert any("shares no documented terms" in msg for msg in issues) + + assert image_prompt_alignment_violations(spec, corpus_text="") == [] + + def test_subject_beat_coverage_allows_dwell_rejects_missed_topics() -> None: narr = ( "The bootstrap pipeline seeds the cluster. " diff --git a/tests/test_scene_spec_generate.py b/tests/test_scene_spec_generate.py index 32f0a10..c294c5f 100644 --- a/tests/test_scene_spec_generate.py +++ b/tests/test_scene_spec_generate.py @@ -370,6 +370,60 @@ def test_user_message_includes_computed_layout_stack_budgets() -> None: assert "13.22" in msg # horizontal safe width (FRAME_WIDTH - 1.0) +def test_user_message_includes_source_snippets() -> None: + msg = build_scene_spec_user_message( + seg_id="01", + seg_name="01-x", + class_name="XScene", + narration_text="The bootstrap pipeline seeds the cluster.", + timing_enrichment="(no timing)", + hints=[], + extra_hints=[], + reference_scenes="", + source_snippets=[("docs/architecture.md", "The bootstrap pipeline writes a lockfile.")], + ) + assert "SOURCE DOCUMENTATION" in msg + assert "docs/architecture.md" in msg + assert "writes a lockfile" in msg + assert "Do not invent product names" in msg + + +def test_generate_rejects_unaligned_image_prompt(tmp_path: Path) -> None: + cfg = _bundle(tmp_path) + + def fake_llm(**_kwargs: object) -> str: + return """```yaml +segment_id: "08" +class_name: ExtrasScene +title: + text: "Synthetic" + font_size: 40 + color: C_WHITE +rows: + - run_time: 1.2 + boxes: + - label: "Hello" + color: C_ORANGE + width: 4.0 + height: 1.0 + font_size: 20 + - image: images/widget.png + width: 4.0 + height: 2.0 + prompt: "isometric render of the WidgetX orchestrator" +```""" + + with pytest.raises(SceneGenerationError, match="image prompt alignment"): + generate_scene_spec( + cfg, + "08", + extra_paths=[], + extra_hints=[], + dry_run=False, + llm=fake_llm, + ) + + def test_scene_spec_generate_all_uses_default_when_all_missing(tmp_path: Path) -> None: from click.testing import CliRunner diff --git a/tests/test_validate.py b/tests/test_validate.py index e2fb80e..6395904 100644 --- a/tests/test_validate.py +++ b/tests/test_validate.py @@ -623,6 +623,101 @@ def test_validate_passes_covered_beats(self, cfg_dir: Path) -> None: assert check["passed"], check["details"] +class TestImagePromptAlignmentValidate: + def test_validate_flags_unaligned_image_prompt(self, cfg_dir: Path) -> None: + cfg_raw = yaml.safe_load((cfg_dir / "docgen.yaml").read_text(encoding="utf-8")) + cfg_raw["visual_map"]["01"] = {"type": "manim", "source": "Scene01.mp4"} + cfg_raw.setdefault("segment_names", {})["01"] = "01-test" + (cfg_dir / "docgen.yaml").write_text(yaml.dump(cfg_raw), encoding="utf-8") + (cfg_dir / "narration" / "01-test.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", + encoding="utf-8", + ) + specs = cfg_dir / "animations" / "specs" + specs.mkdir(parents=True, exist_ok=True) + (specs / "01-test.scene.yaml").write_text( + yaml.dump( + { + "segment_id": "01", + "class_name": "DemoScene", + "title": {"text": "T", "font_size": 36, "color": "C_WHITE"}, + "rows": [ + { + "run_time": 1.0, + "boxes": [ + { + "label": "bootstrap pipeline", + "color": "C_ORANGE", + "width": 4.0, + "height": 1.0, + "font_size": 18, + }, + { + "image": "images/arch.png", + "width": 4.0, + "height": 2.0, + "prompt": "isometric render of the WidgetX orchestrator", + }, + ], + } + ], + } + ), + encoding="utf-8", + ) + config = Config.from_yaml(cfg_dir / "docgen.yaml") + report = Validator(config).validate_segment("01") + check = next(c for c in report["checks"] if c["name"] == "image_prompt_alignment") + assert not check["passed"] + assert any("documented terms" in d for d in check["details"]) + + def test_validate_passes_grounded_image_prompt(self, cfg_dir: Path) -> None: + cfg_raw = yaml.safe_load((cfg_dir / "docgen.yaml").read_text(encoding="utf-8")) + cfg_raw["visual_map"]["01"] = {"type": "manim", "source": "Scene01.mp4"} + cfg_raw.setdefault("segment_names", {})["01"] = "01-test" + (cfg_dir / "docgen.yaml").write_text(yaml.dump(cfg_raw), encoding="utf-8") + (cfg_dir / "narration" / "01-test.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", + encoding="utf-8", + ) + specs = cfg_dir / "animations" / "specs" + specs.mkdir(parents=True, exist_ok=True) + (specs / "01-test.scene.yaml").write_text( + yaml.dump( + { + "segment_id": "01", + "class_name": "DemoScene", + "title": {"text": "T", "font_size": 36, "color": "C_WHITE"}, + "rows": [ + { + "run_time": 1.0, + "boxes": [ + { + "label": "bootstrap pipeline", + "color": "C_ORANGE", + "width": 4.0, + "height": 1.0, + "font_size": 18, + }, + { + "image": "images/arch.png", + "width": 4.0, + "height": 2.0, + "prompt": "clean diagram of the bootstrap pipeline", + }, + ], + } + ], + } + ), + encoding="utf-8", + ) + config = Config.from_yaml(cfg_dir / "docgen.yaml") + report = Validator(config).validate_segment("01") + check = next(c for c in report["checks"] if c["name"] == "image_prompt_alignment") + assert check["passed"], check["details"] + + # ── ffprobe JSON probes honor returncode ────────────────────────────── class _FakeProbe: @@ -686,7 +781,7 @@ def test_drift_pass_when_ffprobe_succeeds(self, config, tmp_path, monkeypatch): @pytest.mark.parametrize( "check_name", - ("av_sync", "subject_beat_coverage", "ocr_scan", "layout", "freeze_ratio"), + ("av_sync", "subject_beat_coverage", "image_prompt_alignment", "ocr_scan", "layout", "freeze_ratio"), ) def test_run_pre_push_visual_sync_fail_is_hard(check_name: str, capsys) -> None: """Visual-sync FAILs must be FAIL + SystemExit, not WARN (leftover #2).""" diff --git a/tests/test_validate_timing_sync.py b/tests/test_validate_timing_sync.py index 835a797..235c44d 100644 --- a/tests/test_validate_timing_sync.py +++ b/tests/test_validate_timing_sync.py @@ -506,7 +506,10 @@ def _write_paced_hand_authored_scenes(cfg: Config) -> None: ) -@pytest.mark.parametrize("check_name", ("story_end", "subject_beat_coverage")) +@pytest.mark.parametrize( + "check_name", + ("story_end", "subject_beat_coverage", "image_prompt_alignment"), +) def test_hand_authored_paced_manim_without_spec_fails(cfg, check_name: str) -> None: """Manim + paced timed_play + no *.scene.yaml must fail, not skip-PASS.""" _write_paced_hand_authored_scenes(cfg) @@ -517,6 +520,8 @@ def test_hand_authored_paced_manim_without_spec_fails(cfg, check_name: str) -> N v = Validator(cfg) if check_name == "story_end": check = v._check_story_end("01") + elif check_name == "image_prompt_alignment": + check = v._check_image_prompt_alignment("01") else: check = v._check_subject_beat_coverage("01") assert check.name == check_name diff --git a/tests/test_yaml_generate.py b/tests/test_yaml_generate.py index 8041e4f..5501f5f 100644 --- a/tests/test_yaml_generate.py +++ b/tests/test_yaml_generate.py @@ -85,6 +85,7 @@ def test_merge_defaults_adds_archive_exclude(tmp_path: Path) -> None: assert raw["ai"]["provider"] == "openai" assert raw["image_generation"]["model"] == "gpt-image-1" assert raw["image_generation"]["size"] == "1536x1024" + assert raw["image_generation"]["align_with_docs"] is True def test_merge_defaults_idempotent_archive(tmp_path: Path) -> None: From f7abb4adb14be007717c2c199fa67c4a26cc9e0a Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Sat, 26 Sep 2026 23:12:21 +0000 Subject: [PATCH 15/17] Fix the scene-spec-generate alignment test to use a spoken image label. The image stem was treated as an invented subject-beat label and failed coverage before the new prompt-alignment gate could run. Co-authored-by: jmjava --- tests/test_scene_spec_generate.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tests/test_scene_spec_generate.py b/tests/test_scene_spec_generate.py index c294c5f..199c0d7 100644 --- a/tests/test_scene_spec_generate.py +++ b/tests/test_scene_spec_generate.py @@ -407,9 +407,10 @@ def fake_llm(**_kwargs: object) -> str: width: 4.0 height: 1.0 font_size: 20 - - image: images/widget.png + - image: images/hello.png width: 4.0 height: 2.0 + label: "Hello" prompt: "isometric render of the WidgetX orchestrator" ```""" From 370bda478ce0673bbe58b091cc2a05ba5beb9ed0 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Sat, 26 Sep 2026 23:21:26 +0000 Subject: [PATCH 16/17] Review generated image pixels against the documentation. Prompt grounding only checked the caption. image-generate now OCRs the PNG for invented labels and vision-reviews it (OpenAI / Grok / Claude), retries once with the critique, and deletes a failing asset. validate adds image_asset_alignment (OCR on by default; vision opt-in). Co-authored-by: jmjava --- AGENTS.md | 5 +- README.md | 24 +++-- src/docgen/ai_client.py | 104 +++++++++++++++++++- src/docgen/config.py | 55 +++++++++++ src/docgen/image_align.py | 150 +++++++++++++++++++++++++++++ src/docgen/image_generate.py | 121 +++++++++++++++++++++-- src/docgen/init.py | 1 + src/docgen/scene_spec.py | 28 ++++++ src/docgen/validate.py | 105 +++++++++++++++++++- src/docgen/yaml_generate.py | 1 + tests/test_ai_client.py | 64 ++++++++++++ tests/test_config.py | 29 +++++- tests/test_image_align.py | 73 ++++++++++++++ tests/test_image_generate.py | 72 ++++++++++++++ tests/test_scene_spec.py | 16 +++ tests/test_validate.py | 60 +++++++++++- tests/test_validate_timing_sync.py | 4 +- tests/test_yaml_generate.py | 1 + 18 files changed, 889 insertions(+), 24 deletions(-) create mode 100644 src/docgen/image_align.py create mode 100644 tests/test_image_align.py diff --git a/AGENTS.md b/AGENTS.md index e14875f..5b141a8 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -70,10 +70,10 @@ Commands registered on the **`docgen`** CLI include: - **`freeze`** — ``docgen freeze`` builds the **`docgen-gui`** onedir (`pip install 'docgen[packaging]'`). Optional ``--smoke`` runs the binary headless. Do not run a full freeze in routine pytest; set ``DOCGEN_FREEZE_SMOKE=1`` for the optional test. - **`tts`** — text-to-speech for segment files (OpenAI or xAI `/v1/tts`). - **`timestamps`** — word/segment timing (`timing.json`). Default engine **`local`** aligns the known narration text against the mp3 offline (ffmpeg silencedetect, no API); **`--engine whisper`** uses OpenAI whisper-1 or xAI `/v1/stt` when `ai.provider` is grok. Both emit the same Whisper-shaped blocks. Failed ffmpeg silencedetect raises `AlignmentError` (empty stderr is not treated as full-span speech). OpenAI whisper-1 word/segment `start`/`end` must be finite JSON numbers (bool/NaN raise `AIError`). Grok `/v1/stt` rejects empty word tokens and inverted `end < start` intervals (`AIError`). Empty ``segments.all`` raises ``TimestampError`` (same as TTS) and does not leave a stale ``timing.json`` as success. -- **`image-generate`** — render scene-spec **image elements** (`image:` + `prompt:` boxes) via OpenAI Images or xAI Imagine into the bundle (also runs for missing assets inside `generate-all`). Default **`align_with_docs`** wraps each prompt with narration + source snippets and fails when the authored prompt shares no documented terms. +- **`image-generate`** — render scene-spec **image elements** (`image:` + `prompt:` boxes) via OpenAI Images or xAI Imagine into the bundle (also runs for missing assets inside `generate-all`). Default **`align_with_docs`** wraps each prompt with narration + source snippets and fails when the authored prompt shares no documented terms. After the PNG is written, **OCR** plus a **vision review** (`align_review`, default true) check the pixels against the same corpus and retry once on FAIL. - **`manim`** — render Manim scenes declared in config. - **`compose`** — mux narration audio with visual sources via ffmpeg. With no segment ids, uses ``segments.all`` (same as ``generate-all``), not ``segments.default``. A ``type: mixed`` row raises ``ComposeError`` if any listed source is missing (no silent subset mux). -- **`validate`** / **`validate --pre-push`** — drift, narration lint, Manim hints, **`timing_sync`**, **`story_end`** (last paced reveal vs audio end; hard fail), **`scene_assets`** (pre-render: stuck-board cadence, frame-budget overlaps, `MANIM_FONT` consistency, stale helpers / stale compiled class including hand-edited generated-region labels and `run_time` — hard fail; also a `generate-all` gate before Manim), **`av_sync`** (hard fail on `--pre-push` / `generate-all`; prefers scene-spec labels as OCR anchors), **`subject_beat_coverage`** (declarative specs vs narration topic beats; hard fail when enabled), **`image_prompt_alignment`** (image-element prompts must share documented terms with narration/source; hard fail when enabled), and related visual-sync checks (`ocr_scan`, `layout`, `freeze_ratio` — hard fail on `--pre-push` / `generate-all`). Missing tesseract fails `ocr_scan` / `av_sync` / `layout` (not skip-PASS). Missing audio or an LFS pointer fails `timing_sync` and recording media gates (`stream_presence`, `av_drift`, `ocr_scan`, `av_sync`) (not skip-PASS). Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` / `image_prompt_alignment` for `type: manim` (not skip-PASS). +- **`validate`** / **`validate --pre-push`** — drift, narration lint, Manim hints, **`timing_sync`**, **`story_end`** (last paced reveal vs audio end; hard fail), **`scene_assets`** (pre-render: stuck-board cadence, frame-budget overlaps, `MANIM_FONT` consistency, stale helpers / stale compiled class including hand-edited generated-region labels and `run_time` — hard fail; also a `generate-all` gate before Manim), **`av_sync`** (hard fail on `--pre-push` / `generate-all`; prefers scene-spec labels as OCR anchors), **`subject_beat_coverage`** (declarative specs vs narration topic beats; hard fail when enabled), **`image_prompt_alignment`** (image-element prompts must share documented terms with narration/source; hard fail when enabled), **`image_asset_alignment`** (OCR of generated PNGs vs docs; optional vision when `validation.image_asset_alignment.review`; hard fail when enabled), and related visual-sync checks (`ocr_scan`, `layout`, `freeze_ratio` — hard fail on `--pre-push` / `generate-all`). Missing tesseract fails `ocr_scan` / `av_sync` / `layout` / `image_asset_alignment` (not skip-PASS). Missing audio or an LFS pointer fails `timing_sync` and recording media gates (`stream_presence`, `av_drift`, `ocr_scan`, `av_sync`) (not skip-PASS). Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` / `image_prompt_alignment` / `image_asset_alignment` for `type: manim` (not skip-PASS). - **`lint`** — narration lint helper. - **`narration-generate`** — LLM-assisted narration from hints and repo context; optional **`--revise --revision-notes`** for in-place edits (same contract as the wizard Revise button). - **`scene-spec-generate`** — LLM emits declarative **`*.scene.yaml`**; enforces frame budget + **subject-beat coverage** (dwell OK; cover topic shifts; reject invented labels) and **image-prompt alignment** (image `prompt:` must use documented terms from narration/source). @@ -92,6 +92,7 @@ Commands registered on the **`docgen`** CLI include: - **Beat sync (fail-closed):** when `timing.json` has words, every story box label must match a spoken phrase (`wait_word`); unmatched labels and leftover LLM indices are rejected. Opt out with ``pace: none``. Legacy row-level ``wait_segment`` is upgraded to ``wait_word`` and written back on ``scene-compile`` (direct ``compile_scene_class`` still rejects leftover ``wait_segment``). Fuzzy containment matching is not used. **`scene-compile` clamps FadeIn / page-fade `run_time` against the next word start** so `_TimedScene._clock` cannot race past waits (issue #66 — do not emit cascading first-board dumps). After a reveal, a **dwell** slot may play `Indicate` / `Circumscribe` when the gap to the next `wait_word` is long enough (also clamped). Long holds emit additional mid-hold pulses (`timed_wait` + emphasis) so the board does not freeze after the first Indicate. Optional box fields: `shape` (rounded/pill/diamond), `reveal` (fade/grow/slide), `emphasis` (none/pulse/ring). Page transitions FadeOut revealed boxes, not the parent `VGroup`. `scene-compile` refreshes stale `_box` / `_arrow` / `_TimedScene` helpers in `scenes.py`. - **Subject-beat coverage:** implemented in `scene_spec.layout_density_violations` / `cluster_subject_beats`; enforced by **`scene-spec-generate`** and **`validate`** (`validation.subject_beat_coverage.enabled`, default true). Not a blind label count. - **Image-prompt alignment:** implemented in `scene_spec.image_prompt_alignment_violations`; **`scene-spec-generate`** and **`image-generate`** reject prompts that share no documented terms; **`validate`** (`validation.image_prompt_alignment.enabled`, default true) is the same gate. `image-generate` also wraps the Images API prompt with narration/source (`image_generation.align_with_docs`, default true). +- **Image-asset alignment:** OCR + vision review of the generated PNG (`docgen.image_align`). **`image-generate`** fails closed and retries once (`image_generation.align_review`, default true). **`validate`** (`validation.image_asset_alignment`, OCR on by default, vision opt-in) is the same pixel gate. - **`docgen benchmark` is required** after clock / compile / `_TimedScene` / dwell changes. Pytest string assertions are not a substitute. Do not remove the CI `benchmark` job. See **Required gate** below. - Prefer **stable CLI / library contracts** and **documented exit codes** so CI can depend on them. - **`narration_from_source`:** hints in config + **`docgen narration-generate`** — owner-supplied context paths, not opaque bulk edits to outputs. diff --git a/README.md b/README.md index e6802e1..0e6e018 100644 --- a/README.md +++ b/README.md @@ -58,23 +58,27 @@ If you still need the legacy behaviour, pin a pre-removal commit (`grok-imagine-image-2.0` when `ai.provider` is `grok`). By default the authored prompt is **grounded** in the segment narration plus `manim_scene_generation` source snippets, and prompts that share no - documented terms fail closed (`image_prompt_alignment`). The compiled scene - shows the asset with the `_image` helper. `generate-all` fills in missing - assets automatically. + documented terms fail closed (`image_prompt_alignment`). After the PNG is + written, OCR rejects invented on-image labels and a vision model reviews + the pixels against the same docs (`image_asset_alignment` / + `image_generation.align_review`). The compiled scene shows the asset with + the `_image` helper. `generate-all` fills in missing assets automatically. - **ffmpeg composition** — combine narration audio and Manim video into final segments, with a freeze-tail guard. - **Validation** — A/V drift, freeze ratio, OCR error scan, layout, narration lint, Manim scene lint, **timing_sync** (stale `timing.json` vs regenerated mp3 — hard fail), **story_end** (paced visual story finishes long before narration — hard fail), **image_prompt_alignment** (image-element `prompt:` must share - documented terms with narration/source — hard fail), and **av_sync** (OCR + documented terms with narration/source — hard fail), **image_asset_alignment** + (OCR of generated PNGs vs docs; optional vision review — hard fail), and **av_sync** (OCR check that scene-spec label anchors appear on screen near their spoken time — hard fail on `--pre-push` / `generate-all`). Missing tesseract fails `ocr_scan` / `av_sync` / `layout` instead of skip-PASS. Missing audio or an LFS pointer fails `timing_sync` and recording media gates (`stream_presence`, `av_drift`, `ocr_scan`, `av_sync`) instead of skip-PASS. Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` / - `image_prompt_alignment` for `type: manim` instead of skip-PASS. A hand-edited generated-region label + `image_prompt_alignment` / `image_asset_alignment` for `type: manim` + instead of skip-PASS. A hand-edited generated-region label or `run_time` in `scenes.py` fails `scene_assets` compile_sync (clock / benchmark still execute a fresh `compile_scene_class`, not the on-disk file). - **GitHub Pages** — auto-generate `index.html`, deploy workflow, LFS rules, @@ -134,6 +138,8 @@ image_generation: model: gpt-image-1 # or dall-e-3, gpt-image-1-mini, … size: 1536x1024 align_with_docs: true # wrap prompts with narration/source; fail invented terms + align_review: true # vision-review the PNG; retry once on FAIL + # review_model: gpt-4o # quality: high # style: "…" # optional override of the educational-diagram prefix ``` @@ -248,7 +254,7 @@ docgen --repo /path/to/your-project generate-all | `docgen freeze [--dist DIR] [--smoke]` | PyInstaller onedir for **`docgen-gui` only** (`pip install 'docgen[packaging]'`). Not the full Manim CLI | | `docgen tts [--segment 01] [--dry-run]` | Generate TTS audio | | `docgen timestamps [--engine local\|whisper]` | Extract word/segment timestamps from TTS audio → `timing.json` (default `local`: offline narration-text alignment; `whisper`: OpenAI transcription). Empty `segments.all` is `TimestampError` (stale `timing.json` is not success) | -| `docgen image-generate [--segment 01 \| --all \| --spec PATH] [--force] [--dry-run] [--model …] [--size …]` | Generate scene-spec image assets (`image:` + `prompt:` boxes) via the OpenAI Images API; grounds prompts in narration/source and fails unaligned prompts unless `image_generation.align_with_docs` is false | +| `docgen image-generate [--segment 01 \| --all \| --spec PATH] [--force] [--dry-run] [--model …] [--size …]` | Generate scene-spec image assets (`image:` + `prompt:` boxes) via the OpenAI Images API; grounds prompts in narration/source, then OCR + vision-reviews the PNG unless `align_with_docs` / `align_review` is false | | `docgen manim [--scene StackDAGScene]` | Render Manim animations | | `docgen compose [01 02 03] [--ffmpeg-timeout 900]` | Compose segments (audio + video). Omit ids to walk `segments.all` (same as `generate-all`); a mapped segment with missing audio/visuals is a hard fail | | `docgen validate [--max-drift 2.75] [--pre-push]` | Run all validation checks | @@ -352,6 +358,8 @@ image_generation: # scene-spec image elements (docgen image-generate) model: gpt-image-1 # Cursor/OpenAI Images; Grok remaps gpt-image-* to Imagine size: 1536x1024 align_with_docs: true # ground prompts in narration/source (default) + align_review: true # vision-review generated PNGs (default) + # review_model: gpt-4o # quality: high # optional, model-specific # style: "…" # optional Images-API prefix override @@ -366,6 +374,10 @@ validation: enabled: true # scene-spec-generate + validate: cover narration topic beats image_prompt_alignment: enabled: true # image-element prompts must use documented terms + image_asset_alignment: + enabled: true # OCR generated PNGs vs narration/source + ocr: true + review: false # set true to vision-review existing assets in validate compose: ffmpeg_timeout_sec: 300 # can also be overridden with: docgen compose --ffmpeg-timeout N diff --git a/src/docgen/ai_client.py b/src/docgen/ai_client.py index 16a3bb2..536ae97 100644 --- a/src/docgen/ai_client.py +++ b/src/docgen/ai_client.py @@ -9,7 +9,8 @@ Chat + images for OpenAI/Grok go through the ``openai`` SDK. xAI is ``base_url=https://api.x.ai/v1`` plus model aliases. Anthropic chat uses -``POST https://api.anthropic.com/v1/messages`` (no TTS/images there). +``POST https://api.anthropic.com/v1/messages`` (no image *generation* there; +Claude can still *review* a PNG via ``chat_completion_with_image``). TTS/STT: OpenAI ``/v1/audio/speech`` + ``whisper-1``, or xAI ``/v1/tts`` and ``/v1/stt``. @@ -393,6 +394,89 @@ def _create() -> Any: return text +def chat_completion_with_image( + *, + system_prompt: str, + user_message: str, + image_bytes: bytes, + media_type: str = "image/png", + model: str, + temperature: float, + cfg: "Config | None" = None, +) -> str: + """Vision chat: attach one image plus text (OpenAI, Grok, or Anthropic).""" + import base64 + + settings = resolve_ai_settings(cfg) + resolved = resolve_chat_model(model, settings) + if not image_bytes: + raise AIError("vision chat needs non-empty image bytes") + mime = (media_type or "image/png").strip() or "image/png" + b64 = base64.b64encode(image_bytes).decode("ascii") + if settings.is_anthropic: + return _anthropic_chat( + system_prompt=system_prompt, + user_message=user_message, + model=resolved, + temperature=temperature, + settings=settings, + image_b64=b64, + image_media_type=mime, + ) + + import openai + + from docgen.openai_retry import call_with_rate_limit_retries + + client = openai_client(cfg) + + def _create() -> Any: + return client.chat.completions.create( + model=resolved, + messages=[ + {"role": "system", "content": system_prompt}, + { + "role": "user", + "content": [ + {"type": "text", "text": user_message}, + { + "type": "image_url", + "image_url": {"url": f"data:{mime};base64,{b64}"}, + }, + ], + }, + ], + temperature=float(temperature), + ) + + try: + response = call_with_rate_limit_retries(_create) + except openai.AuthenticationError as exc: + raise AIError( + f"{_vendor(settings)} rejected {settings.api_key_env} (authentication failed): {exc}. " + f"{settings.auth_help()}" + ) from exc + except openai.PermissionDeniedError as exc: + raise AIError( + f"{_vendor(settings)} permission denied for model {resolved!r}: {exc}." + ) from exc + except openai.RateLimitError as exc: + raise AIError( + f"{_vendor(settings)} rate-limited for model {resolved!r}: {exc}." + ) from exc + except openai.APIConnectionError as exc: + raise AIError( + f"{_vendor(settings)} connection error: {exc} — re-run when connectivity is restored." + ) from exc + try: + text = (response.choices[0].message.content or "").strip() + except (IndexError, AttributeError): + text = "" + if not text: + raise AIError(f"{_vendor(settings)} vision chat returned no text content.") + return text + + def synthesize_speech( *, text: str, @@ -556,15 +640,31 @@ def _anthropic_chat( model: str, temperature: float, settings: AISettings, + image_b64: str | None = None, + image_media_type: str = "image/png", ) -> str: if not settings.api_key: raise AIError(f"Anthropic chat needs ANTHROPIC_API_KEY. {settings.auth_help()}") + if image_b64: + user_content: Any = [ + { + "type": "image", + "source": { + "type": "base64", + "media_type": image_media_type or "image/png", + "data": image_b64, + }, + }, + {"type": "text", "text": user_message}, + ] + else: + user_content = user_message payload = { "model": model, "max_tokens": ANTHROPIC_MAX_TOKENS, "temperature": float(temperature), "system": system_prompt, - "messages": [{"role": "user", "content": user_message}], + "messages": [{"role": "user", "content": user_content}], } raw = _http_json( _anthropic_messages_url(settings), diff --git a/src/docgen/config.py b/src/docgen/config.py index 184cd44..0aba5d2 100644 --- a/src/docgen/config.py +++ b/src/docgen/config.py @@ -332,6 +332,7 @@ def __post_init__(self) -> None: "narration_lint", "subject_beat_coverage", "image_prompt_alignment", + "image_asset_alignment", ): self._sub_block(validation, nested, label=f"validation.{nested}") # List-valued keys: a string must not be iterated as characters. @@ -494,6 +495,22 @@ def __post_init__(self) -> None: ) if ig.get("style") is not None: require_yaml_string(ig["style"], label="image_generation.style", source=src) + if ig.get("align_review") is not None: + require_yaml_bool( + ig["align_review"], + label="image_generation.align_review", + source=src, + ) + if ig.get("review_model") is not None: + require_yaml_string( + ig["review_model"], label="image_generation.review_model", source=src + ) + if ig.get("align_review_retries") is not None: + require_yaml_number( + ig["align_review_retries"], + label="image_generation.align_review_retries", + source=src, + ) ai = self._block("ai") if ai.get("provider") is not None: require_yaml_string(ai["provider"], label="ai.provider", source=src) @@ -786,6 +803,29 @@ def __post_init__(self) -> None: label="validation.image_prompt_alignment.enabled", source=src, ) + iaa = self._sub_block( + validation, + "image_asset_alignment", + label="validation.image_asset_alignment", + ) + if iaa.get("enabled") is not None: + require_yaml_bool( + iaa["enabled"], + label="validation.image_asset_alignment.enabled", + source=src, + ) + if iaa.get("ocr") is not None: + require_yaml_bool( + iaa["ocr"], + label="validation.image_asset_alignment.ocr", + source=src, + ) + if iaa.get("review") is not None: + require_yaml_bool( + iaa["review"], + label="validation.image_asset_alignment.review", + source=src, + ) def _source_label(self) -> str: return self.yaml_path.name if self.yaml_path else "docgen.yaml" @@ -971,6 +1011,8 @@ def image_generation_config(self) -> dict[str, Any]: "model": "gpt-image-1", "size": "1536x1024", "align_with_docs": True, + "align_review": True, + "align_review_retries": 1, } defaults.update(self._block("image_generation")) return defaults @@ -998,6 +1040,19 @@ def image_prompt_alignment_enabled(self) -> bool: return bool(block.get("enabled")) return True + @property + def image_asset_alignment_config(self) -> dict[str, Any]: + """Pixel checks on generated scene images (OCR + optional vision).""" + defaults: dict[str, Any] = {"enabled": True, "ocr": True, "review": False} + defaults.update( + self._sub_block( + self._block("validation"), + "image_asset_alignment", + label="validation.image_asset_alignment", + ) + ) + return defaults + # -- Manim ----------------------------------------------------------------- @property diff --git a/src/docgen/image_align.py b/src/docgen/image_align.py new file mode 100644 index 0000000..b31dc4d --- /dev/null +++ b/src/docgen/image_align.py @@ -0,0 +1,150 @@ +"""Pixel-level alignment of generated scene images against documentation. + +Prompt grounding (``image_prompt_alignment``) only checks the caption. This +module looks at the **PNG**: + +* **OCR** — readable tokens that are not in narration/source fail (invented + on-image labels). Empty OCR is fine (many diagrams have no text). +* **Vision review** — a chat model that accepts images (OpenAI, Grok, or + Claude) answers PASS/FAIL against the same documentation corpus. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path +from typing import TYPE_CHECKING, Callable + +if TYPE_CHECKING: + from docgen.config import Config + +DEFAULT_REVIEW_MODEL = "gpt-4o" + +_REVIEW_SYSTEM = """You review one illustration for a documentation video. + +Decide whether the IMAGE depicts the DOCUMENTED SUBJECT (narration + source). +PASS when the picture is a reasonable, simplified depiction of those concepts. +FAIL when it shows a different subject, invented product/API names, decorative +text that is not copied from the docs, fake UI chrome, or generic clip-art +unrelated to the documentation. + +Reply with EXACTLY two lines and nothing else: +PASS + +or +FAIL + +""" + + +@dataclass(frozen=True) +class ImageReviewResult: + passed: bool + reason: str + ocr_text: str = "" + + +def ocr_image_text(path: Path) -> str | None: + """OCR a still image. ``None`` if unreadable or tesseract is unavailable. + + Empty string means the file decoded but no text was found. + """ + try: + import cv2 + import pytesseract + + pytesseract.get_tesseract_version() + except Exception: + return None + img = cv2.imread(str(path)) + if img is None: + return None + gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) + _, thresh = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU) + try: + return str(pytesseract.image_to_string(thresh) or "") + except Exception: + return None + + +def parse_review_verdict(text: str) -> ImageReviewResult: + """Parse a PASS/FAIL vision reply. Unparseable text fails closed.""" + lines = [ln.strip() for ln in (text or "").splitlines() if ln.strip()] + if not lines: + return ImageReviewResult(False, "vision review returned empty text") + verdict = lines[0].split()[0].upper().strip(".:") + reason = lines[1] if len(lines) > 1 else " ".join(lines[1:]) or lines[0] + if verdict == "PASS": + return ImageReviewResult(True, reason) + if verdict == "FAIL": + return ImageReviewResult(False, reason) + preview = (text or "").strip() + if len(preview) > 80: + preview = preview[:77] + "..." + return ImageReviewResult(False, f"vision review reply was not PASS/FAIL: {preview!r}") + + +def build_review_user_message( + *, + corpus_text: str, + authored_prompt: str = "", + label: str = "", +) -> str: + parts = [ + "DOCUMENTED SUBJECT:", + (corpus_text or "").strip() or "(none)", + ] + if (authored_prompt or "").strip(): + parts.extend(["", "AUTHORED IMAGE PROMPT:", authored_prompt.strip()]) + if (label or "").strip(): + parts.extend(["", "ON-SCREEN LABEL:", label.strip()]) + parts.extend(["", "Review the attached image."]) + return "\n".join(parts) + + +def _media_type_for(path: Path) -> str: + ext = path.suffix.lower() + return { + ".jpg": "image/jpeg", + ".jpeg": "image/jpeg", + ".webp": "image/webp", + ".gif": "image/gif", + }.get(ext, "image/png") + + +def review_image_against_docs( + path: Path, + *, + corpus_text: str, + authored_prompt: str = "", + label: str = "", + cfg: "Config | None" = None, + model: str = "", + chat_fn: Callable[..., str] | None = None, +) -> ImageReviewResult: + """Send ``path`` + documentation to a vision-capable chat model.""" + from docgen.ai_client import AIError, chat_completion_with_image + + try: + raw = path.read_bytes() + except OSError as exc: + return ImageReviewResult(False, f"could not read image {path}: {exc}") + if not raw: + return ImageReviewResult(False, f"image {path} is empty") + user = build_review_user_message( + corpus_text=corpus_text, authored_prompt=authored_prompt, label=label + ) + invoke = chat_fn or chat_completion_with_image + try: + text = invoke( + system_prompt=_REVIEW_SYSTEM, + user_message=user, + image_bytes=raw, + media_type=_media_type_for(path), + model=(model or "").strip() or DEFAULT_REVIEW_MODEL, + temperature=0.0, + cfg=cfg, + ) + except (AIError, RuntimeError, OSError, TypeError, ValueError) as exc: + return ImageReviewResult(False, f"vision review failed: {exc}") + return parse_review_verdict(text) diff --git a/src/docgen/image_generate.py b/src/docgen/image_generate.py index b78a627..980cbb8 100644 --- a/src/docgen/image_generate.py +++ b/src/docgen/image_generate.py @@ -15,8 +15,11 @@ the segment narration plus ``manim_scene_generation`` source snippets so the image model sees the documentation, not only a short caption. Prompts that share no documented terms fail closed (same check as ``validate`` / -``image_prompt_alignment``). ``docgen manim`` then loads the asset via the -``_image`` helper in ``scenes.py``. +``image_prompt_alignment``). After the PNG is written, **OCR** rejects +invented on-image labels and a **vision review** (OpenAI / Grok / Claude) +checks the pixels against the same corpus (``image_generation.align_review``). +A failed review retries once with the critique, then deletes the asset. +``docgen manim`` then loads the asset via the ``_image`` helper in ``scenes.py``. """ from __future__ import annotations @@ -26,8 +29,14 @@ from pathlib import Path from typing import TYPE_CHECKING, Any, Callable +from docgen.image_align import ( + ImageReviewResult, + ocr_image_text, + review_image_against_docs, +) from docgen.openai_retry import call_with_rate_limit_retries from docgen.scene_spec import ( + image_ocr_alignment_violations, image_prompt_alignment_violations, iter_image_elements, load_scene_spec, @@ -227,6 +236,8 @@ def generate_images_for_spec( model_override: str | None = None, size_override: str | None = None, image_fn: Callable[[str], bytes] | None = None, + review_fn: Callable[..., ImageReviewResult] | None = None, + ocr_fn: Callable[[Path], str | None] | None = None, ) -> list[ImageAssetResult]: """Generate missing image assets referenced by one ``*.scene.yaml``. @@ -234,7 +245,9 @@ def generate_images_for_spec( missing **and** has no ``prompt`` fails loud — either commit the file or give the toolchain a prompt to generate it from. - ``image_fn`` is an injection point for tests (prompt → PNG bytes). + ``image_fn`` / ``review_fn`` / ``ocr_fn`` are injection points for tests. + Live vision review runs when ``align_review`` is on and ``image_fn`` is + not injected (or ``review_fn`` is provided). """ spec = load_scene_spec(spec_path) elements = iter_image_elements(spec) @@ -244,8 +257,16 @@ def generate_images_for_spec( quality = icfg.get("quality") quality = str(quality).strip() if quality else None align = bool(icfg.get("align_with_docs", True)) + align_review = bool(icfg.get("align_review", True)) + try: + pixel_retries = int(icfg.get("align_review_retries", 1) or 0) + except (TypeError, ValueError): + pixel_retries = 1 + pixel_retries = max(0, pixel_retries) + review_model = str(icfg.get("review_model") or "").strip() style = str(icfg.get("style") or "").strip() or DEFAULT_IMAGE_STYLE corpus = collect_alignment_corpus(cfg, spec) if align else "" + live_review = align and align_review and (review_fn is not None or image_fn is None) if align: issues = image_prompt_alignment_violations(spec, corpus_text=corpus) @@ -288,17 +309,95 @@ def generate_images_for_spec( prompt=p, model=model, size=size, quality=quality, cfg=cfg ) ) - data = fn(effective) - if not data: + critique = "" + kept = False + last_issues: list[str] = [] + for attempt in range(1 + pixel_retries): + to_send = effective + if critique: + to_send = ( + f"{effective}\n\n--- PIXEL REVIEW FAILED ---\n{critique}\n" + "Redraw so the image matches the documented subject. " + "Do not invent labels or extra components." + ) + data = fn(to_send) + if not data: + raise ImageGenerationError( + f"{spec_path}: image element {rel!r} — provider returned empty bytes" + ) + out.parent.mkdir(parents=True, exist_ok=True) + out.write_bytes(data) + last_issues = _pixel_alignment_issues( + out, + relpath=rel, + corpus=corpus, + authored_prompt=prompt, + label=label, + cfg=cfg, + align=align, + live_review=live_review, + review_model=review_model, + review_fn=review_fn, + ocr_fn=ocr_fn, + ) + if not last_issues: + kept = True + break + critique = "\n".join(last_issues) + if not kept: + out.unlink(missing_ok=True) + joined = "\n ".join(last_issues) raise ImageGenerationError( - f"{spec_path}: image element {rel!r} — provider returned empty bytes" + f"{spec_path}: image {rel!r} failed pixel alignment " + f"(OCR / vision review):\n {joined}" ) - out.parent.mkdir(parents=True, exist_ok=True) - out.write_bytes(data) results.append(ImageAssetResult(rel, out, "generated", prompt, effective)) return results +def _pixel_alignment_issues( + path: Path, + *, + relpath: str, + corpus: str, + authored_prompt: str, + label: str, + cfg: "Config", + align: bool, + live_review: bool, + review_model: str, + review_fn: Callable[..., ImageReviewResult] | None, + ocr_fn: Callable[[Path], str | None] | None, +) -> list[str]: + if not align or not corpus.strip(): + return [] + issues: list[str] = [] + scanned = (ocr_fn or ocr_image_text)(path) + if scanned: + issues.extend( + image_ocr_alignment_violations( + scanned, corpus_text=corpus, relpath=relpath + ) + ) + if live_review: + if review_fn is not None: + verdict = review_fn( + path, corpus_text=corpus, authored_prompt=authored_prompt, label=label + ) + else: + verdict = review_image_against_docs( + path, + corpus_text=corpus, + authored_prompt=authored_prompt, + label=label, + cfg=cfg, + model=review_model, + ) + if not verdict.passed: + issues.append(f"{relpath}: vision review FAIL — {verdict.reason}") + return issues + + def spec_files_for_bundle(cfg: "Config") -> list[Path]: """All committed ``animations/specs/*.scene.yaml`` files, sorted.""" specs_dir = cfg.animations_dir / "specs" @@ -311,6 +410,8 @@ def generate_missing_images_for_bundle( cfg: "Config", *, image_fn: Callable[[str], bytes] | None = None, + review_fn: Callable[..., ImageReviewResult] | None = None, + ocr_fn: Callable[[Path], str | None] | None = None, ) -> list[str]: """Generate only **missing** image assets across all bundle specs. @@ -319,7 +420,9 @@ def generate_missing_images_for_bundle( """ msgs: list[str] = [] for spec_path in spec_files_for_bundle(cfg): - for res in generate_images_for_spec(cfg, spec_path, image_fn=image_fn): + for res in generate_images_for_spec( + cfg, spec_path, image_fn=image_fn, review_fn=review_fn, ocr_fn=ocr_fn + ): if res.status == "generated": msgs.append(f"{spec_path.name}: generated {res.relpath}") return msgs diff --git a/src/docgen/init.py b/src/docgen/init.py index 2a9d583..98ed0d6 100644 --- a/src/docgen/init.py +++ b/src/docgen/init.py @@ -339,6 +339,7 @@ def _write_config(plan: InitPlan) -> str: "model": "gpt-image-1", # Cursor/OpenAI Images; override with --model "size": "1536x1024", "align_with_docs": True, # ground prompts in narration/source; fail invented terms + "align_review": True, # vision-review the PNG against the same docs }, "compose": { "ffmpeg_timeout_sec": 300, diff --git a/src/docgen/scene_spec.py b/src/docgen/scene_spec.py index 4af804d..d22fbd2 100644 --- a/src/docgen/scene_spec.py +++ b/src/docgen/scene_spec.py @@ -1397,6 +1397,34 @@ def image_prompt_alignment_violations( return issues +def image_ocr_alignment_violations( + ocr_text: str, + *, + corpus_text: str, + relpath: str = "image", +) -> list[str]: + """Reject OCR tokens that look like invented documented terms. + + Style words and a missing corpus are ignored. A single short OCR ghost + (``<6`` letters) is tolerated; two confident invented tokens, or one + token of length ≥ 6, fail. + """ + corpus = content_tokens(corpus_text) + if not corpus: + return [] + substance = content_tokens(ocr_text) - _IMAGE_STYLE_TOKENS + if not substance: + return [] + invented = {t for t in substance if t not in corpus} + confident = {t for t in invented if len(t) >= 4} + if len(confident) >= 2 or any(len(t) >= 6 for t in confident): + sample = ", ".join(repr(x) for x in sorted(confident)[:6]) + return [ + f"{relpath}: on-image OCR has terms not in narration/source: {sample}" + ] + return [] + + def cluster_subject_beats( sentences: list[str], *, diff --git a/src/docgen/validate.py b/src/docgen/validate.py index 30ebc83..cfd7818 100644 --- a/src/docgen/validate.py +++ b/src/docgen/validate.py @@ -395,6 +395,7 @@ def validate_segment( report.checks.append(self._check_manim_scene_lint()) report.checks.append(self._check_subject_beat_coverage(seg_id)) report.checks.append(self._check_image_prompt_alignment(seg_id)) + report.checks.append(self._check_image_asset_alignment(seg_id)) report.checks.append(self._check_scene_assets(seg_id)) return report.to_dict() @@ -405,8 +406,9 @@ def run_pre_push(self) -> None: Missing recordings are reported as warnings, not failures — a project that hasn't generated videos yet should still be pushable. Quality checks on *existing* recordings — including visual-sync (``av_sync``, - ``subject_beat_coverage``, ``image_prompt_alignment``, ``ocr_scan``, - ``layout``, ``freeze_ratio``) — and narration lint are hard failures so + ``subject_beat_coverage``, ``image_prompt_alignment``, + ``image_asset_alignment``, ``ocr_scan``, ``layout``, ``freeze_ratio``) + — and narration lint are hard failures so ``generate-all`` cannot print ``Pipeline complete`` after a desynced mux. """ reports = self.run_all() @@ -735,6 +737,105 @@ def _check_image_prompt_alignment(self, seg_id: str) -> CheckResult: ["Image prompts share documented terms with narration/source"], ) + def _check_image_asset_alignment(self, seg_id: str) -> CheckResult: + """OCR (and optional vision) of generated scene-spec PNGs vs documentation.""" + iaa = self.config.image_asset_alignment_config + if not iaa.get("enabled", True): + return CheckResult( + "image_asset_alignment", + True, + ["validation.image_asset_alignment disabled in config (skipped)"], + ) + + seg_name = self.config.resolve_segment_name(seg_id) + spec_path = self.config.animations_dir / "specs" / f"{seg_name}.scene.yaml" + if not spec_path.is_file(): + return _missing_manim_spec_result("image_asset_alignment", spec_path.name) + + import yaml + + from docgen.image_align import ocr_image_text, review_image_against_docs + from docgen.image_generate import collect_alignment_corpus + from docgen.scene_spec import image_ocr_alignment_violations, iter_image_elements + + try: + raw = yaml.safe_load(spec_path.read_text(encoding="utf-8")) + except (OSError, yaml.YAMLError) as exc: + return CheckResult( + "image_asset_alignment", + False, + [f"could not load {spec_path.name}: {exc}"], + ) + if not isinstance(raw, dict): + return CheckResult( + "image_asset_alignment", + False, + [f"{spec_path.name}: root must be a mapping"], + ) + elements = iter_image_elements(raw) + if not elements: + return CheckResult( + "image_asset_alignment", + True, + ["No image elements (skipped)"], + ) + + corpus = collect_alignment_corpus(self.config, raw) + if not corpus.strip(): + return CheckResult( + "image_asset_alignment", + True, + ["No narration/source corpus yet — image pixels not checked"], + ) + + issues: list[str] = [] + scanned_any = False + for el in elements: + rel = str(el.get("image") or "").strip() + if not rel: + continue + asset = self.config.base_dir / rel + if not asset.is_file(): + continue + scanned_any = True + if iaa.get("ocr", True): + unavail = _tesseract_unavailable_detail() + if unavail: + return CheckResult("image_asset_alignment", False, [unavail]) + ocr_text = ocr_image_text(asset) + if ocr_text is None: + issues.append(f"{rel}: could not OCR image (unreadable file)") + elif ocr_text: + issues.extend( + image_ocr_alignment_violations( + ocr_text, corpus_text=corpus, relpath=rel + ) + ) + if iaa.get("review"): + verdict = review_image_against_docs( + asset, + corpus_text=corpus, + authored_prompt=str(el.get("prompt") or ""), + label=str(el.get("label") or ""), + cfg=self.config, + ) + if not verdict.passed: + issues.append(f"{rel}: vision review FAIL — {verdict.reason}") + + if issues: + return CheckResult("image_asset_alignment", False, issues) + if not scanned_any: + return CheckResult( + "image_asset_alignment", + True, + ["Image assets not on disk yet (skipped) — run `docgen image-generate`"], + ) + return CheckResult( + "image_asset_alignment", + True, + ["Generated image pixels match documented terms"], + ) + def _check_scene_assets(self, seg_id: str) -> CheckResult: """Pre-render stuck / overlap / font / compile-sync gate (no video required).""" sa_cfg = self.config.scene_assets_config diff --git a/src/docgen/yaml_generate.py b/src/docgen/yaml_generate.py index 0e87256..82272ef 100644 --- a/src/docgen/yaml_generate.py +++ b/src/docgen/yaml_generate.py @@ -221,6 +221,7 @@ def merge_defaults( "model": "gpt-image-1", "size": "1536x1024", "align_with_docs": True, + "align_review": True, } changes.append( "image_generation: added model gpt-image-1 (Cursor/OpenAI Images; override with --model)" diff --git a/tests/test_ai_client.py b/tests/test_ai_client.py index aece8ae..e9a7757 100644 --- a/tests/test_ai_client.py +++ b/tests/test_ai_client.py @@ -15,6 +15,7 @@ DEFAULT_GROK_IMAGE_MODEL, GROK_BASE_URL, chat_completion, + chat_completion_with_image, openai_client, resolve_ai_settings, resolve_chat_model, @@ -137,6 +138,41 @@ class _Resp: assert captured["model"] == DEFAULT_GROK_CHAT_MODEL +def test_chat_completion_with_image_sends_data_uri( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setenv("CURSOR_API_KEY", "sk-proj-cursor") + captured: dict = {} + + class _Msg: + content = "PASS\nok" + + class _Choice: + message = _Msg() + + class _Resp: + choices = [_Choice()] + + fake = MagicMock() + fake.chat.completions.create.side_effect = lambda **kw: captured.update(kw) or _Resp() + + with patch("docgen.ai_client.openai_client", return_value=fake): + out = chat_completion_with_image( + system_prompt="sys", + user_message="review this", + image_bytes=b"png-bytes", + media_type="image/png", + model="gpt-4o", + temperature=0.0, + cfg=_cfg(tmp_path, {}), + ) + assert out == "PASS\nok" + content = captured["messages"][1]["content"] + assert content[0]["type"] == "text" + assert content[1]["type"] == "image_url" + assert content[1]["image_url"]["url"].startswith("data:image/png;base64,") + + def test_openai_empty_chat_content_raises( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, clear_ai_env: None ) -> None: @@ -583,6 +619,34 @@ def _http(url: str, *, data: bytes, headers: dict, **_kwargs) -> bytes: assert captured["payload"]["model"] == DEFAULT_ANTHROPIC_CHAT_MODEL +def test_anthropic_vision_chat_posts_image_block( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, clear_ai_env: None +) -> None: + monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-test") + cfg = _cfg(tmp_path, {"ai": {"provider": "anthropic"}}) + captured: dict = {} + + def _http(url: str, *, data: bytes, headers: dict, **_kwargs) -> bytes: + captured["payload"] = json.loads(data.decode()) + return json.dumps({"content": [{"type": "text", "text": "PASS\nok"}]}).encode() + + with patch("docgen.ai_client._http_with_retries", side_effect=_http): + out = chat_completion_with_image( + system_prompt="sys", + user_message="review", + image_bytes=b"png-bytes", + media_type="image/png", + model="claude-sonnet-4-5", + temperature=0.0, + cfg=cfg, + ) + assert out == "PASS\nok" + content = captured["payload"]["messages"][0]["content"] + assert content[0]["type"] == "image" + assert content[0]["source"]["type"] == "base64" + assert content[1]["type"] == "text" + + def test_anthropic_chat_honors_base_url( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, clear_ai_env: None ) -> None: diff --git a/tests/test_config.py b/tests/test_config.py index 217355a..bc88250 100644 --- a/tests/test_config.py +++ b/tests/test_config.py @@ -1137,7 +1137,8 @@ def test_from_yaml_validation_enable_bools_allowed(tmp_path: Path) -> None: " story_end:\n enabled: false\n" " layout:\n check_overlap: false\n" " subject_beat_coverage:\n enabled: false\n" - " image_prompt_alignment:\n enabled: false\n", + " image_prompt_alignment:\n enabled: false\n" + " image_asset_alignment:\n enabled: false\n ocr: true\n review: false\n", encoding="utf-8", ) c = Config.from_yaml(p) @@ -1150,6 +1151,9 @@ def test_from_yaml_validation_enable_bools_allowed(tmp_path: Path) -> None: assert c.layout_config["check_overlap"] is False assert c.subject_beat_coverage_enabled is False assert c.image_prompt_alignment_enabled is False + assert c.image_asset_alignment_config["enabled"] is False + assert c.image_asset_alignment_config["ocr"] is True + assert c.image_asset_alignment_config["review"] is False def test_from_yaml_int_image_prompt_alignment_enabled_raises(tmp_path: Path) -> None: @@ -1165,6 +1169,29 @@ def test_from_yaml_int_image_prompt_alignment_enabled_raises(tmp_path: Path) -> Config.from_yaml(p) +def test_from_yaml_int_image_asset_alignment_enabled_raises(tmp_path: Path) -> None: + p = tmp_path / "docgen.yaml" + p.write_text( + "validation:\n image_asset_alignment:\n enabled: 0\n", + encoding="utf-8", + ) + with pytest.raises( + ConfigError, + match="validation.image_asset_alignment.enabled must be a YAML boolean", + ): + Config.from_yaml(p) + + +def test_from_yaml_int_image_align_review_raises(tmp_path: Path) -> None: + p = tmp_path / "docgen.yaml" + p.write_text("image_generation:\n align_review: 1\n", encoding="utf-8") + with pytest.raises( + ConfigError, + match="image_generation.align_review must be a YAML boolean", + ): + Config.from_yaml(p) + + def test_from_yaml_int_image_align_with_docs_raises(tmp_path: Path) -> None: p = tmp_path / "docgen.yaml" p.write_text("image_generation:\n align_with_docs: 1\n", encoding="utf-8") diff --git a/tests/test_image_align.py b/tests/test_image_align.py new file mode 100644 index 0000000..5842b8a --- /dev/null +++ b/tests/test_image_align.py @@ -0,0 +1,73 @@ +"""Pixel-level image alignment: OCR verdicts and vision PASS/FAIL parsing.""" + +from __future__ import annotations + +from pathlib import Path + +from docgen.image_align import ( + build_review_user_message, + parse_review_verdict, + review_image_against_docs, +) + + +def test_parse_review_verdict_pass_and_fail() -> None: + ok = parse_review_verdict("PASS\nDepicts the checkout service lock.") + assert ok.passed is True + assert "checkout" in ok.reason + bad = parse_review_verdict("FAIL\nShows a WidgetX console instead.") + assert bad.passed is False + assert "WidgetX" in bad.reason + + +def test_parse_review_verdict_unparseable_fails_closed() -> None: + out = parse_review_verdict("looks fine to me") + assert out.passed is False + assert "PASS/FAIL" in out.reason + + +def test_parse_review_verdict_empty_fails_closed() -> None: + out = parse_review_verdict(" ") + assert out.passed is False + + +def test_build_review_user_message_includes_docs() -> None: + msg = build_review_user_message( + corpus_text="The checkout service owns the cart lock.", + authored_prompt="clean diagram of the checkout service", + label="checkout service", + ) + assert "DOCUMENTED SUBJECT" in msg + assert "cart lock" in msg + assert "AUTHORED IMAGE PROMPT" in msg + assert "ON-SCREEN LABEL" in msg + + +def test_review_image_against_docs_uses_chat_fn(tmp_path: Path) -> None: + png = tmp_path / "x.png" + png.write_bytes(b"\x89PNG\r\n\x1a\nnot-a-real-png") + captured: dict = {} + + def _chat(**kwargs: object) -> str: + captured.update(kwargs) + return "PASS\nMatches the checkout service." + + verdict = review_image_against_docs( + png, + corpus_text="The checkout service owns the cart lock.", + authored_prompt="diagram of checkout", + chat_fn=_chat, + ) + assert verdict.passed is True + assert captured["image_bytes"] == png.read_bytes() + assert "checkout service" in str(captured["user_message"]) + + +def test_review_empty_file_fails(tmp_path: Path) -> None: + png = tmp_path / "empty.png" + png.write_bytes(b"") + verdict = review_image_against_docs( + png, corpus_text="checkout service", chat_fn=lambda **_k: "PASS\nok" + ) + assert verdict.passed is False + assert "empty" in verdict.reason diff --git a/tests/test_image_generate.py b/tests/test_image_generate.py index 9f1b916..6e52fb0 100644 --- a/tests/test_image_generate.py +++ b/tests/test_image_generate.py @@ -8,6 +8,7 @@ import yaml from docgen.config import Config +from docgen.image_align import ImageReviewResult from docgen.image_generate import ( DEFAULT_IMAGE_STYLE, ImageGenerationError, @@ -216,6 +217,77 @@ def test_source_docs_ground_prompt_without_narration(tmp_path: Path) -> None: assert "cart lock" in seen[0] +def test_ocr_invented_text_deletes_asset(cfg: Config) -> None: + spec = _write_spec( + cfg.animations_dir / "specs" / "01-x.scene.yaml", + prompt="clean diagram of the bootstrap pipeline", + ) + cfg.narration_dir.mkdir(parents=True, exist_ok=True) + (cfg.narration_dir / "1.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", encoding="utf-8" + ) + with pytest.raises(ImageGenerationError, match="pixel alignment"): + generate_images_for_spec( + cfg, + spec, + image_fn=lambda p: _PNG_BYTES, + ocr_fn=lambda _path: "WidgetX Orchestrator console", + ) + assert not (cfg.base_dir / "images" / "arch.png").exists() + + +def test_vision_fail_retries_then_keeps_pass(cfg: Config) -> None: + spec = _write_spec( + cfg.animations_dir / "specs" / "01-x.scene.yaml", + prompt="clean diagram of the bootstrap pipeline", + ) + cfg.narration_dir.mkdir(parents=True, exist_ok=True) + (cfg.narration_dir / "1.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", encoding="utf-8" + ) + seen: list[str] = [] + reviews = iter( + [ + ImageReviewResult(False, "shows a generic city skyline"), + ImageReviewResult(True, "now shows the bootstrap pipeline"), + ] + ) + + def _review(*_a: object, **_k: object) -> ImageReviewResult: + return next(reviews) + + generate_images_for_spec( + cfg, + spec, + image_fn=lambda p: seen.append(p) or _PNG_BYTES, + review_fn=_review, + ocr_fn=lambda _path: "", + ) + assert len(seen) == 2 + assert "PIXEL REVIEW FAILED" in seen[1] + assert (cfg.base_dir / "images" / "arch.png").read_bytes() == _PNG_BYTES + + +def test_vision_fail_exhausted_deletes_asset(cfg: Config) -> None: + spec = _write_spec( + cfg.animations_dir / "specs" / "01-x.scene.yaml", + prompt="clean diagram of the bootstrap pipeline", + ) + cfg.narration_dir.mkdir(parents=True, exist_ok=True) + (cfg.narration_dir / "1.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", encoding="utf-8" + ) + with pytest.raises(ImageGenerationError, match="vision review FAIL"): + generate_images_for_spec( + cfg, + spec, + image_fn=lambda p: _PNG_BYTES, + review_fn=lambda *_a, **_k: ImageReviewResult(False, "wrong subject"), + ocr_fn=lambda _path: "", + ) + assert not (cfg.base_dir / "images" / "arch.png").exists() + + def test_build_aligned_image_prompt_includes_corpus_and_label() -> None: out = build_aligned_image_prompt( "clean diagram of the bootstrap pipeline", diff --git a/tests/test_scene_spec.py b/tests/test_scene_spec.py index 16cf2a0..f8f00bc 100644 --- a/tests/test_scene_spec.py +++ b/tests/test_scene_spec.py @@ -17,6 +17,7 @@ disk_spec_with_merged_wait_words, cluster_subject_beats, count_spec_labels, + image_ocr_alignment_violations, image_prompt_alignment_violations, iter_paced_label_anchors, last_paced_reveal_time, @@ -603,6 +604,21 @@ def test_image_prompt_alignment_requires_documented_terms() -> None: assert image_prompt_alignment_violations(spec, corpus_text="") == [] +def test_image_ocr_alignment_flags_invented_on_image_text() -> None: + narr = "The bootstrap pipeline seeds the cluster." + assert image_ocr_alignment_violations( + "bootstrap pipeline", corpus_text=narr, relpath="images/arch.png" + ) == [] + assert image_ocr_alignment_violations("", corpus_text=narr) == [] + issues = image_ocr_alignment_violations( + "WidgetX Orchestrator console", + corpus_text=narr, + relpath="images/arch.png", + ) + assert issues + assert any("OCR" in msg for msg in issues) + + def test_subject_beat_coverage_allows_dwell_rejects_missed_topics() -> None: narr = ( "The bootstrap pipeline seeds the cluster. " diff --git a/tests/test_validate.py b/tests/test_validate.py index 6395904..0ca1dd0 100644 --- a/tests/test_validate.py +++ b/tests/test_validate.py @@ -717,6 +717,64 @@ def test_validate_passes_grounded_image_prompt(self, cfg_dir: Path) -> None: check = next(c for c in report["checks"] if c["name"] == "image_prompt_alignment") assert check["passed"], check["details"] + def test_validate_flags_invented_ocr_on_asset(self, cfg_dir: Path, monkeypatch) -> None: + cfg_raw = yaml.safe_load((cfg_dir / "docgen.yaml").read_text(encoding="utf-8")) + cfg_raw["visual_map"]["01"] = {"type": "manim", "source": "Scene01.mp4"} + cfg_raw.setdefault("segment_names", {})["01"] = "01-test" + (cfg_dir / "docgen.yaml").write_text(yaml.dump(cfg_raw), encoding="utf-8") + (cfg_dir / "narration" / "01-test.md").write_text( + "The bootstrap pipeline seeds the cluster.\n", + encoding="utf-8", + ) + specs = cfg_dir / "animations" / "specs" + specs.mkdir(parents=True, exist_ok=True) + (specs / "01-test.scene.yaml").write_text( + yaml.dump( + { + "segment_id": "01", + "class_name": "DemoScene", + "title": {"text": "T", "font_size": 36, "color": "C_WHITE"}, + "rows": [ + { + "run_time": 1.0, + "boxes": [ + { + "label": "bootstrap pipeline", + "color": "C_ORANGE", + "width": 4.0, + "height": 1.0, + "font_size": 18, + }, + { + "image": "images/arch.png", + "width": 4.0, + "height": 2.0, + "prompt": "clean diagram of the bootstrap pipeline", + }, + ], + } + ], + } + ), + encoding="utf-8", + ) + asset = cfg_dir / "images" / "arch.png" + asset.parent.mkdir(parents=True, exist_ok=True) + asset.write_bytes(b"\x89PNG\r\n\x1a\nfake") + monkeypatch.setattr( + "docgen.image_align.ocr_image_text", + lambda _path: "WidgetX Orchestrator console", + ) + monkeypatch.setattr( + "docgen.validate._tesseract_unavailable_detail", + lambda: None, + ) + config = Config.from_yaml(cfg_dir / "docgen.yaml") + report = Validator(config).validate_segment("01") + check = next(c for c in report["checks"] if c["name"] == "image_asset_alignment") + assert not check["passed"] + assert any("OCR" in d for d in check["details"]) + # ── ffprobe JSON probes honor returncode ────────────────────────────── @@ -781,7 +839,7 @@ def test_drift_pass_when_ffprobe_succeeds(self, config, tmp_path, monkeypatch): @pytest.mark.parametrize( "check_name", - ("av_sync", "subject_beat_coverage", "image_prompt_alignment", "ocr_scan", "layout", "freeze_ratio"), + ("av_sync", "subject_beat_coverage", "image_prompt_alignment", "image_asset_alignment", "ocr_scan", "layout", "freeze_ratio"), ) def test_run_pre_push_visual_sync_fail_is_hard(check_name: str, capsys) -> None: """Visual-sync FAILs must be FAIL + SystemExit, not WARN (leftover #2).""" diff --git a/tests/test_validate_timing_sync.py b/tests/test_validate_timing_sync.py index 235c44d..b4249d8 100644 --- a/tests/test_validate_timing_sync.py +++ b/tests/test_validate_timing_sync.py @@ -508,7 +508,7 @@ def _write_paced_hand_authored_scenes(cfg: Config) -> None: @pytest.mark.parametrize( "check_name", - ("story_end", "subject_beat_coverage", "image_prompt_alignment"), + ("story_end", "subject_beat_coverage", "image_prompt_alignment", "image_asset_alignment"), ) def test_hand_authored_paced_manim_without_spec_fails(cfg, check_name: str) -> None: """Manim + paced timed_play + no *.scene.yaml must fail, not skip-PASS.""" @@ -522,6 +522,8 @@ def test_hand_authored_paced_manim_without_spec_fails(cfg, check_name: str) -> N check = v._check_story_end("01") elif check_name == "image_prompt_alignment": check = v._check_image_prompt_alignment("01") + elif check_name == "image_asset_alignment": + check = v._check_image_asset_alignment("01") else: check = v._check_subject_beat_coverage("01") assert check.name == check_name diff --git a/tests/test_yaml_generate.py b/tests/test_yaml_generate.py index 5501f5f..0bf7781 100644 --- a/tests/test_yaml_generate.py +++ b/tests/test_yaml_generate.py @@ -86,6 +86,7 @@ def test_merge_defaults_adds_archive_exclude(tmp_path: Path) -> None: assert raw["image_generation"]["model"] == "gpt-image-1" assert raw["image_generation"]["size"] == "1536x1024" assert raw["image_generation"]["align_with_docs"] is True + assert raw["image_generation"]["align_review"] is True def test_merge_defaults_idempotent_archive(tmp_path: Path) -> None: From 10f92957e9df960e1254197f1d19b9d0df871118 Mon Sep 17 00:00:00 2001 From: John Menke Date: Wed, 7 Oct 2026 20:04:18 -0400 Subject: [PATCH 17/17] Keep image-doc alignment under the complexity gates. Move the new checks into quiet modules and thin wrappers so hotspot files match main and new functions stay within the existing CCN and NLOC limits. Co-authored-by: Cursor --- AGENTS.md | 6 +- README.md | 10 +- src/docgen/ai_client.py | 159 ++-- src/docgen/cli.py | 1304 ++++++++++++++++++---------- src/docgen/config.py | 99 --- src/docgen/image_align.py | 272 +++++- src/docgen/image_generate.py | 393 ++++++--- src/docgen/init.py | 2 - src/docgen/scene_asset_validate.py | 3 + src/docgen/scene_spec.py | 86 -- src/docgen/scene_spec_flow.py | 633 ++++++++++++++ src/docgen/scene_spec_generate.py | 459 +--------- src/docgen/validate.py | 164 +--- src/docgen/yaml_generate.py | 2 - tests/test_ai_client.py | 3 +- tests/test_config.py | 34 +- tests/test_scene_spec.py | 6 +- tests/test_validate.py | 17 +- tests/test_validate_timing_sync.py | 9 +- tests/test_yaml_generate.py | 2 - 20 files changed, 2197 insertions(+), 1466 deletions(-) create mode 100644 src/docgen/scene_spec_flow.py diff --git a/AGENTS.md b/AGENTS.md index 5b141a8..9627091 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -73,7 +73,7 @@ Commands registered on the **`docgen`** CLI include: - **`image-generate`** — render scene-spec **image elements** (`image:` + `prompt:` boxes) via OpenAI Images or xAI Imagine into the bundle (also runs for missing assets inside `generate-all`). Default **`align_with_docs`** wraps each prompt with narration + source snippets and fails when the authored prompt shares no documented terms. After the PNG is written, **OCR** plus a **vision review** (`align_review`, default true) check the pixels against the same corpus and retry once on FAIL. - **`manim`** — render Manim scenes declared in config. - **`compose`** — mux narration audio with visual sources via ffmpeg. With no segment ids, uses ``segments.all`` (same as ``generate-all``), not ``segments.default``. A ``type: mixed`` row raises ``ComposeError`` if any listed source is missing (no silent subset mux). -- **`validate`** / **`validate --pre-push`** — drift, narration lint, Manim hints, **`timing_sync`**, **`story_end`** (last paced reveal vs audio end; hard fail), **`scene_assets`** (pre-render: stuck-board cadence, frame-budget overlaps, `MANIM_FONT` consistency, stale helpers / stale compiled class including hand-edited generated-region labels and `run_time` — hard fail; also a `generate-all` gate before Manim), **`av_sync`** (hard fail on `--pre-push` / `generate-all`; prefers scene-spec labels as OCR anchors), **`subject_beat_coverage`** (declarative specs vs narration topic beats; hard fail when enabled), **`image_prompt_alignment`** (image-element prompts must share documented terms with narration/source; hard fail when enabled), **`image_asset_alignment`** (OCR of generated PNGs vs docs; optional vision when `validation.image_asset_alignment.review`; hard fail when enabled), and related visual-sync checks (`ocr_scan`, `layout`, `freeze_ratio` — hard fail on `--pre-push` / `generate-all`). Missing tesseract fails `ocr_scan` / `av_sync` / `layout` / `image_asset_alignment` (not skip-PASS). Missing audio or an LFS pointer fails `timing_sync` and recording media gates (`stream_presence`, `av_drift`, `ocr_scan`, `av_sync`) (not skip-PASS). Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` / `image_prompt_alignment` / `image_asset_alignment` for `type: manim` (not skip-PASS). +- **`validate`** / **`validate --pre-push`** — drift, narration lint, Manim hints, **`timing_sync`**, **`story_end`** (last paced reveal vs audio end; hard fail), **`scene_assets`** (pre-render: stuck-board cadence, frame-budget overlaps, `MANIM_FONT` consistency, stale helpers / stale compiled class including hand-edited generated-region labels and `run_time` — hard fail; also a `generate-all` gate before Manim), **`av_sync`** (hard fail on `--pre-push` / `generate-all`; prefers scene-spec labels as OCR anchors), **`subject_beat_coverage`** (declarative specs vs narration topic beats; hard fail when enabled), and related visual-sync checks (`ocr_scan`, `layout`, `freeze_ratio` — hard fail on `--pre-push` / `generate-all`). **`scene_assets`** also fails when an image-element prompt shares no documented terms with narration/source, or when OCR (and optional vision, `validation.image_asset_alignment.review`) finds invented on-image terms. Missing tesseract fails `ocr_scan` / `av_sync` / `layout` (not skip-PASS). Missing audio or an LFS pointer fails `timing_sync` and recording media gates (`stream_presence`, `av_drift`, `ocr_scan`, `av_sync`) (not skip-PASS). Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` for `type: manim` (not skip-PASS). - **`lint`** — narration lint helper. - **`narration-generate`** — LLM-assisted narration from hints and repo context; optional **`--revise --revision-notes`** for in-place edits (same contract as the wizard Revise button). - **`scene-spec-generate`** — LLM emits declarative **`*.scene.yaml`**; enforces frame budget + **subject-beat coverage** (dwell OK; cover topic shifts; reject invented labels) and **image-prompt alignment** (image `prompt:` must use documented terms from narration/source). @@ -91,8 +91,8 @@ Commands registered on the **`docgen`** CLI include: - **Manim / `scenes.py` (marker blocks):** Fix generators under `src/docgen/**` (`manim_scene_support.py`, `scene_spec.py`, `scene_spec_generate.py`, `validate`, `yaml_generate`, tests). **Do not** patch generated classes inside a consumer's **`animations/scenes.py`** between **`BEGIN/END GENERATED SCENE`** markers; re-run **`scene-spec-generate`** / **`scene-compile --retime`** and **`manim`** instead. Preferred consumer order: narration → TTS → timestamps → scene-spec/compile → Manim → compose. - **Beat sync (fail-closed):** when `timing.json` has words, every story box label must match a spoken phrase (`wait_word`); unmatched labels and leftover LLM indices are rejected. Opt out with ``pace: none``. Legacy row-level ``wait_segment`` is upgraded to ``wait_word`` and written back on ``scene-compile`` (direct ``compile_scene_class`` still rejects leftover ``wait_segment``). Fuzzy containment matching is not used. **`scene-compile` clamps FadeIn / page-fade `run_time` against the next word start** so `_TimedScene._clock` cannot race past waits (issue #66 — do not emit cascading first-board dumps). After a reveal, a **dwell** slot may play `Indicate` / `Circumscribe` when the gap to the next `wait_word` is long enough (also clamped). Long holds emit additional mid-hold pulses (`timed_wait` + emphasis) so the board does not freeze after the first Indicate. Optional box fields: `shape` (rounded/pill/diamond), `reveal` (fade/grow/slide), `emphasis` (none/pulse/ring). Page transitions FadeOut revealed boxes, not the parent `VGroup`. `scene-compile` refreshes stale `_box` / `_arrow` / `_TimedScene` helpers in `scenes.py`. - **Subject-beat coverage:** implemented in `scene_spec.layout_density_violations` / `cluster_subject_beats`; enforced by **`scene-spec-generate`** and **`validate`** (`validation.subject_beat_coverage.enabled`, default true). Not a blind label count. -- **Image-prompt alignment:** implemented in `scene_spec.image_prompt_alignment_violations`; **`scene-spec-generate`** and **`image-generate`** reject prompts that share no documented terms; **`validate`** (`validation.image_prompt_alignment.enabled`, default true) is the same gate. `image-generate` also wraps the Images API prompt with narration/source (`image_generation.align_with_docs`, default true). -- **Image-asset alignment:** OCR + vision review of the generated PNG (`docgen.image_align`). **`image-generate`** fails closed and retries once (`image_generation.align_review`, default true). **`validate`** (`validation.image_asset_alignment`, OCR on by default, vision opt-in) is the same pixel gate. +- **Image-prompt alignment:** implemented in `image_align.image_prompt_alignment_violations`; **`scene-spec-generate`** and **`image-generate`** reject prompts that share no documented terms; **`validate`** reports the same failures on **`scene_assets`** (`validation.image_prompt_alignment.enabled`, default true). `image-generate` also wraps the Images API prompt with narration/source (`image_generation.align_with_docs`, default true). +- **Image-asset alignment:** OCR + vision review of the generated PNG (`docgen.image_align`). **`image-generate`** fails closed and retries once (`image_generation.align_review`, default true). **`validate`** folds the same pixel gate into **`scene_assets`** (`validation.image_asset_alignment`, OCR on by default, vision opt-in). - **`docgen benchmark` is required** after clock / compile / `_TimedScene` / dwell changes. Pytest string assertions are not a substitute. Do not remove the CI `benchmark` job. See **Required gate** below. - Prefer **stable CLI / library contracts** and **documented exit codes** so CI can depend on them. - **`narration_from_source`:** hints in config + **`docgen narration-generate`** — owner-supplied context paths, not opaque bulk edits to outputs. diff --git a/README.md b/README.md index 0e6e018..4f926d1 100644 --- a/README.md +++ b/README.md @@ -68,17 +68,15 @@ If you still need the legacy behaviour, pin a pre-removal commit - **Validation** — A/V drift, freeze ratio, OCR error scan, layout, narration lint, Manim scene lint, **timing_sync** (stale `timing.json` vs regenerated mp3 — hard fail), **story_end** (paced visual story finishes long before narration — - hard fail), **image_prompt_alignment** (image-element `prompt:` must share - documented terms with narration/source — hard fail), **image_asset_alignment** - (OCR of generated PNGs vs docs; optional vision review — hard fail), and **av_sync** (OCR + hard fail), **scene_assets** (also fails image prompts or PNG OCR/vision + that do not match narration/source), and **av_sync** (OCR check that scene-spec label anchors appear on screen near their spoken time — hard fail on `--pre-push` / `generate-all`). Missing tesseract fails `ocr_scan` / `av_sync` / `layout` instead of skip-PASS. Missing audio or an LFS pointer fails `timing_sync` and recording media gates (`stream_presence`, `av_drift`, `ocr_scan`, `av_sync`) instead of skip-PASS. - Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` / - `image_prompt_alignment` / `image_asset_alignment` for `type: manim` - instead of skip-PASS. A hand-edited generated-region label + Missing `*.scene.yaml` fails `story_end` / `subject_beat_coverage` for + `type: manim` instead of skip-PASS. A hand-edited generated-region label or `run_time` in `scenes.py` fails `scene_assets` compile_sync (clock / benchmark still execute a fresh `compile_scene_class`, not the on-disk file). - **GitHub Pages** — auto-generate `index.html`, deploy workflow, LFS rules, diff --git a/src/docgen/ai_client.py b/src/docgen/ai_client.py index 536ae97..b17b2ff 100644 --- a/src/docgen/ai_client.py +++ b/src/docgen/ai_client.py @@ -394,36 +394,40 @@ def _create() -> Any: return text -def chat_completion_with_image( +def _raise_openai_chat_error(exc: Exception, settings: AISettings, resolved: str) -> None: + import openai + + if isinstance(exc, openai.AuthenticationError): + raise AIError( + f"{_vendor(settings)} rejected {settings.api_key_env} (authentication failed): {exc}. " + f"{settings.auth_help()}" + ) from exc + if isinstance(exc, openai.PermissionDeniedError): + raise AIError( + f"{_vendor(settings)} permission denied for model {resolved!r}: {exc}." + ) from exc + if isinstance(exc, openai.RateLimitError): + raise AIError( + f"{_vendor(settings)} rate-limited for model {resolved!r}: {exc}." + ) from exc + if isinstance(exc, openai.APIConnectionError): + raise AIError( + f"{_vendor(settings)} connection error: {exc} — re-run when connectivity is restored." + ) from exc + raise exc + + +def _openai_vision_text( *, + cfg: "Config | None", + settings: AISettings, + resolved: str, system_prompt: str, user_message: str, - image_bytes: bytes, - media_type: str = "image/png", - model: str, + mime: str, + b64: str, temperature: float, - cfg: "Config | None" = None, ) -> str: - """Vision chat: attach one image plus text (OpenAI, Grok, or Anthropic).""" - import base64 - - settings = resolve_ai_settings(cfg) - resolved = resolve_chat_model(model, settings) - if not image_bytes: - raise AIError("vision chat needs non-empty image bytes") - mime = (media_type or "image/png").strip() or "image/png" - b64 = base64.b64encode(image_bytes).decode("ascii") - if settings.is_anthropic: - return _anthropic_chat( - system_prompt=system_prompt, - user_message=user_message, - model=resolved, - temperature=temperature, - settings=settings, - image_b64=b64, - image_media_type=mime, - ) - import openai from docgen.openai_retry import call_with_rate_limit_retries @@ -451,23 +455,13 @@ def _create() -> Any: try: response = call_with_rate_limit_retries(_create) - except openai.AuthenticationError as exc: - raise AIError( - f"{_vendor(settings)} rejected {settings.api_key_env} (authentication failed): {exc}. " - f"{settings.auth_help()}" - ) from exc - except openai.PermissionDeniedError as exc: - raise AIError( - f"{_vendor(settings)} permission denied for model {resolved!r}: {exc}." - ) from exc - except openai.RateLimitError as exc: - raise AIError( - f"{_vendor(settings)} rate-limited for model {resolved!r}: {exc}." - ) from exc - except openai.APIConnectionError as exc: - raise AIError( - f"{_vendor(settings)} connection error: {exc} — re-run when connectivity is restored." - ) from exc + except openai.OpenAIError as exc: + _raise_openai_chat_error(exc, settings, resolved) + raise + return _vision_message_text(response, settings) + + +def _vision_message_text(response: Any, settings: AISettings) -> str: try: text = (response.choices[0].message.content or "").strip() except (IndexError, AttributeError): @@ -477,6 +471,51 @@ def _create() -> Any: return text +def _vision_mime_and_b64(image_bytes: bytes, media_type: str) -> tuple[str, str]: + import base64 + + mime = (media_type or "image/png").strip() or "image/png" + return mime, base64.b64encode(image_bytes).decode("ascii") + + +def chat_completion_with_image( + *, + system_prompt: str, + user_message: str, + image_bytes: bytes, + media_type: str = "image/png", + model: str, + temperature: float, + cfg: "Config | None" = None, +) -> str: + """Vision chat: attach one image plus text (OpenAI, Grok, or Anthropic).""" + settings = resolve_ai_settings(cfg) + resolved = resolve_chat_model(model, settings) + if not image_bytes: + raise AIError("vision chat needs non-empty image bytes") + mime, b64 = _vision_mime_and_b64(image_bytes, media_type) + if settings.is_anthropic: + return _anthropic_chat( + system_prompt=system_prompt, + user_message=user_message, + model=resolved, + temperature=temperature, + settings=settings, + image_b64=b64, + image_media_type=mime, + ) + return _openai_vision_text( + cfg=cfg, + settings=settings, + resolved=resolved, + system_prompt=system_prompt, + user_message=user_message, + mime=mime, + b64=b64, + temperature=temperature, + ) + + def synthesize_speech( *, text: str, @@ -633,6 +672,27 @@ def _anthropic_messages_url(settings: AISettings) -> str: return f"{base}/v1/messages" +def _anthropic_user_content( + user_message: str, + image_b64: str | None, + image_media_type: str, +) -> Any: + if not image_b64: + return user_message + media = (image_media_type or "image/png").strip() or "image/png" + return [ + { + "type": "image", + "source": { + "type": "base64", + "media_type": media, + "data": image_b64, + }, + }, + {"type": "text", "text": user_message}, + ] + + def _anthropic_chat( *, system_prompt: str, @@ -645,20 +705,7 @@ def _anthropic_chat( ) -> str: if not settings.api_key: raise AIError(f"Anthropic chat needs ANTHROPIC_API_KEY. {settings.auth_help()}") - if image_b64: - user_content: Any = [ - { - "type": "image", - "source": { - "type": "base64", - "media_type": image_media_type or "image/png", - "data": image_b64, - }, - }, - {"type": "text", "text": user_message}, - ] - else: - user_content = user_message + user_content = _anthropic_user_content(user_message, image_b64, image_media_type) payload = { "model": model, "max_tokens": ANTHROPIC_MAX_TOKENS, diff --git a/src/docgen/cli.py b/src/docgen/cli.py index 49c075b..d58ac06 100644 --- a/src/docgen/cli.py +++ b/src/docgen/cli.py @@ -46,6 +46,37 @@ def _echo_ai_status(cfg: Config | None) -> None: echo_ai_status(cfg) +def _apply_all_env_pairs(pairs: list[tuple[str, str]]) -> None: + for key, value in pairs: + os.environ[key] = value + + +def _warn_ignored_env_key(key: str, value: str, warn_keys: set[str]) -> None: + if key not in warn_keys or not value or not os.environ.get(key): + return + click.echo( + f"[docgen] {key} already set in the process environment; " + "env_file value is ignored for this key (shell wins). " + f"Unset {key} or set DOCGEN_ENV_OVERRIDES=1 to load all keys " + f"from env_file, or DOCGEN_ENV_OVERRIDES={key} to override just " + "this key.", + err=True, + ) + + +def _apply_selective_env_pairs( + pairs: list[tuple[str, str]], + override_keys: set[str], + warn_keys: set[str], +) -> None: + for key, value in pairs: + if key in override_keys: + os.environ[key] = value + continue + _warn_ignored_env_key(key, value, warn_keys) + os.environ.setdefault(key, value) + + def _load_env(cfg: Config | None) -> None: """Load .env file from config if specified, so OPENAI_API_KEY / XAI_API_KEY etc. are available. @@ -59,27 +90,12 @@ def _load_env(cfg: Config | None) -> None: pairs = _parse_env_file_pairs(cfg.env_file) mode = _docgen_env_override_mode() if mode == "all": - for k, v in pairs: - os.environ[k] = v + _apply_all_env_pairs(pairs) return override_keys = mode if isinstance(mode, set) else set() from docgen.ai_client import conflicting_api_key_envs - warn_keys = set(conflicting_api_key_envs()) - for k, v in pairs: - if k in override_keys: - os.environ[k] = v - continue - if k in warn_keys and v and os.environ.get(k): - click.echo( - f"[docgen] {k} already set in the process environment; " - "env_file value is ignored for this key (shell wins). " - f"Unset {k} or set DOCGEN_ENV_OVERRIDES=1 to load all keys " - f"from env_file, or DOCGEN_ENV_OVERRIDES={k} to override just " - "this key.", - err=True, - ) - os.environ.setdefault(k, v) + _apply_selective_env_pairs(pairs, override_keys, set(conflicting_api_key_envs())) def _require_config(ctx: click.Context) -> Config: @@ -112,6 +128,69 @@ def _cli_version_string(ctx: click.Context, param: click.Parameter, value: bool) ctx.exit() +def _resolve_cli_repo(repo_spec: str | None) -> Path | None: + if not repo_spec: + return None + from docgen.target_repo import TargetRepoError, resolve_repo + + try: + repo_root = resolve_repo(repo_spec) + except TargetRepoError as exc: + raise click.ClickException(str(exc)) from exc + click.echo(f"[docgen] target repo: {repo_root}", err=True) + return repo_root + + +def _config_path_under_repo(config_path: str, repo_root: Path | None) -> Path: + cfg_path = Path(config_path) + if cfg_path.is_absolute() or repo_root is None: + return cfg_path + nested = repo_root / cfg_path + if nested.exists(): + return nested + return cfg_path + + +def _config_from_repo(ctx: click.Context, repo_root: Path) -> Config | None: + from docgen.target_repo import find_bundle_yaml + + found = find_bundle_yaml(repo_root) + if found is not None: + click.echo(f"[docgen] bundle: {found}", err=True) + return Config.from_yaml(found) + if ctx.invoked_subcommand != "init": + click.echo( + f"[docgen] No docgen.yaml under {repo_root}; " + "run `docgen --repo … init --defaults` to scaffold " + "docs/demos (docgen is not copied into the consumer src/).", + err=True, + ) + return None + + +def _load_cli_config( + ctx: click.Context, + config_path: str | None, + repo_root: Path | None, +) -> Config | None: + try: + if config_path: + return Config.from_yaml(_config_path_under_repo(config_path, repo_root)) + if repo_root is not None: + return _config_from_repo(ctx, repo_root) + return Config.discover() + except FileNotFoundError: + click.echo( + "[docgen] No docgen.yaml found in this directory tree; pass " + "`--config PATH/to/docgen.yaml`, `--repo PATH_OR_URL`, or `cd` " + "to your demos bundle directory.", + err=True, + ) + return None + except ConfigError as exc: + raise click.ClickException(str(exc)) from exc + + @click.group() @click.option( "--config", @@ -158,52 +237,9 @@ def main( plus ``XAI_API_KEY`` uses xAI. Run ``docgen ai-status`` to see the resolved key. """ ctx.ensure_object(dict) - repo_root = None - if repo_spec: - from docgen.target_repo import TargetRepoError, resolve_repo - - try: - repo_root = resolve_repo(repo_spec) - except TargetRepoError as exc: - raise click.ClickException(str(exc)) from exc - click.echo(f"[docgen] target repo: {repo_root}", err=True) + repo_root = _resolve_cli_repo(repo_spec) ctx.obj["repo_root"] = repo_root - - cfg = None - try: - if config_path: - cfg_path = Path(config_path) - if not cfg_path.is_absolute() and repo_root is not None: - nested = repo_root / cfg_path - if nested.exists(): - cfg_path = nested - cfg = Config.from_yaml(cfg_path) - elif repo_root is not None: - from docgen.target_repo import find_bundle_yaml - - found = find_bundle_yaml(repo_root) - if found is not None: - click.echo(f"[docgen] bundle: {found}", err=True) - cfg = Config.from_yaml(found) - elif ctx.invoked_subcommand != "init": - click.echo( - f"[docgen] No docgen.yaml under {repo_root}; " - "run `docgen --repo … init --defaults` to scaffold " - "docs/demos (docgen is not copied into the consumer src/).", - err=True, - ) - else: - cfg = Config.discover() - except FileNotFoundError: - cfg = None - click.echo( - "[docgen] No docgen.yaml found in this directory tree; pass " - "`--config PATH/to/docgen.yaml`, `--repo PATH_OR_URL`, or `cd` " - "to your demos bundle directory.", - err=True, - ) - except ConfigError as exc: - raise click.ClickException(str(exc)) from exc + cfg = _load_cli_config(ctx, config_path, repo_root) ctx.obj["config"] = cfg _load_env(cfg) @@ -459,6 +495,50 @@ def manim(ctx: click.Context, scene: str | None) -> None: raise click.ClickException(str(exc)) from exc +def _compose_target_ids( + cfg: Config, + segments: tuple[str, ...], + only_visual_types: tuple[str, ...], +) -> list[str]: + from docgen.compose import filter_segments_by_visual_types + + target = list(segments) if segments else list(cfg.segments_all) + target = filter_segments_by_visual_types(cfg, target, only_visual_types) + if only_visual_types and not target: + raise click.ClickException( + "[compose] No segments left after --only-visual-type filter " + f"({', '.join(only_visual_types)})." + ) + if not target: + raise click.ClickException( + "[compose] no segments to compose — set segments.all, or pass segment ids" + ) + return target + + +def _segment_has_visual_type(cfg: Config, sid: str) -> bool: + row = cfg.visual_map.get(sid) + if not isinstance(row, dict): + return False + return bool(str(row.get("type", "")).strip()) + + +def _expected_compose_count(cfg: Config, target: list[str]) -> int: + mapped = [sid for sid in target if _segment_has_visual_type(cfg, sid)] + if mapped: + return len(mapped) + return len(target) + + +def _require_composed_count(cfg: Config, target: list[str], composed: int) -> None: + expected = _expected_compose_count(cfg, target) + if expected and composed < expected: + raise click.ClickException( + f"[compose] produced {composed}/{expected} segment videos " + "(missing audio or visuals)." + ) + + @main.command() @click.argument("segments", nargs=-1) @click.option( @@ -489,38 +569,17 @@ def compose( Pass segment IDs to compose specific ones, or omit for ``segments.all`` (same set as ``generate-all``). """ - from docgen.compose import ComposeError, Composer, filter_segments_by_visual_types + from docgen.compose import ComposeError, Composer cfg = _require_config(ctx) comp = Composer(cfg, ffmpeg_timeout_sec=ffmpeg_timeout) - target = list(segments) if segments else list(cfg.segments_all) - target = filter_segments_by_visual_types(cfg, target, only_visual_types) - if only_visual_types and not target: - raise click.ClickException( - "[compose] No segments left after --only-visual-type filter " - f"({', '.join(only_visual_types)})." - ) - if not target: - raise click.ClickException( - "[compose] no segments to compose — set segments.all, or pass segment ids" - ) + target = _compose_target_ids(cfg, segments, only_visual_types) click.echo(f"=== Composing {len(target)} segments ===") try: composed = comp.compose_segments(target) except ComposeError as exc: raise click.ClickException(str(exc)) from exc - mapped = [ - sid - for sid in target - if isinstance(cfg.visual_map.get(sid), dict) - and str(cfg.visual_map[sid].get("type", "")).strip() - ] - expected = len(mapped) if mapped else len(target) - if expected and composed < expected: - raise click.ClickException( - f"[compose] produced {composed}/{expected} segment videos " - "(missing audio or visuals)." - ) + _require_composed_count(cfg, target, composed) @main.command() @@ -579,6 +638,89 @@ def lint(ctx: click.Context, segment: str | None) -> None: raise SystemExit(1) +def _reject_narration_selector( + *, + all_segments: bool, + segment: str | None, + revise: bool, + revision_notes: str, +) -> None: + if all_segments and segment: + raise click.ClickException("--all and --segment are mutually exclusive") + if not all_segments and not segment: + raise click.ClickException("provide --segment or --all") + notes = revision_notes if revision_notes is not None else "" + if revise and not str(notes).strip(): + raise click.ClickException("--revise requires --revision-notes") + + +def _generate_one_narration( + cfg: Config, + seg_str: str, + *, + extra_paths: tuple[str, ...], + extra_hints: tuple[str, ...], + revision_notes: str, + mode: str, + dry_run: bool, + write_force: bool, + all_segments: bool, +) -> None: + from docgen.narrate_from_source import generate_narration_markdown, write_narration_markdown + + try: + body = generate_narration_markdown( + cfg, + seg_str, + extra_paths=list(extra_paths), + extra_hints=list(extra_hints), + revision_notes=revision_notes, + mode=mode, + ) + except ValueError as exc: + raise click.ClickException(f"segment {seg_str}: {exc}") from exc + if dry_run: + click.echo(body) + return + try: + out = write_narration_markdown(cfg, seg_str, body, force=write_force) + except FileExistsError as exc: + raise click.ClickException(f"segment {seg_str}: {exc} (use --force)") from exc + if all_segments: + click.echo(f" -> {out}") + return + click.echo(f"[narration-generate] wrote {out}") + + +def _narration_generate_all( + cfg: Config, + *, + extra_paths: tuple[str, ...], + extra_hints: tuple[str, ...], + revision_notes: str, + mode: str, + dry_run: bool, + write_force: bool, +) -> None: + ids = list(cfg.segments_all) + if not ids: + raise click.ClickException("segments.all is empty in docgen.yaml") + for seg_id in ids: + seg_str = str(seg_id) + click.echo(f"=== narration-generate --segment {seg_str} ===") + _generate_one_narration( + cfg, + seg_str, + extra_paths=extra_paths, + extra_hints=extra_hints, + revision_notes=revision_notes, + mode=mode, + dry_run=dry_run, + write_force=write_force, + all_segments=True, + ) + + @main.command("narration-generate") @click.option( "--segment", @@ -651,54 +793,105 @@ def narration_generate( ``--revise`` reads the current narration file and applies ``--revision-notes`` with minimal edits (same contract as the wizard Revise button). """ - if all_segments and segment: - raise click.ClickException("--all and --segment are mutually exclusive") - if not all_segments and not segment: - raise click.ClickException("provide --segment or --all") - if revise and not str(revision_notes or "").strip(): - raise click.ClickException("--revise requires --revision-notes") - - from docgen.narrate_from_source import generate_narration_markdown, write_narration_markdown + _reject_narration_selector( + all_segments=all_segments, + segment=segment, + revise=revise, + revision_notes=revision_notes, + ) cfg = _require_config(ctx) _echo_ai_status(cfg) mode = "revise" if revise else "generate" # Revising always overwrites the existing script. - write_force = force or revise - - def _one(seg_str: str) -> None: - try: - body = generate_narration_markdown( - cfg, - seg_str, - extra_paths=list(extra_paths), - extra_hints=list(extra_hints), - revision_notes=revision_notes, - mode=mode, - ) - except ValueError as exc: - raise click.ClickException(f"segment {seg_str}: {exc}") from exc - if dry_run: - click.echo(body) - return - try: - out = write_narration_markdown(cfg, seg_str, body, force=write_force) - except FileExistsError as exc: - raise click.ClickException(f"segment {seg_str}: {exc} (use --force)") from exc - click.echo(f" -> {out}" if all_segments else f"[narration-generate] wrote {out}") - + write_force = True if revise else force if all_segments: - ids = list(cfg.segments_all) - if not ids: - raise click.ClickException("segments.all is empty in docgen.yaml") - for seg_id in ids: - seg_str = str(seg_id) - click.echo(f"=== narration-generate --segment {seg_str} ===") - _one(seg_str) + _narration_generate_all( + cfg, + extra_paths=extra_paths, + extra_hints=extra_hints, + revision_notes=revision_notes, + mode=mode, + dry_run=dry_run, + write_force=write_force, + ) return assert segment is not None # for type-checker - _one(segment) + _generate_one_narration( + cfg, + segment, + extra_paths=extra_paths, + extra_hints=extra_hints, + revision_notes=revision_notes, + mode=mode, + dry_run=dry_run, + write_force=write_force, + all_segments=False, + ) + + +def _reject_scene_compile_selector(all_specs: bool, spec_path: Path | None) -> None: + if all_specs and spec_path is not None: + raise click.ClickException("Pass SPEC_PATH or --all, not both.") + if not all_specs and spec_path is None: + raise click.ClickException("Pass SPEC_PATH or --all.") + + +def _scene_compile_paths( + cfg: Config, + all_specs: bool, + spec_path: Path | None, +) -> list[Path]: + from docgen.scene_retime import list_scene_spec_paths + + paths = list_scene_spec_paths(cfg) if all_specs else [spec_path] + if not paths: + raise click.ClickException("No animations/specs/*.scene.yaml files found.") + return [path for path in paths if path is not None] + + +def _compile_one_scene_spec( + cfg: Config, + path: Path, + *, + all_specs: bool, + retime: bool, + dry_run: bool, +) -> str | None: + """Compile one spec. Return the file name when ``--all`` records a failure.""" + from docgen.manim_scene_support import SceneGenerationError + from docgen.scene_retime import retime_compile_spec + from docgen.scene_spec import SceneSpecError + + try: + result = retime_compile_spec(cfg, path, dry_run=dry_run) + except (SceneGenerationError, SceneSpecError) as exc: + if not all_specs: + raise click.ClickException(str(exc)) from exc + click.echo(f"[scene-compile] FAIL {path.name}: {exc}", err=True) + return path.name + if dry_run: + click.echo(result["class_block"], nl=False) + if all_specs: + click.echo(f"\n--- end {path.name} ---\n") + return None + suffix = "" + if retime or all_specs: + suffix = ", retime" + click.echo( + f"[scene-compile] wrote {result['class_name']} to {result['scenes_path']} " + f"(segment {result['segment_id']} → timing_key {result['timing_key']!r}" + f"{suffix})" + ) + return None + + +def _fail_scene_compile_all(failures: list[str]) -> None: + if failures: + raise click.ClickException( + f"scene-compile --all: {len(failures)} failed: " + ", ".join(failures) + ) @main.command("scene-compile") @@ -746,68 +939,287 @@ def scene_compile( ``generate-all``, which retimes existing specs automatically) so beat sync uses fresh ``timing.json`` without calling OpenAI. """ - if all_specs and spec_path is not None: - raise click.ClickException("Pass SPEC_PATH or --all, not both.") - if not all_specs and spec_path is None: - raise click.ClickException("Pass SPEC_PATH or --all.") - - from docgen.manim_scene_support import SceneGenerationError - from docgen.scene_retime import list_scene_spec_paths, retime_compile_spec - from docgen.scene_spec import SceneSpecError - + _reject_scene_compile_selector(all_specs, spec_path) cfg = _require_config(ctx) # --retime is the same compile path (label sync + pacing gate); the flag # documents intent and is the recommended post-timestamps invocation. - _ = retime - - paths = list_scene_spec_paths(cfg) if all_specs else [spec_path] - if not paths: - raise click.ClickException("No animations/specs/*.scene.yaml files found.") - + paths = _scene_compile_paths(cfg, all_specs, spec_path) failures: list[str] = [] for path in paths: - assert path is not None - try: - result = retime_compile_spec(cfg, path, dry_run=dry_run) - except (SceneGenerationError, SceneSpecError) as exc: - if all_specs: - click.echo(f"[scene-compile] FAIL {path.name}: {exc}", err=True) - failures.append(path.name) - continue - raise click.ClickException(str(exc)) from exc - if dry_run: - click.echo(result["class_block"], nl=False) - if all_specs: - click.echo(f"\n--- end {path.name} ---\n") - continue - click.echo( - f"[scene-compile] wrote {result['class_name']} to {result['scenes_path']} " - f"(segment {result['segment_id']} → timing_key {result['timing_key']!r}" - f"{', retime' if retime or all_specs else ''})" + failed = _compile_one_scene_spec( + cfg, + path, + all_specs=all_specs, + retime=retime, + dry_run=dry_run, ) - if failures: + if failed: + failures.append(failed) + _fail_scene_compile_all(failures) + + +def _reject_scene_spec_selector( + *, + dry_run: bool, + print_only: bool, + all_segments: bool, + segment: str | None, +) -> None: + if dry_run and print_only: + raise click.ClickException("--dry-run and --print-only are mutually exclusive") + if all_segments and segment: + raise click.ClickException("--all and --segment are mutually exclusive") + if not all_segments and not segment: + raise click.ClickException("provide --segment or --all") + + +def _reject_scene_spec_all_options( + *, + all_segments: bool, + class_name_override: str | None, + output_path: Path | None, +) -> None: + if all_segments and class_name_override: + raise click.ClickException("--class-name cannot be combined with --all") + if all_segments and output_path: raise click.ClickException( - f"scene-compile --all: {len(failures)} failed: " + ", ".join(failures) + "--output cannot be combined with --all (per-segment paths are used)" ) -@main.command("scene-spec-generate") -@click.option( - "--segment", - "segment", - default=None, - help="Segment id (e.g. 01). Mutually exclusive with --all.", -) -@click.option( - "--all", - "all_segments", - is_flag=True, - help="Generate a scene spec for every segment in segments.all whose visual_map " - "type is manim (or unset) and that has no scripts/**.py. " - "--class-name is ignored with --all.", -) -@click.option( - "--class-name", +def _call_generate_scene_spec( + cfg: Config, + sid: str, + *, + extra_paths: tuple[str, ...], + extra_hints: tuple[str, ...], + class_name_override: str | None, + dry_run: bool, + model: str | None, + error_prefix: str, +): + from docgen.manim_scene_support import SceneGenerationError + from docgen.scene_spec_generate import generate_scene_spec + + try: + return generate_scene_spec( + cfg, + sid, + extra_paths=list(extra_paths), + extra_hints=list(extra_hints), + class_name_override=class_name_override, + dry_run=dry_run, + model_override=model, + ) + except SceneGenerationError as exc: + detail = f"{error_prefix}{exc}" if error_prefix else str(exc) + raise click.ClickException(detail) from exc + + +def _compile_generated_scene_spec(cfg: Config, result) -> None: + from docgen.manim_scene_support import SceneGenerationError + from docgen.scene_spec_generate import ( + inject_class_block_into_scenes_py, + linted_class_block_from_spec, + ) + + try: + class_block, merged = linted_class_block_from_spec( + cfg, result.spec, timing_key=result.seg_name + ) + inject_class_block_into_scenes_py( + cfg, + seg_id=merged["segment_id"], + class_name=merged["class_name"], + class_block=class_block, + ) + except SceneGenerationError as exc: + raise click.ClickException(str(exc)) from exc + click.echo( + f"[scene-spec-generate] compiled → {cfg.animations_dir / 'scenes.py'} " + f"({result.class_name}, timing_key {result.seg_name!r})" + ) + + +def _write_default_scene_spec(cfg: Config, result) -> None: + specs_dir = cfg.animations_dir / "specs" + specs_dir.mkdir(parents=True, exist_ok=True) + wpath = specs_dir / f"{result.seg_name}.scene.yaml" + wpath.write_text(result.yaml_text, encoding="utf-8") + click.echo(f"[scene-spec-generate] wrote {wpath}") + + +def _emit_all_segment_scene_spec( + cfg: Config, + sid: str, + *, + extra_paths: tuple[str, ...], + extra_hints: tuple[str, ...], + class_name_override: str | None, + dry_run: bool, + print_only: bool, + do_compile: bool, + model: str | None, +) -> None: + res = _call_generate_scene_spec( + cfg, + sid, + extra_paths=extra_paths, + extra_hints=extra_hints, + class_name_override=class_name_override, + dry_run=dry_run, + model=model, + error_prefix=f"segment {sid}: ", + ) + if dry_run: + click.echo(res.prompt) + return + if print_only: + click.echo(res.yaml_text, nl=False) + else: + _write_default_scene_spec(cfg, res) + if do_compile: + _compile_generated_scene_spec(cfg, res) + + +def _single_scene_spec_write_path( + cfg: Config, + result, + *, + print_only: bool, + output_path: Path | None, +) -> Path | None: + if not print_only: + specs_dir = cfg.animations_dir / "specs" + specs_dir.mkdir(parents=True, exist_ok=True) + return output_path or (specs_dir / f"{result.seg_name}.scene.yaml") + if output_path is not None: + return output_path + return None + + +def _emit_single_scene_spec( + cfg: Config, + segment: str, + *, + extra_paths: tuple[str, ...], + extra_hints: tuple[str, ...], + class_name_override: str | None, + dry_run: bool, + print_only: bool, + output_path: Path | None, + do_compile: bool, + model: str | None, +) -> None: + result = _call_generate_scene_spec( + cfg, + segment, + extra_paths=extra_paths, + extra_hints=extra_hints, + class_name_override=class_name_override, + dry_run=dry_run, + model=model, + error_prefix="", + ) + if dry_run: + click.echo(result.prompt) + return + if print_only: + click.echo(result.yaml_text, nl=False) + write_path = _single_scene_spec_write_path( + cfg, result, print_only=print_only, output_path=output_path + ) + if write_path is not None: + write_path.parent.mkdir(parents=True, exist_ok=True) + write_path.write_text(result.yaml_text, encoding="utf-8") + click.echo(f"[scene-spec-generate] wrote {write_path}") + if do_compile: + _compile_generated_scene_spec(cfg, result) + + +def _scene_spec_all_skip_reason(cfg: Config, sid: str) -> str | None: + script_path = cfg.find_segment_asset(cfg.base_dir / "scripts", sid, ".py") + if script_path: + return "existing capture script" + vm_row = cfg.visual_map.get(sid) + if isinstance(vm_row, dict): + vtype = str(vm_row.get("type", "")).strip().lower() + if vtype and vtype != "manim": + return f"visual_map type is {vtype!r} (not manim)" + return None + + +def _segment_label(names: dict[str, str], seg_id: str, sid: str) -> str: + return names.get(sid) or names.get(seg_id) or sid + + +def _fail_scene_spec_all(failures: list[str]) -> None: + if failures: + raise click.ClickException( + f"scene-spec-generate --all: {len(failures)} segment(s) failed: " + + ", ".join(failures) + ) + + +def _scene_spec_generate_all( + cfg: Config, + *, + extra_paths: tuple[str, ...], + extra_hints: tuple[str, ...], + class_name_override: str | None, + dry_run: bool, + print_only: bool, + do_compile: bool, + model: str | None, +) -> None: + ids = list(cfg.segments_all) + if not ids: + raise click.ClickException("segments.all is empty in docgen.yaml") + names = cfg.segment_names + failures: list[str] = [] + for seg_id in ids: + sid = str(seg_id) + name = _segment_label(names, seg_id, sid) + reason = _scene_spec_all_skip_reason(cfg, sid) + if reason: + click.echo(f"[scene-spec-generate --all] skip {sid} ({name}): {reason}") + continue + click.echo(f"=== scene-spec-generate --segment {sid} ===") + try: + _emit_all_segment_scene_spec( + cfg, + sid, + extra_paths=extra_paths, + extra_hints=extra_hints, + class_name_override=class_name_override, + dry_run=dry_run, + print_only=print_only, + do_compile=do_compile, + model=model, + ) + except click.ClickException as exc: + click.echo(f"[scene-spec-generate --all] FAIL {sid}: {exc}", err=True) + failures.append(sid) + _fail_scene_spec_all(failures) + + +@main.command("scene-spec-generate") +@click.option( + "--segment", + "segment", + default=None, + help="Segment id (e.g. 01). Mutually exclusive with --all.", +) +@click.option( + "--all", + "all_segments", + is_flag=True, + help="Generate a scene spec for every segment in segments.all whose visual_map " + "type is manim (or unset) and that has no scripts/**.py. " + "--class-name is ignored with --all.", +) +@click.option( + "--class-name", "class_name_override", default=None, help="Override class name (default: manim_scene_generation.segments..class_name or CamelCase+Scene). Ignored with --all.", @@ -873,157 +1285,157 @@ def scene_spec_generate_cmd( The model outputs YAML only (see :mod:`docgen.scene_spec`); layout is deterministic in :func:`docgen.scene_spec.compile_scene_class`. """ - if dry_run and print_only: - raise click.ClickException("--dry-run and --print-only are mutually exclusive") - if all_segments and segment: - raise click.ClickException("--all and --segment are mutually exclusive") - if not all_segments and not segment: - raise click.ClickException("provide --segment or --all") - if all_segments and class_name_override: - raise click.ClickException("--class-name cannot be combined with --all") - if all_segments and output_path: - raise click.ClickException("--output cannot be combined with --all (per-segment paths are used)") - - from docgen.manim_scene_support import SceneGenerationError - from docgen.scene_spec_generate import ( - generate_scene_spec, - inject_class_block_into_scenes_py, - linted_class_block_from_spec, + _reject_scene_spec_selector( + dry_run=dry_run, + print_only=print_only, + all_segments=all_segments, + segment=segment, + ) + _reject_scene_spec_all_options( + all_segments=all_segments, + class_name_override=class_name_override, + output_path=output_path, ) - cfg = _require_config(ctx) if not dry_run: _echo_ai_status(cfg) - - def _one_sid(sid: str) -> None: - try: - res = generate_scene_spec( - cfg, - sid, - extra_paths=list(extra_paths), - extra_hints=list(extra_hints), - class_name_override=class_name_override, - dry_run=dry_run, - model_override=model, - ) - except SceneGenerationError as exc: - raise click.ClickException(f"segment {sid}: {exc}") from exc - if dry_run: - click.echo(res.prompt) - return - if print_only: - click.echo(res.yaml_text, nl=False) - else: - specs_dir = cfg.animations_dir / "specs" - specs_dir.mkdir(parents=True, exist_ok=True) - wpath = specs_dir / f"{res.seg_name}.scene.yaml" - wpath.write_text(res.yaml_text, encoding="utf-8") - click.echo(f"[scene-spec-generate] wrote {wpath}") - if do_compile: - try: - class_block, merged = linted_class_block_from_spec( - cfg, res.spec, timing_key=res.seg_name - ) - inject_class_block_into_scenes_py( - cfg, - seg_id=merged["segment_id"], - class_name=merged["class_name"], - class_block=class_block, - ) - except SceneGenerationError as exc: - raise click.ClickException(str(exc)) from exc - click.echo( - f"[scene-spec-generate] compiled → {cfg.animations_dir / 'scenes.py'} " - f"({res.class_name}, timing_key {res.seg_name!r})" - ) - if all_segments: - ids = list(cfg.segments_all) - if not ids: - raise click.ClickException("segments.all is empty in docgen.yaml") - names = cfg.segment_names - scripts_dir = cfg.base_dir / "scripts" - failures: list[str] = [] - for seg_id in ids: - sid = str(seg_id) - name = names.get(sid) or names.get(seg_id) or sid - script_path = cfg.find_segment_asset(scripts_dir, sid, ".py") - if script_path: - click.echo( - f"[scene-spec-generate --all] skip {sid} ({name}): existing capture script" - ) - continue - vm_row = cfg.visual_map.get(sid) - if isinstance(vm_row, dict): - vtype = str(vm_row.get("type", "")).strip().lower() - if vtype and vtype != "manim": - click.echo( - f"[scene-spec-generate --all] skip {sid} ({name}): " - f"visual_map type is {vtype!r} (not manim)" - ) - continue - click.echo(f"=== scene-spec-generate --segment {sid} ===") - try: - _one_sid(sid) - except click.ClickException as exc: - click.echo(f"[scene-spec-generate --all] FAIL {sid}: {exc}", err=True) - failures.append(sid) - if failures: - raise click.ClickException( - f"scene-spec-generate --all: {len(failures)} segment(s) failed: " - + ", ".join(failures) - ) - return - - assert segment is not None # type-checker - try: - result = generate_scene_spec( + _scene_spec_generate_all( cfg, - segment, - extra_paths=list(extra_paths), - extra_hints=list(extra_hints), + extra_paths=extra_paths, + extra_hints=extra_hints, class_name_override=class_name_override, dry_run=dry_run, - model_override=model, + print_only=print_only, + do_compile=do_compile, + model=model, ) - except SceneGenerationError as exc: - raise click.ClickException(str(exc)) from exc - - if dry_run: - click.echo(result.prompt) return + assert segment is not None # type-checker + _emit_single_scene_spec( + cfg, + segment, + extra_paths=extra_paths, + extra_hints=extra_hints, + class_name_override=class_name_override, + dry_run=dry_run, + print_only=print_only, + output_path=output_path, + do_compile=do_compile, + model=model, + ) - if print_only: - click.echo(result.yaml_text, nl=False) - write_path: Path | None = None - if not print_only: - specs_dir = cfg.animations_dir / "specs" - specs_dir.mkdir(parents=True, exist_ok=True) - write_path = output_path or (specs_dir / f"{result.seg_name}.scene.yaml") - elif output_path: - write_path = output_path +def _reject_image_selector( + segment: str | None, + all_segments: bool, + spec_path: Path | None, +) -> None: + chosen = [bool(segment), all_segments, spec_path is not None] + if sum(chosen) != 1: + raise click.ClickException("provide exactly one of --segment, --all, or --spec") - if write_path is not None: - write_path.parent.mkdir(parents=True, exist_ok=True) - write_path.write_text(result.yaml_text, encoding="utf-8") - click.echo(f"[scene-spec-generate] wrote {write_path}") - if do_compile: +def _manim_segment_ids(cfg: Config) -> list[str]: + ids: list[str] = [] + for sid in cfg.segments_all: + row = cfg.visual_map.get(sid) + if not isinstance(row, dict): + continue + vtype = str(row.get("type", "")).strip().lower() + if vtype == "manim": + ids.append(str(sid)) + return ids + + +def _image_specs_for_all(cfg: Config) -> list[Path] | None: + from docgen.image_generate import spec_files_for_bundle + + targets = spec_files_for_bundle(cfg) + if targets: + return targets + manim_ids = _manim_segment_ids(cfg) + if manim_ids: + raise click.ClickException( + "[image-generate] no *.scene.yaml specs in animations/specs/ " + f"but manim segments exist ({', '.join(manim_ids)}) — " + "run `docgen scene-spec-generate` first" + ) + click.echo("[image-generate] no *.scene.yaml specs found in animations/specs/") + return None + + +def _image_spec_for_segment(cfg: Config, segment: str) -> list[Path]: + stem = cfg.resolve_segment_name(segment) + candidate = cfg.animations_dir / "specs" / f"{stem}.scene.yaml" + if not candidate.is_file(): + raise click.ClickException( + f"spec not found: {candidate} — run `docgen scene-spec-generate --segment {segment}` " + "or author the spec by hand." + ) + return [candidate] + + +def _image_spec_targets( + cfg: Config, + segment: str | None, + all_segments: bool, + spec_path: Path | None, +) -> list[Path] | None: + if spec_path is not None: + return [spec_path] + if all_segments: + return _image_specs_for_all(cfg) + assert segment is not None + return _image_spec_for_segment(cfg, segment) + + +def _echo_image_results(target_name: str, results: list) -> int: + if not results: + click.echo(f"[image-generate] {target_name}: no image elements") + return 0 + written = 0 + for res in results: + if res.status == "dry-run": + click.echo(f"[image-generate] {target_name}: would generate {res.relpath}") + click.echo(f" prompt: {res.prompt}") + elif res.status == "exists": + click.echo( + f"[image-generate] {target_name}: {res.relpath} exists (skip; use --force)" + ) + else: + click.echo(f"[image-generate] {target_name}: wrote {res.path}") + written += 1 + return written + + +def _generate_images_for_targets( + cfg: Config, + targets: list[Path], + *, + force: bool, + dry_run: bool, + model: str | None, + size: str | None, +) -> int: + from docgen.image_generate import ImageGenerationError, generate_images_for_spec + from docgen.scene_spec import SceneSpecError + + total = 0 + for target in targets: try: - class_block, merged = linted_class_block_from_spec(cfg, result.spec, timing_key=result.seg_name) - inject_class_block_into_scenes_py( + results = generate_images_for_spec( cfg, - seg_id=merged["segment_id"], - class_name=merged["class_name"], - class_block=class_block, + target, + force=force, + dry_run=dry_run, + model_override=model, + size_override=size, ) - except SceneGenerationError as exc: + except (ImageGenerationError, SceneSpecError) as exc: raise click.ClickException(str(exc)) from exc - click.echo( - f"[scene-spec-generate] compiled → {cfg.animations_dir / 'scenes.py'} " - f"({result.class_name}, timing_key {result.seg_name!r})" - ) + total += _echo_image_results(target.name, results) + return total @main.command("image-generate") @@ -1078,82 +1490,86 @@ def image_generate_cmd( box). This command writes the referenced PNG under the bundle directory so ``docgen manim`` can render them. Existing assets are kept unless --force. """ - chosen = [bool(segment), all_segments, spec_path is not None] - if sum(chosen) != 1: - raise click.ClickException("provide exactly one of --segment, --all, or --spec") - - from docgen.image_generate import ( - ImageGenerationError, - generate_images_for_spec, - spec_files_for_bundle, - ) - from docgen.scene_spec import SceneSpecError - + _reject_image_selector(segment, all_segments, spec_path) cfg = _require_config(ctx) if not dry_run: _echo_ai_status(cfg) - - if spec_path is not None: - targets = [spec_path] - elif all_segments: - targets = spec_files_for_bundle(cfg) - if not targets: - manim_ids = [ - sid - for sid in cfg.segments_all - if isinstance(cfg.visual_map.get(sid), dict) - and str(cfg.visual_map[sid].get("type", "")).strip().lower() == "manim" - ] - if manim_ids: - raise click.ClickException( - "[image-generate] no *.scene.yaml specs in animations/specs/ " - f"but manim segments exist ({', '.join(manim_ids)}) — " - "run `docgen scene-spec-generate` first" - ) - click.echo("[image-generate] no *.scene.yaml specs found in animations/specs/") - return - else: - stem = cfg.resolve_segment_name(str(segment)) - candidate = cfg.animations_dir / "specs" / f"{stem}.scene.yaml" - if not candidate.is_file(): - raise click.ClickException( - f"spec not found: {candidate} — run `docgen scene-spec-generate --segment {segment}` " - "or author the spec by hand." - ) - targets = [candidate] - - total = 0 - for target in targets: - try: - results = generate_images_for_spec( - cfg, - target, - force=force, - dry_run=dry_run, - model_override=model, - size_override=size, - ) - except (ImageGenerationError, SceneSpecError) as exc: - raise click.ClickException(str(exc)) from exc - if not results: - click.echo(f"[image-generate] {target.name}: no image elements") - continue - for res in results: - if res.status == "dry-run": - click.echo(f"[image-generate] {target.name}: would generate {res.relpath}") - click.echo(f" prompt: {res.prompt}") - if res.effective_prompt and res.effective_prompt != res.prompt: - click.echo(" aligned prompt:") - click.echo(res.effective_prompt) - elif res.status == "exists": - click.echo(f"[image-generate] {target.name}: {res.relpath} exists (skip; use --force)") - else: - click.echo(f"[image-generate] {target.name}: wrote {res.path}") - total += 1 + targets = _image_spec_targets(cfg, segment, all_segments, spec_path) + if targets is None: + return + total = _generate_images_for_targets( + cfg, + targets, + force=force, + dry_run=dry_run, + model=model, + size=size, + ) if not dry_run: click.echo(f"[image-generate] generated {total} asset(s)") +def _print_yaml_gaps(cfg: Config, raw: dict) -> None: + from docgen import yaml_generate as yg + + gaps = yg.narration_not_in_segments(raw, cfg.narration_dir) + if not gaps: + click.echo("[yaml-generate] no narration segments missing from segments.all") + return + for seg_id, stem in gaps: + click.echo(f"gap: {seg_id} ({stem}.md) not in segments.all") + raise SystemExit(1) + + +def _merge_yaml_defaults(raw: dict, cfg: Config, merge_hint_segments: bool) -> list[str]: + from docgen import yaml_generate as yg + + try: + return list(yg.merge_defaults(raw, cfg, merge_hint_segments=merge_hint_segments)) + except ValueError as exc: + raise click.ClickException(str(exc)) from exc + + +def _apply_yaml_llm(raw: dict, cfg: Config, model: str | None) -> str: + from docgen import yaml_generate as yg + + _echo_ai_status(cfg) + try: + hints = yg.generate_llm_hints(cfg, model=model) + except ValueError as exc: + raise click.ClickException(str(exc)) from exc + except RuntimeError as exc: + raise click.ClickException(str(exc)) from exc + yg.apply_llm_hints(raw, hints) + return "tts.instructions + wizard.system_prompt: refreshed via chat" + + +def _echo_merged_yaml(raw: dict) -> None: + click.echo("--- merged yaml ---") + click.echo( + yaml.safe_dump( + raw, default_flow_style=False, sort_keys=False, allow_unicode=True, width=120 + ), + nl=False, + ) + + +def _finish_yaml_generate(path: Path, raw: dict, changes: list[str], dry_run: bool) -> None: + from docgen import yaml_generate as yg + + if not changes and not dry_run: + click.echo("[yaml-generate] nothing to do (already up to date)") + return + for line in changes: + click.echo(f"[yaml-generate] {line}") + if dry_run: + _echo_merged_yaml(raw) + return + header = yg.default_header(path) if changes else None + yg.write_docgen_yaml(path, raw, header=header) + click.echo(f"[yaml-generate] wrote {path}") + + @main.command("yaml-generate") @click.option( "--merge-defaults/--no-merge-defaults", @@ -1200,7 +1616,6 @@ def yaml_generate_cmd( Rewrites the config file with PyYAML (comments are not preserved). Use Git to review. """ - from docgen import yaml_generate as yg from docgen.config import load_yaml_mapping cfg = _require_config(ctx) @@ -1211,51 +1626,15 @@ def yaml_generate_cmd( raise click.ClickException(str(exc)) from exc if list_gaps: - gaps = yg.narration_not_in_segments(raw, cfg.narration_dir) - if not gaps: - click.echo("[yaml-generate] no narration segments missing from segments.all") - return - for seg_id, stem in gaps: - click.echo(f"gap: {seg_id} ({stem}.md) not in segments.all") - raise SystemExit(1) + _print_yaml_gaps(cfg, raw) + return changes: list[str] = [] if merge_defaults: - try: - changes.extend(yg.merge_defaults(raw, cfg, merge_hint_segments=merge_hint_segments)) - except ValueError as exc: - raise click.ClickException(str(exc)) from exc + changes.extend(_merge_yaml_defaults(raw, cfg, merge_hint_segments)) if llm: - _echo_ai_status(cfg) - try: - hints = yg.generate_llm_hints(cfg, model=model) - except ValueError as exc: - raise click.ClickException(str(exc)) from exc - except RuntimeError as exc: - raise click.ClickException(str(exc)) from exc - yg.apply_llm_hints(raw, hints) - changes.append("tts.instructions + wizard.system_prompt: refreshed via chat") - - if not changes and not dry_run: - click.echo("[yaml-generate] nothing to do (already up to date)") - return - - for line in changes: - click.echo(f"[yaml-generate] {line}") - - if dry_run: - click.echo("--- merged yaml ---") - click.echo( - yaml.safe_dump( - raw, default_flow_style=False, sort_keys=False, allow_unicode=True, width=120 - ), - nl=False, - ) - return - - header = yg.default_header(path) if changes else None - yg.write_docgen_yaml(path, raw, header=header) - click.echo(f"[yaml-generate] wrote {path}") + changes.append(_apply_yaml_llm(raw, cfg, model)) + _finish_yaml_generate(path, raw, changes, dry_run) @main.command("clean-bundle") @@ -1413,6 +1792,60 @@ def rebuild_after_audio(ctx: click.Context, regen_scene_specs: bool) -> None: _run_pipeline(Pipeline(cfg), skip_tts=True, regen_scene_specs=regen_scene_specs) +def _launch_benchmark_gui(ctx: click.Context, case_id: str | None) -> None: + from docgen.gui.desktop import launch_desktop + + path = "/?view=benchmark" + if case_id: + path += "&case=" + case_id + cfg = ctx.obj.get("config") if ctx.obj else None + launch_desktop(cfg, path=path) + + +def _maybe_update_benchmark_baseline( + scores, + case_id: str | None, + update_baseline: bool, + base_path: Path, +) -> None: + from docgen.scene_benchmark import write_baseline + + if not update_baseline: + return + if case_id: + raise click.ClickException("--update-baseline requires the full corpus (omit --case)") + written = write_baseline(scores, base_path) + click.echo(f"wrote baseline {written}") + + +def _echo_benchmark_text(scores, regressions: list[str], base_path: Path) -> None: + from docgen.scene_benchmark import format_table + + click.echo(format_table(scores)) + if regressions: + click.echo("") + click.echo("regressions vs baseline:") + for note in regressions: + click.echo(f" - {note}") + return + if base_path.is_file(): + click.echo("") + click.echo(f"meets baseline {base_path}") + + +def _echo_benchmark_report( + scores, + regressions: list[str], + report: dict, + fmt: str, + base_path: Path, +) -> None: + if fmt == "json": + click.echo(json.dumps(report, indent=2)) + return + _echo_benchmark_text(scores, regressions, base_path) + + @main.command("benchmark") @click.option( "--case", @@ -1470,21 +1903,13 @@ def benchmark( from docgen.scene_benchmark import ( compare_to_baseline, default_baseline_path, - format_table, load_baseline, run_benchmark, scores_as_json, - write_baseline, ) if gui: - from docgen.gui.desktop import launch_desktop - - path = "/?view=benchmark" - if case_id: - path += "&case=" + case_id - cfg = ctx.obj.get("config") if ctx.obj else None - launch_desktop(cfg, path=path) + _launch_benchmark_gui(ctx, case_id) return try: @@ -1492,28 +1917,13 @@ def benchmark( except ValueError as exc: raise click.ClickException(str(exc)) from exc base_path = baseline_path or default_baseline_path() - if update_baseline: - if case_id: - raise click.ClickException("--update-baseline requires the full corpus (omit --case)") - written = write_baseline(scores, base_path) - click.echo(f"wrote baseline {written}") + _maybe_update_benchmark_baseline(scores, case_id, update_baseline, base_path) baseline = load_baseline(base_path) regressions = compare_to_baseline(scores, baseline) report = scores_as_json(scores, regressions=regressions) if output_path: output_path.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") - if fmt == "json": - click.echo(json.dumps(report, indent=2)) - else: - click.echo(format_table(scores)) - if regressions: - click.echo("") - click.echo("regressions vs baseline:") - for note in regressions: - click.echo(f" - {note}") - elif base_path.is_file(): - click.echo("") - click.echo(f"meets baseline {base_path}") + _echo_benchmark_report(scores, regressions, report, fmt, base_path) if regressions and not update_baseline: raise SystemExit(1) diff --git a/src/docgen/config.py b/src/docgen/config.py index 0aba5d2..9f37b47 100644 --- a/src/docgen/config.py +++ b/src/docgen/config.py @@ -331,8 +331,6 @@ def __post_init__(self) -> None: "story_end", "narration_lint", "subject_beat_coverage", - "image_prompt_alignment", - "image_asset_alignment", ): self._sub_block(validation, nested, label=f"validation.{nested}") # List-valued keys: a string must not be iterated as characters. @@ -487,30 +485,6 @@ def __post_init__(self) -> None: require_yaml_string(ig["size"], label="image_generation.size", source=src) if ig.get("quality") is not None: require_yaml_string(ig["quality"], label="image_generation.quality", source=src) - if ig.get("align_with_docs") is not None: - require_yaml_bool( - ig["align_with_docs"], - label="image_generation.align_with_docs", - source=src, - ) - if ig.get("style") is not None: - require_yaml_string(ig["style"], label="image_generation.style", source=src) - if ig.get("align_review") is not None: - require_yaml_bool( - ig["align_review"], - label="image_generation.align_review", - source=src, - ) - if ig.get("review_model") is not None: - require_yaml_string( - ig["review_model"], label="image_generation.review_model", source=src - ) - if ig.get("align_review_retries") is not None: - require_yaml_number( - ig["align_review_retries"], - label="image_generation.align_review_retries", - source=src, - ) ai = self._block("ai") if ai.get("provider") is not None: require_yaml_string(ai["provider"], label="ai.provider", source=src) @@ -792,40 +766,6 @@ def __post_init__(self) -> None: label="validation.subject_beat_coverage.enabled", source=src, ) - ipa = self._sub_block( - validation, - "image_prompt_alignment", - label="validation.image_prompt_alignment", - ) - if ipa.get("enabled") is not None: - require_yaml_bool( - ipa["enabled"], - label="validation.image_prompt_alignment.enabled", - source=src, - ) - iaa = self._sub_block( - validation, - "image_asset_alignment", - label="validation.image_asset_alignment", - ) - if iaa.get("enabled") is not None: - require_yaml_bool( - iaa["enabled"], - label="validation.image_asset_alignment.enabled", - source=src, - ) - if iaa.get("ocr") is not None: - require_yaml_bool( - iaa["ocr"], - label="validation.image_asset_alignment.ocr", - source=src, - ) - if iaa.get("review") is not None: - require_yaml_bool( - iaa["review"], - label="validation.image_asset_alignment.review", - source=src, - ) def _source_label(self) -> str: return self.yaml_path.name if self.yaml_path else "docgen.yaml" @@ -1010,49 +950,10 @@ def image_generation_config(self) -> dict[str, Any]: defaults: dict[str, Any] = { "model": "gpt-image-1", "size": "1536x1024", - "align_with_docs": True, - "align_review": True, - "align_review_retries": 1, } defaults.update(self._block("image_generation")) return defaults - @property - def image_align_with_docs(self) -> bool: - """When true (default), ground image prompts in narration/source and fail unaligned ones.""" - block = self._block("image_generation") - if "align_with_docs" in block: - return bool(block.get("align_with_docs")) - return True - - @property - def image_prompt_alignment_enabled(self) -> bool: - """When true (default), validate + scene-spec-generate enforce image-prompt grounding. - - Config: ``validation.image_prompt_alignment.enabled`` (bool). - """ - block = self._sub_block( - self._block("validation"), - "image_prompt_alignment", - label="validation.image_prompt_alignment", - ) - if "enabled" in block: - return bool(block.get("enabled")) - return True - - @property - def image_asset_alignment_config(self) -> dict[str, Any]: - """Pixel checks on generated scene images (OCR + optional vision).""" - defaults: dict[str, Any] = {"enabled": True, "ocr": True, "review": False} - defaults.update( - self._sub_block( - self._block("validation"), - "image_asset_alignment", - label="validation.image_asset_alignment", - ) - ) - return defaults - # -- Manim ----------------------------------------------------------------- @property diff --git a/src/docgen/image_align.py b/src/docgen/image_align.py index b31dc4d..cfb210e 100644 --- a/src/docgen/image_align.py +++ b/src/docgen/image_align.py @@ -67,21 +67,35 @@ def ocr_image_text(path: Path) -> str | None: return None +def _verdict_lines(text: str) -> list[str]: + return [ln.strip() for ln in (text or "").splitlines() if ln.strip()] + + +def _verdict_reason(lines: list[str]) -> str: + if len(lines) > 1: + return lines[1] + return lines[0] + + +def _unparsed_verdict(text: str) -> ImageReviewResult: + preview = (text or "").strip() + if len(preview) > 80: + preview = preview[:77] + "..." + return ImageReviewResult(False, f"vision review reply was not PASS/FAIL: {preview!r}") + + def parse_review_verdict(text: str) -> ImageReviewResult: """Parse a PASS/FAIL vision reply. Unparseable text fails closed.""" - lines = [ln.strip() for ln in (text or "").splitlines() if ln.strip()] + lines = _verdict_lines(text) if not lines: return ImageReviewResult(False, "vision review returned empty text") verdict = lines[0].split()[0].upper().strip(".:") - reason = lines[1] if len(lines) > 1 else " ".join(lines[1:]) or lines[0] + reason = _verdict_reason(lines) if verdict == "PASS": return ImageReviewResult(True, reason) if verdict == "FAIL": return ImageReviewResult(False, reason) - preview = (text or "").strip() - if len(preview) > 80: - preview = preview[:77] + "..." - return ImageReviewResult(False, f"vision review reply was not PASS/FAIL: {preview!r}") + return _unparsed_verdict(text) def build_review_user_message( @@ -148,3 +162,249 @@ def review_image_against_docs( except (AIError, RuntimeError, OSError, TypeError, ValueError) as exc: return ImageReviewResult(False, f"vision review failed: {exc}") return parse_review_verdict(text) + + +# Visual-style words an image prompt may use without being documented terms. +_IMAGE_STYLE_TOKENS = frozenset( + """ + diagram diagrams illustration illustrations icon icons flat clean vector + isometric cartoon watercolor sketch photo photographic realistic abstract + artwork image images picture pictures visual background foreground style + colored colour color palette high contrast simple minimal educational + documentary video watermark signature logo logos chrome banner poster + scene board rounded pill diamond arrow arrows flow flowchart infographic + thumbnail render rendering drawing draw depict showing shows show + depicting depicts labeled labelled label labels text white black blue + green orange red gray grey dark light bright soft hard wide tall small + large tiny big thin thick line lines box boxes node nodes panel panels + layout grid row rows column columns + """.split() +) + + +def _optional_bool(cfg: "Config", block: dict, key: str, label: str) -> bool | None: + if key not in block or block.get(key) is None: + return None + from docgen.config import require_yaml_bool + + return require_yaml_bool(block[key], label=label, source=cfg._source_label()) + + +def align_with_docs(cfg: "Config") -> bool: + """Default true. ``image_generation.align_with_docs`` must be a YAML bool.""" + block = cfg._block("image_generation") + flag = _optional_bool(cfg, block, "align_with_docs", "image_generation.align_with_docs") + if flag is None: + return True + return flag + + +def align_review_enabled(cfg: "Config") -> bool: + block = cfg._block("image_generation") + flag = _optional_bool(cfg, block, "align_review", "image_generation.align_review") + if flag is None: + return True + return flag + + +def image_prompt_alignment_enabled(cfg: "Config") -> bool: + block = cfg._sub_block( + cfg._block("validation"), + "image_prompt_alignment", + label="validation.image_prompt_alignment", + ) + flag = _optional_bool( + cfg, block, "enabled", "validation.image_prompt_alignment.enabled" + ) + if flag is None: + return True + return flag + + +def image_asset_alignment_settings(cfg: "Config") -> dict: + """OCR on by default; vision review off unless ``review: true``.""" + block = cfg._sub_block( + cfg._block("validation"), + "image_asset_alignment", + label="validation.image_asset_alignment", + ) + enabled = _optional_bool(cfg, block, "enabled", "validation.image_asset_alignment.enabled") + ocr = _optional_bool(cfg, block, "ocr", "validation.image_asset_alignment.ocr") + review = _optional_bool(cfg, block, "review", "validation.image_asset_alignment.review") + return { + "enabled": True if enabled is None else enabled, + "ocr": True if ocr is None else ocr, + "review": False if review is None else review, + } + + +def _prompt_preview(prompt: str) -> str: + if len(prompt) <= 80: + return prompt + return prompt[:77] + "..." + + +def _prompt_alignment_issue(el: dict, corpus: set[str]) -> str | None: + from docgen.scene_spec import content_tokens + + rel = str(el.get("image") or "").strip() or "(unnamed image)" + prompt = str(el.get("prompt") or "").strip() + if not prompt: + return None + substance = content_tokens(prompt) - _IMAGE_STYLE_TOKENS + if not substance: + return f"{rel}: image prompt is only visual style (no documented subject terms)" + if substance & corpus: + return None + preview = _prompt_preview(prompt) + return ( + f"{rel}: image prompt shares no documented terms with " + f"narration/source: {preview!r}" + ) + + +def image_prompt_alignment_violations( + spec: dict, + *, + corpus_text: str, +) -> list[str]: + """Reject image prompts that share no documented terms with narration/source.""" + from docgen.scene_spec import content_tokens, iter_image_elements + + elements = iter_image_elements(spec) + if not elements: + return [] + corpus = content_tokens(corpus_text) + if not corpus: + return [] + issues: list[str] = [] + for el in elements: + issue = _prompt_alignment_issue(el, corpus) + if issue: + issues.append(issue) + return issues + + +def _confident_invented(ocr_text: str, corpus: set[str]) -> set[str]: + from docgen.scene_spec import content_tokens + + substance = content_tokens(ocr_text) - _IMAGE_STYLE_TOKENS + invented = {token for token in substance if token not in corpus} + return {token for token in invented if len(token) >= 4} + + +def _ocr_should_fail(confident: set[str]) -> bool: + if len(confident) >= 2: + return True + return any(len(token) >= 6 for token in confident) + + +def image_ocr_alignment_violations( + ocr_text: str, + *, + corpus_text: str, + relpath: str = "image", +) -> list[str]: + """Reject OCR tokens that look like invented documented terms.""" + from docgen.scene_spec import content_tokens + + corpus = content_tokens(corpus_text) + if not corpus: + return [] + confident = _confident_invented(ocr_text, corpus) + if not _ocr_should_fail(confident): + return [] + sample = ", ".join(repr(token) for token in sorted(confident)[:6]) + return [f"{relpath}: on-image OCR has terms not in narration/source: {sample}"] + + +def _load_spec_mapping(path: Path) -> tuple[dict | None, str | None]: + import yaml + + try: + raw = yaml.safe_load(path.read_text(encoding="utf-8")) + except (OSError, yaml.YAMLError) as exc: + return None, f"could not load {path.name}: {exc}" + if isinstance(raw, dict): + return raw, None + return None, f"{path.name}: root must be a mapping" + + +def _asset_ocr_issues(asset: Path, rel: str, corpus: str, ocr_on: bool) -> list[str]: + if not ocr_on: + return [] + ocr_text = ocr_image_text(asset) + if ocr_text is None: + return [f"{rel}: could not OCR image (unreadable file)"] + if not ocr_text: + return [] + return image_ocr_alignment_violations(ocr_text, corpus_text=corpus, relpath=rel) + + +def _asset_review_issues( + asset: Path, + el: dict, + corpus: str, + cfg: "Config", + review_on: bool, +) -> list[str]: + if not review_on: + return [] + rel = str(el.get("image") or "").strip() + verdict = review_image_against_docs( + asset, + corpus_text=corpus, + authored_prompt=str(el.get("prompt") or ""), + label=str(el.get("label") or ""), + cfg=cfg, + ) + if verdict.passed: + return [] + return [f"{rel}: vision review FAIL — {verdict.reason}"] + + +def _pixel_issues_for_spec(cfg: "Config", raw: dict, settings: dict) -> list[str]: + from docgen.image_generate import collect_alignment_corpus + from docgen.scene_spec import iter_image_elements + + corpus = collect_alignment_corpus(cfg, raw) + if not corpus.strip(): + return [] + issues: list[str] = [] + for el in iter_image_elements(raw): + rel = str(el.get("image") or "").strip() + if not rel: + continue + asset = cfg.base_dir / rel + if not asset.is_file(): + continue + issues.extend(_asset_ocr_issues(asset, rel, corpus, bool(settings.get("ocr", True)))) + issues.extend(_asset_review_issues(asset, el, corpus, cfg, bool(settings.get("review")))) + return issues + + +def image_doc_alignment_issues(cfg: "Config", seg_id: str) -> list[str]: + """Prompt and pixel issues for one segment. Empty when there is nothing to score. + + Folded into ``scene_assets`` so validate fails closed without a new check name. + """ + from docgen.image_generate import collect_alignment_corpus + + if not image_prompt_alignment_enabled(cfg) and not image_asset_alignment_settings(cfg)["enabled"]: + return [] + seg_name = cfg.resolve_segment_name(seg_id) + spec_path = cfg.animations_dir / "specs" / f"{seg_name}.scene.yaml" + if not spec_path.is_file(): + return [] + raw, err = _load_spec_mapping(spec_path) + if err: + return [err] + assert raw is not None + issues: list[str] = [] + if image_prompt_alignment_enabled(cfg): + corpus = collect_alignment_corpus(cfg, raw) + issues.extend(image_prompt_alignment_violations(raw, corpus_text=corpus)) + asset_settings = image_asset_alignment_settings(cfg) + if asset_settings["enabled"]: + issues.extend(_pixel_issues_for_spec(cfg, raw, asset_settings)) + return issues diff --git a/src/docgen/image_generate.py b/src/docgen/image_generate.py index 980cbb8..ff5b4e3 100644 --- a/src/docgen/image_generate.py +++ b/src/docgen/image_generate.py @@ -31,13 +31,13 @@ from docgen.image_align import ( ImageReviewResult, + image_ocr_alignment_violations, + image_prompt_alignment_violations, ocr_image_text, review_image_against_docs, ) from docgen.openai_retry import call_with_rate_limit_retries from docgen.scene_spec import ( - image_ocr_alignment_violations, - image_prompt_alignment_violations, iter_image_elements, load_scene_spec, ) @@ -190,31 +190,45 @@ def build_aligned_image_prompt( return "\n".join(parts).strip() + "\n" +def _narration_corpus_part(cfg: "Config", seg_id: str) -> str: + found = cfg.find_segment_asset(cfg.narration_dir, seg_id, ".md") + if found is None or not found.is_file(): + return "" + try: + return found.read_text(encoding="utf-8") + except OSError: + return "" + + +def _hint_source_parts(cfg: "Config", seg_id: str) -> list[str]: + from docgen.manim_scene_support import ( + collect_source_snippets, + merged_scene_generation_settings, + ) + + settings = merged_scene_generation_settings(cfg, seg_id) + parts: list[str] = [] + for hint in settings.hints: + if str(hint).strip(): + parts.append(str(hint).strip()) + for label, text in collect_source_snippets(cfg, settings, extra_paths=[]): + body = str(text or "").strip() + if body: + parts.append(f"{label}\n{body}") + return parts + + def collect_alignment_corpus(cfg: "Config", spec: dict[str, Any]) -> str: """Narration + scene-generation hints + source snippets for one spec.""" - parts: list[str] = [] seg_id = str(spec.get("segment_id") or "").strip() - if seg_id: - found = cfg.find_segment_asset(cfg.narration_dir, seg_id, ".md") - if found is not None and found.is_file(): - try: - parts.append(found.read_text(encoding="utf-8")) - except OSError: - pass - from docgen.manim_scene_support import ( - collect_source_snippets, - merged_scene_generation_settings, - ) - - settings = merged_scene_generation_settings(cfg, seg_id) - for h in settings.hints: - if str(h).strip(): - parts.append(str(h).strip()) - for label, text in collect_source_snippets(cfg, settings, extra_paths=[]): - body = str(text or "").strip() - if body: - parts.append(f"{label}\n{body}") - return "\n\n".join(p for p in parts if str(p).strip()) + if not seg_id: + return "" + parts: list[str] = [] + narration = _narration_corpus_part(cfg, seg_id) + if narration: + parts.append(narration) + parts.extend(_hint_source_parts(cfg, seg_id)) + return "\n\n".join(part for part in parts if str(part).strip()) def _resolve_asset_path(cfg: "Config", relpath: str) -> Path: @@ -227,6 +241,218 @@ def _resolve_asset_path(cfg: "Config", relpath: str) -> Path: return cfg.base_dir / p +def _review_retry_count(cfg: "Config") -> int: + block = cfg._block("image_generation") + raw = block.get("align_review_retries") + if raw is None: + return 1 + from docgen.config import require_yaml_number + + return max(0, int(require_yaml_number( + raw, label="image_generation.align_review_retries", source=cfg._source_label() + ))) + + +def _override_or(override: str | None, configured: object, default: str) -> str: + chosen = (override or "").strip() + if chosen: + return chosen + text = str(configured or "").strip() + if text: + return text + return default + + +def _optional_quality(configured: object) -> str | None: + if not configured: + return None + text = str(configured).strip() + if text: + return text + return None + + +def _image_job(cfg: "Config", model_override: str | None, size_override: str | None) -> dict[str, Any]: + from docgen.image_align import align_review_enabled, align_with_docs + + icfg = cfg.image_generation_config + return { + "model": _override_or(model_override, icfg.get("model"), DEFAULT_IMAGE_MODEL), + "size": _override_or(size_override, icfg.get("size"), DEFAULT_IMAGE_SIZE), + "quality": _optional_quality(icfg.get("quality")), + "align": align_with_docs(cfg), + "align_review": align_review_enabled(cfg), + "pixel_retries": _review_retry_count(cfg), + "review_model": str(icfg.get("review_model") or "").strip(), + "style": _override_or(None, icfg.get("style"), DEFAULT_IMAGE_STYLE), + } + + +def _live_review( + align: bool, + align_review: bool, + review_fn: Callable[..., ImageReviewResult] | None, + image_fn: Callable[[str], bytes] | None, +) -> bool: + if not align: + return False + if not align_review: + return False + if review_fn is not None: + return True + return image_fn is None + + +def _reject_unaligned_prompts(spec_path: Path, spec: dict[str, Any], corpus: str, align: bool) -> None: + if not align: + return + issues = image_prompt_alignment_violations(spec, corpus_text=corpus) + if not issues: + return + joined = "\n ".join(issues) + raise ImageGenerationError( + f"{spec_path}: image prompt alignment failed — rewrite each " + f"`prompt` so it uses documented terms from narration/source " + f"(or set image_generation.align_with_docs: false):\n {joined}" + ) + + +def _effective_prompt(prompt: str, *, align: bool, corpus: str, label: str, style: str) -> str: + if align and prompt: + return build_aligned_image_prompt(prompt, corpus_text=corpus, label=label, style=style) + return prompt + + +def _existing_or_dry( + *, + spec_path: Path, + rel: str, + out: Path, + prompt: str, + effective: str, + force: bool, + dry_run: bool, +) -> ImageAssetResult | None: + if out.is_file() and not force: + return ImageAssetResult(rel, out, "exists", prompt, effective) + if not prompt: + raise ImageGenerationError( + f"{spec_path}: image element {rel!r} has no `prompt` and the asset is missing " + f"({out}); add the file to the bundle or set a prompt in the spec." + ) + if dry_run: + return ImageAssetResult(rel, out, "dry-run", prompt, effective) + return None + + +def _prompt_with_critique(effective: str, critique: str) -> str: + if not critique: + return effective + return ( + f"{effective}\n\n--- PIXEL REVIEW FAILED ---\n{critique}\n" + "Redraw so the image matches the documented subject. " + "Do not invent labels or extra components." + ) + + +def _write_or_raise(spec_path: Path, rel: str, out: Path, data: bytes) -> None: + if not data: + raise ImageGenerationError( + f"{spec_path}: image element {rel!r} — provider returned empty bytes" + ) + out.parent.mkdir(parents=True, exist_ok=True) + out.write_bytes(data) + + +def _draw_until_aligned( + *, + spec_path: Path, + rel: str, + out: Path, + prompt: str, + effective: str, + fn: Callable[[str], bytes], + pixel_retries: int, + pixel_kwargs: dict[str, Any], +) -> None: + critique = "" + last_issues: list[str] = [] + kept = False + for _attempt in range(1 + pixel_retries): + _write_or_raise(spec_path, rel, out, fn(_prompt_with_critique(effective, critique))) + last_issues = _pixel_alignment_issues(out, relpath=rel, authored_prompt=prompt, **pixel_kwargs) + if not last_issues: + kept = True + break + critique = "\n".join(last_issues) + if kept: + return + out.unlink(missing_ok=True) + joined = "\n ".join(last_issues) + raise ImageGenerationError( + f"{spec_path}: image {rel!r} failed pixel alignment " + f"(OCR / vision review):\n {joined}" + ) + + +def _one_image( + cfg: "Config", + spec_path: Path, + el: dict[str, Any], + *, + job: dict[str, Any], + corpus: str, + force: bool, + dry_run: bool, + image_fn: Callable[[str], bytes] | None, + review_fn: Callable[..., ImageReviewResult] | None, + ocr_fn: Callable[[Path], str | None] | None, +) -> ImageAssetResult: + rel = str(el["image"]).strip() + prompt = str(el.get("prompt") or "").strip() + out = _resolve_asset_path(cfg, rel) + label = str(el.get("label") or "").strip() + effective = _effective_prompt( + prompt, align=job["align"], corpus=corpus, label=label, style=job["style"] + ) + planned = _existing_or_dry( + spec_path=spec_path, + rel=rel, + out=out, + prompt=prompt, + effective=effective, + force=force, + dry_run=dry_run, + ) + if planned is not None: + return planned + fn = image_fn or ( + lambda p: generate_image_bytes( + prompt=p, model=job["model"], size=job["size"], quality=job["quality"], cfg=cfg + ) + ) + _draw_until_aligned( + spec_path=spec_path, + rel=rel, + out=out, + prompt=prompt, + effective=effective, + fn=fn, + pixel_retries=job["pixel_retries"], + pixel_kwargs={ + "corpus": corpus, + "label": label, + "cfg": cfg, + "align": job["align"], + "live_review": _live_review(job["align"], job["align_review"], review_fn, image_fn), + "review_model": job["review_model"], + "review_fn": review_fn, + "ocr_fn": ocr_fn, + }, + ) + return ImageAssetResult(rel, out, "generated", prompt, effective) + + def generate_images_for_spec( cfg: "Config", spec_path: Path, @@ -250,109 +476,24 @@ def generate_images_for_spec( not injected (or ``review_fn`` is provided). """ spec = load_scene_spec(spec_path) - elements = iter_image_elements(spec) - icfg = cfg.image_generation_config - model = (model_override or "").strip() or str(icfg.get("model") or DEFAULT_IMAGE_MODEL) - size = (size_override or "").strip() or str(icfg.get("size") or DEFAULT_IMAGE_SIZE) - quality = icfg.get("quality") - quality = str(quality).strip() if quality else None - align = bool(icfg.get("align_with_docs", True)) - align_review = bool(icfg.get("align_review", True)) - try: - pixel_retries = int(icfg.get("align_review_retries", 1) or 0) - except (TypeError, ValueError): - pixel_retries = 1 - pixel_retries = max(0, pixel_retries) - review_model = str(icfg.get("review_model") or "").strip() - style = str(icfg.get("style") or "").strip() or DEFAULT_IMAGE_STYLE - corpus = collect_alignment_corpus(cfg, spec) if align else "" - live_review = align and align_review and (review_fn is not None or image_fn is None) - - if align: - issues = image_prompt_alignment_violations(spec, corpus_text=corpus) - if issues: - joined = "\n ".join(issues) - raise ImageGenerationError( - f"{spec_path}: image prompt alignment failed — rewrite each " - f"`prompt` so it uses documented terms from narration/source " - f"(or set image_generation.align_with_docs: false):\n {joined}" - ) - - results: list[ImageAssetResult] = [] - for el in elements: - rel = str(el["image"]).strip() - prompt = str(el.get("prompt") or "").strip() - out = _resolve_asset_path(cfg, rel) - label = str(el.get("label") or "").strip() - effective = ( - build_aligned_image_prompt( - prompt, corpus_text=corpus, label=label, style=style - ) - if align and prompt - else prompt - ) - - if out.is_file() and not force: - results.append(ImageAssetResult(rel, out, "exists", prompt, effective)) - continue - if not prompt: - raise ImageGenerationError( - f"{spec_path}: image element {rel!r} has no `prompt` and the asset is missing " - f"({out}); add the file to the bundle or set a prompt in the spec." - ) - if dry_run: - results.append(ImageAssetResult(rel, out, "dry-run", prompt, effective)) - continue - - fn = image_fn or ( - lambda p: generate_image_bytes( - prompt=p, model=model, size=size, quality=quality, cfg=cfg - ) + job = _image_job(cfg, model_override, size_override) + corpus = collect_alignment_corpus(cfg, spec) if job["align"] else "" + _reject_unaligned_prompts(spec_path, spec, corpus, job["align"]) + return [ + _one_image( + cfg, + spec_path, + el, + job=job, + corpus=corpus, + force=force, + dry_run=dry_run, + image_fn=image_fn, + review_fn=review_fn, + ocr_fn=ocr_fn, ) - critique = "" - kept = False - last_issues: list[str] = [] - for attempt in range(1 + pixel_retries): - to_send = effective - if critique: - to_send = ( - f"{effective}\n\n--- PIXEL REVIEW FAILED ---\n{critique}\n" - "Redraw so the image matches the documented subject. " - "Do not invent labels or extra components." - ) - data = fn(to_send) - if not data: - raise ImageGenerationError( - f"{spec_path}: image element {rel!r} — provider returned empty bytes" - ) - out.parent.mkdir(parents=True, exist_ok=True) - out.write_bytes(data) - last_issues = _pixel_alignment_issues( - out, - relpath=rel, - corpus=corpus, - authored_prompt=prompt, - label=label, - cfg=cfg, - align=align, - live_review=live_review, - review_model=review_model, - review_fn=review_fn, - ocr_fn=ocr_fn, - ) - if not last_issues: - kept = True - break - critique = "\n".join(last_issues) - if not kept: - out.unlink(missing_ok=True) - joined = "\n ".join(last_issues) - raise ImageGenerationError( - f"{spec_path}: image {rel!r} failed pixel alignment " - f"(OCR / vision review):\n {joined}" - ) - results.append(ImageAssetResult(rel, out, "generated", prompt, effective)) - return results + for el in iter_image_elements(spec) + ] def _pixel_alignment_issues( diff --git a/src/docgen/init.py b/src/docgen/init.py index 98ed0d6..8f16208 100644 --- a/src/docgen/init.py +++ b/src/docgen/init.py @@ -338,8 +338,6 @@ def _write_config(plan: InitPlan) -> str: "image_generation": { "model": "gpt-image-1", # Cursor/OpenAI Images; override with --model "size": "1536x1024", - "align_with_docs": True, # ground prompts in narration/source; fail invented terms - "align_review": True, # vision-review the PNG against the same docs }, "compose": { "ffmpeg_timeout_sec": 300, diff --git a/src/docgen/scene_asset_validate.py b/src/docgen/scene_asset_validate.py index 5de31a1..8d13c0a 100644 --- a/src/docgen/scene_asset_validate.py +++ b/src/docgen/scene_asset_validate.py @@ -398,7 +398,10 @@ def scene_asset_violations_for_segment(cfg: "Config", seg_id: str) -> list[str]: validate_scene_spec, ) + from docgen.image_align import image_doc_alignment_issues + issues: list[str] = [] + issues.extend(image_doc_alignment_issues(cfg, seg_id)) scenes_path = cfg.animations_dir / "scenes.py" scenes_text = "" if scenes_path.is_file(): diff --git a/src/docgen/scene_spec.py b/src/docgen/scene_spec.py index d22fbd2..ed7d94c 100644 --- a/src/docgen/scene_spec.py +++ b/src/docgen/scene_spec.py @@ -1339,92 +1339,6 @@ def content_tokens(text: str) -> set[str]: return {w for w in words if len(w) > 2 and w not in _CONTENT_STOPWORDS} -# Visual-style words an image prompt may use without being documented terms. -_IMAGE_STYLE_TOKENS = frozenset( - """ - diagram diagrams illustration illustrations icon icons flat clean vector - isometric cartoon watercolor sketch photo photographic realistic abstract - artwork image images picture pictures visual background foreground style - colored colour color palette high contrast simple minimal educational - documentary video watermark signature logo logos chrome banner poster - scene board rounded pill diamond arrow arrows flow flowchart infographic - thumbnail render rendering drawing draw depict showing shows show - depicting depicts labeled labelled label labels text white black blue - green orange red gray grey dark light bright soft hard wide tall small - large tiny big thin thick line lines box boxes node nodes panel panels - layout grid row rows column columns - """.split() -) - - -def image_prompt_alignment_violations( - spec: dict[str, Any], - *, - corpus_text: str, -) -> list[str]: - """Reject image prompts that share no documented terms with narration/source. - - Style adjectives (``flat``, ``diagram``, ``illustration``, …) are ignored. - A missing corpus (no narration and no source) is a no-op — there is nothing - to align against. Specs with no image elements, or image elements that omit - ``prompt`` (committed assets), also pass. - """ - elements = iter_image_elements(spec) - if not elements: - return [] - corpus = content_tokens(corpus_text) - if not corpus: - return [] - issues: list[str] = [] - for el in elements: - rel = str(el.get("image") or "").strip() or "(unnamed image)" - prompt = str(el.get("prompt") or "").strip() - if not prompt: - continue - toks = content_tokens(prompt) - substance = toks - _IMAGE_STYLE_TOKENS - if not substance: - issues.append( - f"{rel}: image prompt is only visual style (no documented subject terms)" - ) - continue - if not (substance & corpus): - preview = prompt if len(prompt) <= 80 else prompt[:77] + "..." - issues.append( - f"{rel}: image prompt shares no documented terms with " - f"narration/source: {preview!r}" - ) - return issues - - -def image_ocr_alignment_violations( - ocr_text: str, - *, - corpus_text: str, - relpath: str = "image", -) -> list[str]: - """Reject OCR tokens that look like invented documented terms. - - Style words and a missing corpus are ignored. A single short OCR ghost - (``<6`` letters) is tolerated; two confident invented tokens, or one - token of length ≥ 6, fail. - """ - corpus = content_tokens(corpus_text) - if not corpus: - return [] - substance = content_tokens(ocr_text) - _IMAGE_STYLE_TOKENS - if not substance: - return [] - invented = {t for t in substance if t not in corpus} - confident = {t for t in invented if len(t) >= 4} - if len(confident) >= 2 or any(len(t) >= 6 for t in confident): - sample = ", ".join(repr(x) for x in sorted(confident)[:6]) - return [ - f"{relpath}: on-image OCR has terms not in narration/source: {sample}" - ] - return [] - - def cluster_subject_beats( sentences: list[str], *, diff --git a/src/docgen/scene_spec_flow.py b/src/docgen/scene_spec_flow.py new file mode 100644 index 0000000..797df12 --- /dev/null +++ b/src/docgen/scene_spec_flow.py @@ -0,0 +1,633 @@ +"""Scene-spec generate steps kept out of the hotspot module's complex functions. + +``scene_spec_generate`` re-exports the public entry points. Each function here +stays at or under the complexity gate (CCN 10, NLOC 80). +""" + +from __future__ import annotations + +import json +from typing import TYPE_CHECKING, Any, Callable + +if TYPE_CHECKING: + from docgen.config import Config + +import yaml + +from docgen.manim_scene_support import ( + SceneGenerationError, + collect_source_snippets, + derive_class_name, + extract_reference_classes, + merged_scene_generation_settings, +) +from docgen.manim_scene_support import _load_narration as load_narration_for_scene +from docgen.manim_scene_support import _load_timing_segments as load_timing_for_scene +from docgen.scene_spec import ( + FRAME_HEIGHT, + FRAME_WIDTH, + SceneSpecError, + auto_fit_row_widths, + auto_paginate, + cluster_subject_beats, + coerce_legacy_wait_at_to_whisper_rows, + compile_scene_class, + layout_budget_violations, + layout_density_violations, + layout_stack_budget, + narration_sentences, + pacing_violations, + sanitize_pacing_conflicts, + spec_rows_reference_whisper_waits, + sync_row_labels_to_whisper_words, + upgrade_wait_segments_to_wait_words, + validate_scene_spec, +) + +def _beat_lines(narration_text: str, word_count: int) -> list[str]: + beats = cluster_subject_beats(narration_sentences(narration_text)) + if not beats: + return [] + wc_note = f" ({word_count} Whisper words)" if word_count > 0 else "" + lines = [ + f"**SUBJECT BEATS{wc_note} — cover each with ≥1 spoken-phrase label " + f"(hold the board inside a beat; change when the topic shifts):**" + ] + for i, beat in enumerate(beats, start=1): + preview = beat if len(beat) <= 160 else beat[:157] + "..." + lines.append(f" {i}. {preview}") + lines.append( + "scene-spec-generate **rejects** specs that leave beats uncovered or use " + "labels that are not spoken in the narration (not a blind label count)." + ) + lines.append("") + return lines + + +def _hint_lines(hints: list[str], extra_hints: list[str]) -> list[str]: + all_hints = list(hints) + list(extra_hints) + if not all_hints: + return [] + lines = ["", "--- PROJECT-OWNER HINTS ---"] + for hint in all_hints: + if str(hint).strip(): + lines.append(f"- {str(hint).strip()}") + return lines + + +def _source_doc_lines(source_snippets: list[tuple[str, str]]) -> list[str]: + if not source_snippets: + return [] + lines = [ + "", + "--- SOURCE DOCUMENTATION ---", + "Use these excerpts when authoring box labels **and** image prompts. " + "Do not invent product names, APIs, or architecture that is not in this " + "source or the narration.", + ] + for label, text in source_snippets: + body = str(text or "").strip() + if not body: + continue + lines.extend(["", f"### {label}", body]) + return lines + + +def _reference_lines(reference_scenes: str) -> list[str]: + if not reference_scenes: + return [] + return [ + "", + "--- REFERENCE (existing Manim classes — steal **ideas**, output YAML only) ---", + reference_scenes, + ] + + +def _budget_lines() -> list[str]: + horiz_safe = FRAME_WIDTH - 1.0 + budget_default = layout_stack_budget({"font_size": 36}, {"first_row_title_buff": 0.5}) + budget_compact = layout_stack_budget({"font_size": 32}, {"first_row_title_buff": 0.45}) + return [ + "", + "--- FRAME / LAYOUT BUDGET (plan every page; scene-spec-generate rejects overflow) ---", + ( + f"Frame ≈ {FRAME_WIDTH:.2f} × {FRAME_HEIGHT:.2f} Manim units. " + f"Horizontal safe width ≈ {horiz_safe:.2f} u " + "(sum of box widths + (n_boxes-1)*column_gap per row must stay ≤ this)." + ), + ( + "**Vertical stack budgets** (use these numbers unless you change " + "title.font_size / layout.first_row_title_buff):\n" + f" • Default font_size=36, first_row_title_buff=0.5 → " + f"max stack height ≈ {budget_default:.2f} u\n" + f" • Compact font_size=32, first_row_title_buff=0.45 → " + f"max stack height ≈ {budget_compact:.2f} u\n" + "Per page: sum(max box height per row) + (n_rows-1)*row_gap ≤ that budget. " + "When you would exceed it, spill to another page (do not shrink/cram)." + ), + ] + + +def build_scene_spec_user_message( + *, + seg_id: str, + seg_name: str, + class_name: str, + narration_text: str, + timing_enrichment: str, + hints: list[str], + extra_hints: list[str], + reference_scenes: str, + source_snippets: list[tuple[str, str]], + word_count: int = 0, +) -> str: + """User message: narration + timing + hints; demand YAML spec.""" + parts = [ + ( + f"Produce a **scene spec YAML** (not Python) for segment `{seg_id}` / " + f"class `{class_name}` (narration stem `{seg_name}`)." + ), + "", + "**Required YAML fields** — use these exact values:", + f" segment_id: {json.dumps(str(seg_id).strip())}", + f" class_name: {json.dumps(class_name)}", + "", + "--- NARRATION ---", + narration_text.strip() or "(empty)", + "", + ] + parts.extend(_beat_lines(narration_text, word_count)) + parts.append(timing_enrichment.strip()) + parts.extend(_hint_lines(hints, extra_hints)) + parts.extend(_source_doc_lines(source_snippets)) + parts.extend(_reference_lines(reference_scenes)) + parts.extend(_budget_lines()) + return "\n".join(parts) + + +def _merge_timing_key(cfg: "Config", spec: dict[str, Any], timing_key: str | None) -> dict[str, Any]: + merged = dict(spec) + sid = str(merged["segment_id"]).strip() + if timing_key is not None: + merged["timing_key"] = timing_key + return merged + if not merged.get("timing_key"): + merged["timing_key"] = cfg.resolve_segment_name(sid) + return merged + + +def _sync_audio( + merged: dict[str, Any], + words: list[dict[str, Any]], + segments: list[dict[str, Any]], +) -> dict[str, Any]: + merged = coerce_legacy_wait_at_to_whisper_rows(merged, words, segments) + if words and segments: + merged = upgrade_wait_segments_to_wait_words(merged, words, segments) + if words: + merged = sync_row_labels_to_whisper_words(merged, words, overwrite=True) + return merged + + +def _ensure_words_if_paced(merged: dict[str, Any], words: list, tk: str) -> None: + if spec_rows_reference_whisper_waits(merged) and not words: + raise SceneGenerationError( + f"timing.json has no word-level `words` for stem {tk!r}; run `docgen timestamps` " + "before compiling scenes that use wait_word or wait_segment." + ) + + +def _ensure_pacing(merged: dict[str, Any], words: list, tk: str) -> None: + pace_issues = pacing_violations(merged, words_present=bool(words)) + if not pace_issues: + return + shown = "\n ".join(pace_issues[:12]) + more = f"\n (+{len(pace_issues) - 12} more)" if len(pace_issues) > 12 else "" + raise SceneGenerationError( + f"scene pacing failed for timing_key {tk!r} — every story box needs a " + f"spoken label matched in timing.json words (or pace: none):\n {shown}{more}" + ) + + +def _compile_and_lint(cfg: "Config", merged: dict[str, Any], words: list) -> str: + from docgen.manim_scene_support import lint_generated_block + + try: + class_block = compile_scene_class(merged, words=words or None) + except SceneSpecError as exc: + raise SceneGenerationError(str(exc)) from exc + issues = lint_generated_block( + class_block, + min_font_size=cfg.manim_min_font_size, + unsafe_unicode=cfg.manim_unsafe_unicode, + ) + if issues: + joined = "\n ".join(issues[:20]) + raise SceneGenerationError(f"compiled scene failed manim_scene_lint:\n {joined}") + return class_block + + +def linted_class_block_from_spec( + cfg: "Config", + spec: dict[str, Any], + *, + timing_key: str | None = None, +) -> tuple[str, dict[str, Any]]: + """Merge ``timing_key``, auto-paginate + word-align, compile, run ``manim_scene_lint``.""" + from docgen.scene_spec_generate import _load_timing_words + + merged = _merge_timing_key(cfg, spec, timing_key) + merged = auto_paginate(auto_fit_row_widths(merged)) + tk = str(merged["timing_key"]) + segments = load_timing_for_scene(cfg, tk) + words = _load_timing_words(cfg, tk) + merged = _sync_audio(merged, words, segments) + _ensure_words_if_paced(merged, words, tk) + _ensure_pacing(merged, words, tk) + return _compile_and_lint(cfg, merged, words), merged + + +def _load_yaml_mapping(cfg: "Config", seg_id: str, raw: str) -> tuple[str, dict[str, Any]]: + from docgen.scene_spec_generate import _save_draft, strip_yaml_fences + + body = strip_yaml_fences(raw) + try: + loaded = yaml.safe_load(body) + except yaml.YAMLError as exc: + draft = _save_draft(cfg, seg_id, raw) + raise SceneGenerationError( + f"segment {seg_id}: LLM output is not valid YAML ({exc}). Draft: {draft}" + ) from exc + if not isinstance(loaded, dict): + draft = _save_draft(cfg, seg_id, raw) + raise SceneGenerationError( + f"segment {seg_id}: LLM YAML root must be a mapping. Draft: {draft}" + ) + return body, loaded + + +def _reject_joined( + cfg: "Config", + seg_id: str, + body: str, + issues: list[str], + headline: str, +) -> None: + if not issues: + return + from docgen.scene_spec_generate import _save_draft + + draft = _save_draft(cfg, seg_id, body) + joined = "\n ".join(issues) + raise SceneGenerationError(f"segment {seg_id}: {headline}:\n {joined}\nDraft: {draft}") + + +def _ensure_schema(cfg: "Config", seg_id: str, body: str, merged: dict[str, Any]) -> None: + from docgen.scene_spec_generate import _save_draft + + try: + validate_scene_spec(merged, path_label=f"segment {seg_id}") + except SceneSpecError as exc: + draft = _save_draft(cfg, seg_id, body) + raise SceneGenerationError( + f"segment {seg_id}: scene spec invalid: {exc}. Draft: {draft}" + ) from exc + + +def _ensure_density( + cfg: "Config", + *, + seg_id: str, + body: str, + merged: dict[str, Any], + narration_text: str, + word_count: int, + enforce_density: bool, + density_slack: int, +) -> None: + if not enforce_density: + return + if not getattr(cfg, "subject_beat_coverage_enabled", True): + return + issues = layout_density_violations( + merged, + narration_text=narration_text, + word_count=word_count, + slack=density_slack, + ) + _reject_joined(cfg, seg_id, body, issues, "scene spec failed subject-beat coverage") + + +def _ensure_image_alignment( + cfg: "Config", + *, + seg_id: str, + body: str, + merged: dict[str, Any], + corpus_text: str, + narration_text: str, +) -> None: + from docgen.image_align import image_prompt_alignment_enabled, image_prompt_alignment_violations + + if not image_prompt_alignment_enabled(cfg): + return + issues = image_prompt_alignment_violations( + merged, corpus_text=corpus_text or narration_text + ) + _reject_joined(cfg, seg_id, body, issues, "image prompt alignment failed") + + +def _ensure_compiles(cfg: "Config", seg_id: str, body: str, merged: dict[str, Any], seg_name: str) -> None: + from docgen.scene_spec_generate import _save_draft + + try: + linted_class_block_from_spec(cfg, merged, timing_key=seg_name) + except SceneGenerationError as exc: + draft = _save_draft(cfg, seg_id, body) + raise SceneGenerationError(f"{exc} Draft: {draft}") from exc + + +def _parse_and_harden_llm_spec( + cfg: "Config", + *, + seg_id: str, + class_name: str, + seg_name: str, + narration_text: str, + word_count: int, + raw: str, + enforce_density: bool, + density_slack: int = 0, + corpus_text: str = "", +) -> dict[str, Any]: + from docgen.scene_spec_generate import normalize_spec_from_llm + + body, loaded = _load_yaml_mapping(cfg, seg_id, raw) + merged = sanitize_pacing_conflicts(auto_paginate(auto_fit_row_widths( + normalize_spec_from_llm(loaded, seg_id=seg_id, class_name=class_name) + ))) + _ensure_schema(cfg, seg_id, body, merged) + _reject_joined( + cfg, seg_id, body, layout_budget_violations(merged), "scene spec exceeds frame budget" + ) + _ensure_density( + cfg, + seg_id=seg_id, + body=body, + merged=merged, + narration_text=narration_text, + word_count=word_count, + enforce_density=enforce_density, + density_slack=density_slack, + ) + _ensure_image_alignment( + cfg, + seg_id=seg_id, + body=body, + merged=merged, + corpus_text=corpus_text, + narration_text=narration_text, + ) + _ensure_compiles(cfg, seg_id, body, merged, seg_name) + return merged + + +def _corpus_text( + narration_text: str, + hints: list[str], + extra_hints: list[str], + snippets: list[tuple[str, str]], +) -> str: + parts = [narration_text] + for hint in list(hints) + list(extra_hints): + if str(hint).strip(): + parts.append(str(hint).strip()) + for label, text in snippets: + body = str(text or "").strip() + if body: + parts.append(f"{label}\n{body}") + return "\n\n".join(part for part in parts if str(part).strip()) + + +def _prepare_generation(cfg: "Config", seg_id: str, extra_paths: list[str], extra_hints: list[str], class_name_override: str | None) -> dict[str, Any]: + from docgen.manim_scene_support import build_timing_enrichment_for_prompt + from docgen.scene_spec_generate import _load_timing_words, scene_spec_system_prompt + + settings = merged_scene_generation_settings(cfg, seg_id) + seg_name = cfg.resolve_segment_name(seg_id) + class_name = derive_class_name(seg_id, seg_name, class_name_override or settings.class_name) + narration_text = load_narration_for_scene(cfg, seg_id, seg_name) + whisper_segments = load_timing_for_scene(cfg, seg_name) + timing_block = build_timing_enrichment_for_prompt(cfg, seg_id, seg_name, whisper_segments) + word_count = len(_load_timing_words(cfg, seg_name)) + scenes_path = cfg.animations_dir / "scenes.py" + existing = scenes_path.read_text(encoding="utf-8") if scenes_path.exists() else "" + snippets = collect_source_snippets(cfg, settings, extra_paths=extra_paths) + user_message = build_scene_spec_user_message( + seg_id=seg_id, + seg_name=seg_name, + class_name=class_name, + narration_text=narration_text, + timing_enrichment=timing_block, + hints=settings.hints, + extra_hints=extra_hints, + reference_scenes=extract_reference_classes(existing), + source_snippets=snippets, + word_count=word_count, + ) + return { + "settings": settings, + "seg_name": seg_name, + "class_name": class_name, + "narration_text": narration_text, + "word_count": word_count, + "system_prompt": scene_spec_system_prompt(cfg, seg_id), + "user_message": user_message, + "corpus_text": _corpus_text(narration_text, list(settings.hints), extra_hints, snippets), + "extra_hints": extra_hints, + } + + +def _dry_run_result(ctx: dict[str, Any], seg_id: str) -> Any: + from docgen.scene_spec_generate import SceneSpecGenerationResult + + prompt = f"--- system ---\n{ctx['system_prompt']}\n\n--- user ---\n{ctx['user_message']}" + return SceneSpecGenerationResult( + seg_id=seg_id, + seg_name=ctx["seg_name"], + class_name=ctx["class_name"], + spec={}, + yaml_text="", + prompt=prompt, + raw_response="", + ) + + +def _call_llm(cfg: "Config", invoke: Callable[..., str], **kwargs: Any) -> str: + try: + return invoke(**kwargs) + except RuntimeError as exc: + from docgen.ai_client import resolve_ai_settings + + settings = resolve_ai_settings(cfg) + raise SceneGenerationError( + f"Chat call failed ({exc}). " + f"{settings.auth_help()} Set DOCGEN_ENV_OVERRIDES=1 to load the bundle " + "env_file, or use --dry-run to inspect the prompt only." + ) from exc + + +def _message_for_attempt( + user_message: str, + attempt: int, + last_sparse: SceneGenerationError | None, + n_beats: int, +) -> str: + if attempt == 0 or last_sparse is None: + return user_message + err = str(last_sparse) + if "image prompt alignment" in err: + return ( + f"{user_message}\n\n--- RETRY: IMAGE PROMPT ALIGNMENT FAILED ---\n" + f"{last_sparse}\n" + "Rewrite each image prompt so it uses documented terms from the " + "narration and SOURCE DOCUMENTATION. Do not invent product names " + "or generic 'a diagram' artwork with no subject." + ) + return ( + f"{user_message}\n\n--- RETRY: SUBJECT-BEAT COVERAGE FAILED ---\n" + f"{last_sparse}\n" + f"Cover each of the {n_beats} subject beats with a spoken-phrase label. " + "Hold the board across sentences in the same beat; add a new label only " + "when the topic shifts. Do not invent unspoken diagram terms." + ) + + +def _harden_once(cfg: "Config", ctx: dict[str, Any], seg_id: str, raw: str, slack: int) -> dict[str, Any]: + return _parse_and_harden_llm_spec( + cfg, + seg_id=seg_id, + class_name=ctx["class_name"], + seg_name=ctx["seg_name"], + narration_text=ctx["narration_text"], + word_count=ctx["word_count"], + raw=raw, + enforce_density=True, + density_slack=slack, + corpus_text=ctx["corpus_text"], + ) + + +def _alignment_should_stop(exc: SceneGenerationError, attempt: int) -> bool: + if "image prompt alignment" not in str(exc): + return False + return attempt >= 2 + + +def _near_miss_or_raise( + cfg: "Config", + ctx: dict[str, Any], + seg_id: str, + raw: str, + near_miss_slack: int, + attempt: int, + exc: SceneGenerationError, +) -> tuple[dict[str, Any] | None, SceneGenerationError | None]: + if "subject-beat coverage" not in str(exc): + raise exc + try: + return _harden_once(cfg, ctx, seg_id, raw, near_miss_slack), None + except SceneGenerationError as near: + if "subject-beat coverage" not in str(near) or attempt >= 2: + raise + return None, near + + +def _generate_with_retries( + cfg: "Config", + ctx: dict[str, Any], + seg_id: str, + *, + model: str, + temperature: float, + invoke: Callable[..., str], +) -> tuple[dict[str, Any], str]: + n_beats = len(cluster_subject_beats(narration_sentences(ctx["narration_text"]))) + near_miss_slack = max(1, n_beats // 8) if n_beats else 0 + raw = "" + merged: dict[str, Any] = {} + last_sparse: SceneGenerationError | None = None + for attempt in range(3): + raw = _call_llm( + cfg, + invoke, + system_prompt=ctx["system_prompt"], + user_message=_message_for_attempt(ctx["user_message"], attempt, last_sparse, n_beats), + model=model, + temperature=min(0.9, temperature + 0.15 * attempt), + ) + try: + merged = _harden_once(cfg, ctx, seg_id, raw, 0) + return merged, raw + except SceneGenerationError as exc: + if _alignment_should_stop(exc, attempt): + raise + if "image prompt alignment" in str(exc): + last_sparse = exc + continue + merged_or_none, last_sparse = _near_miss_or_raise( + cfg, ctx, seg_id, raw, near_miss_slack, attempt, exc + ) + if merged_or_none is not None: + return merged_or_none, raw + if last_sparse is not None: + raise last_sparse + return merged, raw + + +def _success_result(ctx: dict[str, Any], seg_id: str, merged: dict[str, Any], raw: str) -> Any: + from docgen.scene_spec_generate import SceneSpecGenerationResult, spec_to_yaml_text + + prompt = f"--- system ---\n{ctx['system_prompt']}\n\n--- user ---\n{ctx['user_message']}" + return SceneSpecGenerationResult( + seg_id=seg_id, + seg_name=ctx["seg_name"], + class_name=ctx["class_name"], + spec=merged, + yaml_text=spec_to_yaml_text(merged), + prompt=prompt, + raw_response=raw, + ) + + +def _resolve_temperature(settings: Any, temperature_override: float | None) -> float: + if temperature_override is not None: + return float(temperature_override) + return float(settings.temperature) + + +def generate_scene_spec( + cfg: "Config", + seg_id: str, + *, + extra_paths: list[str], + extra_hints: list[str], + class_name_override: str | None = None, + dry_run: bool = False, + model_override: str | None = None, + temperature_override: float | None = None, + llm: Callable[..., str] | None = None, +) -> Any: + """Prompt OpenAI for YAML, validate schema, compile+lint the merged Python.""" + from docgen.scene_spec_generate import _invoke_llm + + ctx = _prepare_generation(cfg, seg_id, extra_paths, extra_hints, class_name_override) + if dry_run: + return _dry_run_result(ctx, seg_id) + model = (model_override or "").strip() or ctx["settings"].model + temperature = _resolve_temperature(ctx["settings"], temperature_override) + invoke = llm or (lambda **kw: _invoke_llm(cfg=cfg, **kw)) + merged, raw = _generate_with_retries( + cfg, ctx, seg_id, model=model, temperature=temperature, invoke=invoke + ) + return _success_result(ctx, seg_id, merged, raw) diff --git a/src/docgen/scene_spec_generate.py b/src/docgen/scene_spec_generate.py index 73b9410..6428b41 100644 --- a/src/docgen/scene_spec_generate.py +++ b/src/docgen/scene_spec_generate.py @@ -6,46 +6,20 @@ from __future__ import annotations -import json import re from dataclasses import dataclass from pathlib import Path -from typing import TYPE_CHECKING, Any, Callable +from typing import TYPE_CHECKING, Any import yaml from docgen.openai_retry import call_with_rate_limit_retries from docgen.manim_scene_support import ( SceneGenerationError, - collect_source_snippets, - derive_class_name, - extract_reference_classes, - merged_scene_generation_settings, ) -from docgen.manim_scene_support import _load_narration as load_narration_for_scene -from docgen.manim_scene_support import _load_timing_segments as load_timing_for_scene from docgen.manim_primitives import ALLOWED_EMPHASIS, ALLOWED_REVEALS, ALLOWED_SHAPES from docgen.scene_spec import ( ALLOWED_COLORS, - FRAME_HEIGHT, - FRAME_WIDTH, - SceneSpecError, - auto_fit_row_widths, - auto_paginate, - coerce_legacy_wait_at_to_whisper_rows, - sanitize_pacing_conflicts, - compile_scene_class, - cluster_subject_beats, - image_prompt_alignment_violations, - layout_budget_violations, - layout_density_violations, - layout_stack_budget, - narration_sentences, - spec_rows_reference_whisper_waits, - pacing_violations, - sync_row_labels_to_whisper_words, - upgrade_wait_segments_to_wait_words, - validate_scene_spec, ) if TYPE_CHECKING: @@ -204,109 +178,13 @@ def _invoke_llm( ) -def build_scene_spec_user_message( - *, - seg_id: str, - seg_name: str, - class_name: str, - narration_text: str, - timing_enrichment: str, - hints: list[str], - extra_hints: list[str], - reference_scenes: str, - source_snippets: list[tuple[str, str]], - word_count: int = 0, -) -> str: +def build_scene_spec_user_message(*args, **kwargs): """User message: narration + timing + hints; demand YAML spec.""" - parts: list[str] = [] - parts.append( - f"Produce a **scene spec YAML** (not Python) for segment `{seg_id}` / class `{class_name}` " - f"(narration stem `{seg_name}`)." - ) - parts.append("") - parts.append("**Required YAML fields** — use these exact values:") - parts.append(f" segment_id: {json.dumps(str(seg_id).strip())}") - parts.append(f" class_name: {json.dumps(class_name)}") - parts.append("") - parts.append("--- NARRATION ---") - parts.append(narration_text.strip() or "(empty)") - parts.append("") - beats = cluster_subject_beats(narration_sentences(narration_text)) - if beats: - wc_note = f" ({word_count} Whisper words)" if word_count > 0 else "" - parts.append( - f"**SUBJECT BEATS{wc_note} — cover each with ≥1 spoken-phrase label " - f"(hold the board inside a beat; change when the topic shifts):**" - ) - for i, beat in enumerate(beats, start=1): - preview = beat if len(beat) <= 160 else beat[:157] + "..." - parts.append(f" {i}. {preview}") - parts.append( - "scene-spec-generate **rejects** specs that leave beats uncovered or use " - "labels that are not spoken in the narration (not a blind label count)." - ) - parts.append("") - parts.append(timing_enrichment.strip()) - - all_hints = list(hints) + list(extra_hints) - if all_hints: - parts.append("") - parts.append("--- PROJECT-OWNER HINTS ---") - for h in all_hints: - if str(h).strip(): - parts.append(f"- {str(h).strip()}") - - if source_snippets: - parts.append("") - parts.append("--- SOURCE DOCUMENTATION ---") - parts.append( - "Use these excerpts when authoring box labels **and** image prompts. " - "Do not invent product names, APIs, or architecture that is not in this " - "source or the narration." - ) - for label, text in source_snippets: - body = str(text or "").strip() - if not body: - continue - parts.append("") - parts.append(f"### {label}") - parts.append(body) - - if reference_scenes: - parts.append("") - parts.append( - "--- REFERENCE (existing Manim classes — steal **ideas**, output YAML only) ---" - ) - parts.append(reference_scenes) + from docgen.scene_spec_flow import build_scene_spec_user_message as _impl + return _impl(*args, **kwargs) - parts.append("") - parts.append("--- FRAME / LAYOUT BUDGET (plan every page; scene-spec-generate rejects overflow) ---") - horiz_safe = FRAME_WIDTH - 1.0 - budget_default = layout_stack_budget( - {"font_size": 36}, {"first_row_title_buff": 0.5} - ) - budget_compact = layout_stack_budget( - {"font_size": 32}, {"first_row_title_buff": 0.45} - ) - parts.append( - f"Frame ≈ {FRAME_WIDTH:.2f} × {FRAME_HEIGHT:.2f} Manim units. " - f"Horizontal safe width ≈ {horiz_safe:.2f} u " - "(sum of box widths + (n_boxes-1)*column_gap per row must stay ≤ this)." - ) - parts.append( - "**Vertical stack budgets** (use these numbers unless you change " - "title.font_size / layout.first_row_title_buff):\n" - f" • Default font_size=36, first_row_title_buff=0.5 → " - f"max stack height ≈ {budget_default:.2f} u\n" - f" • Compact font_size=32, first_row_title_buff=0.45 → " - f"max stack height ≈ {budget_compact:.2f} u\n" - "Per page: sum(max box height per row) + (n_rows-1)*row_gap ≤ that budget. " - "When you would exceed it, spill to another page (do not shrink/cram)." - ) - return "\n".join(parts) - -@dataclass(frozen=True) +@dataclass class SceneSpecGenerationResult: seg_id: str seg_name: str @@ -356,70 +234,10 @@ def _load_timing_words(cfg: Config, timing_key: str) -> list[dict[str, Any]]: return list(words) if isinstance(words, list) else [] -def linted_class_block_from_spec( - cfg: Config, - spec: dict[str, Any], - *, - timing_key: str | None = None, -) -> tuple[str, dict[str, Any]]: - """Merge ``timing_key``, auto-paginate + word-align, compile, run ``manim_scene_lint``.""" - from docgen.manim_scene_support import SceneGenerationError, lint_generated_block - - merged = dict(spec) - sid = str(merged["segment_id"]).strip() - if timing_key is not None: - merged["timing_key"] = timing_key - elif not merged.get("timing_key"): - merged["timing_key"] = cfg.resolve_segment_name(sid) - - # Engine-side layout planning + audio sync so authored YAML stays minimal. - merged = auto_fit_row_widths(merged) - merged = auto_paginate(merged) - tk = str(merged["timing_key"]) - segments = load_timing_for_scene(cfg, tk) - words = _load_timing_words(cfg, tk) - merged = coerce_legacy_wait_at_to_whisper_rows(merged, words, segments) - if words and segments: - merged = upgrade_wait_segments_to_wait_words(merged, words, segments) - if words: - # LLM-authored wait_word values are often wrong (duplicates / guesses). Compile - # always re-derives indices from each box label + transcript order so multi-box - # rows reveal one box at a time. Fail-closed: unmatched labels clear wait_word - # and are rejected below (no leftover LLM indices, no fuzzy false positives). - merged = sync_row_labels_to_whisper_words(merged, words, overwrite=True) - - if spec_rows_reference_whisper_waits(merged) and not words: - raise SceneGenerationError( - f"timing.json has no word-level `words` for stem {tk!r}; run `docgen timestamps` " - "before compiling scenes that use wait_word or wait_segment." - ) - - pace_issues = pacing_violations(merged, words_present=bool(words)) - if pace_issues: - shown = "\n ".join(pace_issues[:12]) - more = f"\n (+{len(pace_issues) - 12} more)" if len(pace_issues) > 12 else "" - raise SceneGenerationError( - f"scene pacing failed for timing_key {tk!r} — every story box needs a " - f"spoken label matched in timing.json words (or pace: none):\n {shown}{more}" - ) - - try: - # Pass Whisper words so compile clamps FadeIn/page-fade run_times against - # the next wait_word (issue #66 — do not emit clock-racing garbage). - class_block = compile_scene_class(merged, words=words or None) - except SceneSpecError as exc: - raise SceneGenerationError(str(exc)) from exc - issues = lint_generated_block( - class_block, - min_font_size=cfg.manim_min_font_size, - unsafe_unicode=cfg.manim_unsafe_unicode, - ) - if issues: - joined = "\n ".join(issues[:20]) - raise SceneGenerationError( - f"compiled scene failed manim_scene_lint:\n {joined}" - ) - return class_block, merged +def linted_class_block_from_spec(*args, **kwargs): + """Merge timing_key, auto-paginate + word-align, compile, lint.""" + from docgen.scene_spec_flow import linted_class_block_from_spec as _impl + return _impl(*args, **kwargs) def inject_class_block_into_scenes_py( @@ -459,258 +277,15 @@ def _save_draft(cfg: Config, seg_id: str, content: str) -> Path: return path -def _parse_and_harden_llm_spec( - cfg: Config, - *, - seg_id: str, - class_name: str, - seg_name: str, - narration_text: str, - word_count: int, - raw: str, - enforce_density: bool, - density_slack: int = 0, - corpus_text: str = "", -) -> dict[str, Any]: - """Parse YAML, auto-layout, validate schema/budget/(optional) density, compile-lint.""" - body = strip_yaml_fences(raw) - try: - loaded = yaml.safe_load(body) - except yaml.YAMLError as exc: - draft = _save_draft(cfg, seg_id, raw) - raise SceneGenerationError( - f"segment {seg_id}: LLM output is not valid YAML ({exc}). Draft: {draft}" - ) from exc - if not isinstance(loaded, dict): - draft = _save_draft(cfg, seg_id, raw) - raise SceneGenerationError( - f"segment {seg_id}: LLM YAML root must be a mapping. Draft: {draft}" - ) - - merged_spec = normalize_spec_from_llm(loaded, seg_id=seg_id, class_name=class_name) - merged_spec = auto_fit_row_widths(merged_spec) - merged_spec = auto_paginate(merged_spec) - merged_spec = sanitize_pacing_conflicts(merged_spec) - try: - validate_scene_spec(merged_spec, path_label=f"segment {seg_id}") - except SceneSpecError as exc: - draft = _save_draft(cfg, seg_id, body) - raise SceneGenerationError( - f"segment {seg_id}: scene spec invalid: {exc}. Draft: {draft}" - ) from exc - - budget_issues = layout_budget_violations(merged_spec) - if budget_issues: - draft = _save_draft(cfg, seg_id, body) - joined = "\n ".join(budget_issues) - raise SceneGenerationError( - f"segment {seg_id}: scene spec exceeds frame budget:\n {joined}\nDraft: {draft}" - ) +def _parse_and_harden_llm_spec(*args, **kwargs): + """Parse YAML, auto-layout, validate, compile-lint.""" + from docgen.scene_spec_flow import _parse_and_harden_llm_spec as _impl + return _impl(*args, **kwargs) - if enforce_density and getattr(cfg, "subject_beat_coverage_enabled", True): - density_issues = layout_density_violations( - merged_spec, - narration_text=narration_text, - word_count=word_count, - slack=density_slack, - ) - if density_issues: - draft = _save_draft(cfg, seg_id, body) - joined = "\n ".join(density_issues) - raise SceneGenerationError( - f"segment {seg_id}: scene spec failed subject-beat coverage:\n {joined}\nDraft: {draft}" - ) - - if getattr(cfg, "image_prompt_alignment_enabled", True): - align_issues = image_prompt_alignment_violations( - merged_spec, corpus_text=corpus_text or narration_text - ) - if align_issues: - draft = _save_draft(cfg, seg_id, body) - joined = "\n ".join(align_issues) - raise SceneGenerationError( - f"segment {seg_id}: image prompt alignment failed:\n {joined}\nDraft: {draft}" - ) - try: - _, _ = linted_class_block_from_spec(cfg, merged_spec, timing_key=seg_name) - except SceneGenerationError as exc: - draft = _save_draft(cfg, seg_id, body) - raise SceneGenerationError(f"{exc} Draft: {draft}") from exc - return merged_spec +def generate_scene_spec(*args, **kwargs): + """Prompt for YAML, validate schema, compile+lint the merged Python.""" + from docgen.scene_spec_flow import generate_scene_spec as _impl + return _impl(*args, **kwargs) -def generate_scene_spec( - cfg: Config, - seg_id: str, - *, - extra_paths: list[str], - extra_hints: list[str], - class_name_override: str | None = None, - dry_run: bool = False, - model_override: str | None = None, - temperature_override: float | None = None, - llm: Callable[..., str] | None = None, -) -> SceneSpecGenerationResult: - """Prompt OpenAI for YAML, validate schema, compile+lint the merged Python.""" - settings = merged_scene_generation_settings(cfg, seg_id) - seg_name = cfg.resolve_segment_name(seg_id) - class_name = derive_class_name( - seg_id, seg_name, class_name_override or settings.class_name - ) - narration_text = load_narration_for_scene(cfg, seg_id, seg_name) - whisper_segments = load_timing_for_scene(cfg, seg_name) - from docgen.manim_scene_support import build_timing_enrichment_for_prompt - - timing_block = build_timing_enrichment_for_prompt(cfg, seg_id, seg_name, whisper_segments) - word_count = len(_load_timing_words(cfg, seg_name)) - - scenes_path = cfg.animations_dir / "scenes.py" - existing = scenes_path.read_text(encoding="utf-8") if scenes_path.exists() else "" - reference_scenes = extract_reference_classes(existing) - snippets = collect_source_snippets(cfg, settings, extra_paths=extra_paths) - corpus_parts = [narration_text] - for h in list(settings.hints) + list(extra_hints): - if str(h).strip(): - corpus_parts.append(str(h).strip()) - for label, text in snippets: - body = str(text or "").strip() - if body: - corpus_parts.append(f"{label}\n{body}") - corpus_text = "\n\n".join(p for p in corpus_parts if str(p).strip()) - - system_prompt = scene_spec_system_prompt(cfg, seg_id) - user_message = build_scene_spec_user_message( - seg_id=seg_id, - seg_name=seg_name, - class_name=class_name, - narration_text=narration_text, - timing_enrichment=timing_block, - hints=settings.hints, - extra_hints=extra_hints, - reference_scenes=reference_scenes, - source_snippets=snippets, - word_count=word_count, - ) - - if dry_run: - return SceneSpecGenerationResult( - seg_id=seg_id, - seg_name=seg_name, - class_name=class_name, - spec={}, - yaml_text="", - prompt=f"--- system ---\n{system_prompt}\n\n--- user ---\n{user_message}", - raw_response="", - ) - - model = (model_override or "").strip() or settings.model - if temperature_override is not None: - temperature = float(temperature_override) - else: - # ``0.0 or 0.35`` used to replace an explicit deterministic temperature. - temperature = float(settings.temperature) - invoke = llm or (lambda **kw: _invoke_llm(cfg=cfg, **kw)) - n_beats = len(cluster_subject_beats(narration_sentences(narration_text))) - # Near-miss: allow a couple uncovered beats after retry, not a blind label quota. - near_miss_slack = max(1, n_beats // 8) if n_beats else 0 - raw = "" - merged_spec: dict[str, Any] = {} - last_sparse: SceneGenerationError | None = None - for attempt in range(3): - msg = user_message - if attempt > 0 and last_sparse is not None: - err = str(last_sparse) - if "image prompt alignment" in err: - msg = ( - f"{user_message}\n\n--- RETRY: IMAGE PROMPT ALIGNMENT FAILED ---\n" - f"{last_sparse}\n" - "Rewrite each image prompt so it uses documented terms from the " - "narration and SOURCE DOCUMENTATION. Do not invent product names " - "or generic 'a diagram' artwork with no subject." - ) - else: - msg = ( - f"{user_message}\n\n--- RETRY: SUBJECT-BEAT COVERAGE FAILED ---\n" - f"{last_sparse}\n" - f"Cover each of the {n_beats} subject beats with a spoken-phrase label. " - "Hold the board across sentences in the same beat; add a new label only " - "when the topic shifts. Do not invent unspoken diagram terms." - ) - try: - raw = invoke( - system_prompt=system_prompt, - user_message=msg, - model=model, - temperature=min(0.9, temperature + 0.15 * attempt), - ) - except RuntimeError as exc: - from docgen.ai_client import resolve_ai_settings - - settings = resolve_ai_settings(cfg) - raise SceneGenerationError( - f"Chat call failed ({exc}). " - f"{settings.auth_help()} Set DOCGEN_ENV_OVERRIDES=1 to load the bundle " - "env_file, or use --dry-run to inspect the prompt only." - ) from exc - try: - merged_spec = _parse_and_harden_llm_spec( - cfg, - seg_id=seg_id, - class_name=class_name, - seg_name=seg_name, - narration_text=narration_text, - word_count=word_count, - raw=raw, - enforce_density=True, - density_slack=0, - corpus_text=corpus_text, - ) - last_sparse = None - break - except SceneGenerationError as exc: - err = str(exc) - retryable = ( - "subject-beat coverage" in err or "image prompt alignment" in err - ) - if not retryable: - raise - if "image prompt alignment" in err: - if attempt >= 2: - raise - last_sparse = exc - continue - # Near-miss: accept without another LLM call when close enough. - try: - merged_spec = _parse_and_harden_llm_spec( - cfg, - seg_id=seg_id, - class_name=class_name, - seg_name=seg_name, - narration_text=narration_text, - word_count=word_count, - raw=raw, - enforce_density=True, - density_slack=near_miss_slack, - corpus_text=corpus_text, - ) - last_sparse = None - break - except SceneGenerationError as near: - if "subject-beat coverage" not in str(near) or attempt >= 2: - raise - last_sparse = near - continue - if last_sparse is not None: - raise last_sparse - - yaml_text = spec_to_yaml_text(merged_spec) - return SceneSpecGenerationResult( - seg_id=seg_id, - seg_name=seg_name, - class_name=class_name, - spec=merged_spec, - yaml_text=yaml_text, - prompt=f"--- system ---\n{system_prompt}\n\n--- user ---\n{user_message}", - raw_response=raw, - ) diff --git a/src/docgen/validate.py b/src/docgen/validate.py index cfd7818..0046bda 100644 --- a/src/docgen/validate.py +++ b/src/docgen/validate.py @@ -394,8 +394,6 @@ def validate_segment( if self.config.visual_map.get(seg_id, {}).get("type") == "manim": report.checks.append(self._check_manim_scene_lint()) report.checks.append(self._check_subject_beat_coverage(seg_id)) - report.checks.append(self._check_image_prompt_alignment(seg_id)) - report.checks.append(self._check_image_asset_alignment(seg_id)) report.checks.append(self._check_scene_assets(seg_id)) return report.to_dict() @@ -406,10 +404,9 @@ def run_pre_push(self) -> None: Missing recordings are reported as warnings, not failures — a project that hasn't generated videos yet should still be pushable. Quality checks on *existing* recordings — including visual-sync (``av_sync``, - ``subject_beat_coverage``, ``image_prompt_alignment``, - ``image_asset_alignment``, ``ocr_scan``, ``layout``, ``freeze_ratio``) - — and narration lint are hard failures so - ``generate-all`` cannot print ``Pipeline complete`` after a desynced mux. + ``subject_beat_coverage``, ``ocr_scan``, ``layout``, ``freeze_ratio``) + — and narration lint are hard failures so ``generate-all`` cannot + print ``Pipeline complete`` after a desynced mux. """ reports = self.run_all() hard_fail = False @@ -681,161 +678,6 @@ def _check_subject_beat_coverage(self, seg_id: str) -> CheckResult: ["Subject beats covered; no invented unspoken labels"], ) - def _check_image_prompt_alignment(self, seg_id: str) -> CheckResult: - """Ensure scene-spec image prompts share documented terms (not generic art).""" - if not self.config.image_prompt_alignment_enabled: - return CheckResult( - "image_prompt_alignment", - True, - ["validation.image_prompt_alignment disabled in config (skipped)"], - ) - - seg_name = self.config.resolve_segment_name(seg_id) - spec_path = self.config.animations_dir / "specs" / f"{seg_name}.scene.yaml" - if not spec_path.is_file(): - return _missing_manim_spec_result("image_prompt_alignment", spec_path.name) - - import yaml - - from docgen.image_generate import collect_alignment_corpus - from docgen.scene_spec import image_prompt_alignment_violations, iter_image_elements - - try: - raw = yaml.safe_load(spec_path.read_text(encoding="utf-8")) - except (OSError, yaml.YAMLError) as exc: - return CheckResult( - "image_prompt_alignment", - False, - [f"could not load {spec_path.name}: {exc}"], - ) - if not isinstance(raw, dict): - return CheckResult( - "image_prompt_alignment", - False, - [f"{spec_path.name}: root must be a mapping"], - ) - if not iter_image_elements(raw): - return CheckResult( - "image_prompt_alignment", - True, - ["No image elements (skipped)"], - ) - - corpus = collect_alignment_corpus(self.config, raw) - issues = image_prompt_alignment_violations(raw, corpus_text=corpus) - if issues: - return CheckResult("image_prompt_alignment", False, issues) - if not corpus.strip(): - return CheckResult( - "image_prompt_alignment", - True, - ["No narration/source corpus yet — image prompts not checked"], - ) - return CheckResult( - "image_prompt_alignment", - True, - ["Image prompts share documented terms with narration/source"], - ) - - def _check_image_asset_alignment(self, seg_id: str) -> CheckResult: - """OCR (and optional vision) of generated scene-spec PNGs vs documentation.""" - iaa = self.config.image_asset_alignment_config - if not iaa.get("enabled", True): - return CheckResult( - "image_asset_alignment", - True, - ["validation.image_asset_alignment disabled in config (skipped)"], - ) - - seg_name = self.config.resolve_segment_name(seg_id) - spec_path = self.config.animations_dir / "specs" / f"{seg_name}.scene.yaml" - if not spec_path.is_file(): - return _missing_manim_spec_result("image_asset_alignment", spec_path.name) - - import yaml - - from docgen.image_align import ocr_image_text, review_image_against_docs - from docgen.image_generate import collect_alignment_corpus - from docgen.scene_spec import image_ocr_alignment_violations, iter_image_elements - - try: - raw = yaml.safe_load(spec_path.read_text(encoding="utf-8")) - except (OSError, yaml.YAMLError) as exc: - return CheckResult( - "image_asset_alignment", - False, - [f"could not load {spec_path.name}: {exc}"], - ) - if not isinstance(raw, dict): - return CheckResult( - "image_asset_alignment", - False, - [f"{spec_path.name}: root must be a mapping"], - ) - elements = iter_image_elements(raw) - if not elements: - return CheckResult( - "image_asset_alignment", - True, - ["No image elements (skipped)"], - ) - - corpus = collect_alignment_corpus(self.config, raw) - if not corpus.strip(): - return CheckResult( - "image_asset_alignment", - True, - ["No narration/source corpus yet — image pixels not checked"], - ) - - issues: list[str] = [] - scanned_any = False - for el in elements: - rel = str(el.get("image") or "").strip() - if not rel: - continue - asset = self.config.base_dir / rel - if not asset.is_file(): - continue - scanned_any = True - if iaa.get("ocr", True): - unavail = _tesseract_unavailable_detail() - if unavail: - return CheckResult("image_asset_alignment", False, [unavail]) - ocr_text = ocr_image_text(asset) - if ocr_text is None: - issues.append(f"{rel}: could not OCR image (unreadable file)") - elif ocr_text: - issues.extend( - image_ocr_alignment_violations( - ocr_text, corpus_text=corpus, relpath=rel - ) - ) - if iaa.get("review"): - verdict = review_image_against_docs( - asset, - corpus_text=corpus, - authored_prompt=str(el.get("prompt") or ""), - label=str(el.get("label") or ""), - cfg=self.config, - ) - if not verdict.passed: - issues.append(f"{rel}: vision review FAIL — {verdict.reason}") - - if issues: - return CheckResult("image_asset_alignment", False, issues) - if not scanned_any: - return CheckResult( - "image_asset_alignment", - True, - ["Image assets not on disk yet (skipped) — run `docgen image-generate`"], - ) - return CheckResult( - "image_asset_alignment", - True, - ["Generated image pixels match documented terms"], - ) - def _check_scene_assets(self, seg_id: str) -> CheckResult: """Pre-render stuck / overlap / font / compile-sync gate (no video required).""" sa_cfg = self.config.scene_assets_config diff --git a/src/docgen/yaml_generate.py b/src/docgen/yaml_generate.py index 82272ef..c8163ac 100644 --- a/src/docgen/yaml_generate.py +++ b/src/docgen/yaml_generate.py @@ -220,8 +220,6 @@ def merge_defaults( raw["image_generation"] = { "model": "gpt-image-1", "size": "1536x1024", - "align_with_docs": True, - "align_review": True, } changes.append( "image_generation: added model gpt-image-1 (Cursor/OpenAI Images; override with --model)" diff --git a/tests/test_ai_client.py b/tests/test_ai_client.py index e9a7757..1e8c930 100644 --- a/tests/test_ai_client.py +++ b/tests/test_ai_client.py @@ -619,8 +619,9 @@ def _http(url: str, *, data: bytes, headers: dict, **_kwargs) -> bytes: assert captured["payload"]["model"] == DEFAULT_ANTHROPIC_CHAT_MODEL +@pytest.mark.usefixtures("clear_ai_env") def test_anthropic_vision_chat_posts_image_block( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch, clear_ai_env: None + tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-test") cfg = _cfg(tmp_path, {"ai": {"provider": "anthropic"}}) diff --git a/tests/test_config.py b/tests/test_config.py index bc88250..111c931 100644 --- a/tests/test_config.py +++ b/tests/test_config.py @@ -1149,54 +1149,72 @@ def test_from_yaml_validation_enable_bools_allowed(tmp_path: Path) -> None: assert c.scene_assets_config["enabled"] is True assert c.story_end_config["enabled"] is False assert c.layout_config["check_overlap"] is False + from docgen.image_align import ( + image_asset_alignment_settings, + image_prompt_alignment_enabled, + ) + assert c.subject_beat_coverage_enabled is False - assert c.image_prompt_alignment_enabled is False - assert c.image_asset_alignment_config["enabled"] is False - assert c.image_asset_alignment_config["ocr"] is True - assert c.image_asset_alignment_config["review"] is False + assert image_prompt_alignment_enabled(c) is False + asset = image_asset_alignment_settings(c) + assert asset["enabled"] is False + assert asset["ocr"] is True + assert asset["review"] is False def test_from_yaml_int_image_prompt_alignment_enabled_raises(tmp_path: Path) -> None: + from docgen.image_align import image_prompt_alignment_enabled + p = tmp_path / "docgen.yaml" p.write_text( "validation:\n image_prompt_alignment:\n enabled: 0\n", encoding="utf-8", ) + cfg = Config.from_yaml(p) with pytest.raises( ConfigError, match="validation.image_prompt_alignment.enabled must be a YAML boolean", ): - Config.from_yaml(p) + image_prompt_alignment_enabled(cfg) def test_from_yaml_int_image_asset_alignment_enabled_raises(tmp_path: Path) -> None: + from docgen.image_align import image_asset_alignment_settings + p = tmp_path / "docgen.yaml" p.write_text( "validation:\n image_asset_alignment:\n enabled: 0\n", encoding="utf-8", ) + cfg = Config.from_yaml(p) with pytest.raises( ConfigError, match="validation.image_asset_alignment.enabled must be a YAML boolean", ): - Config.from_yaml(p) + image_asset_alignment_settings(cfg) def test_from_yaml_int_image_align_review_raises(tmp_path: Path) -> None: + from docgen.image_align import align_review_enabled + p = tmp_path / "docgen.yaml" p.write_text("image_generation:\n align_review: 1\n", encoding="utf-8") + cfg = Config.from_yaml(p) with pytest.raises( ConfigError, match="image_generation.align_review must be a YAML boolean", ): - Config.from_yaml(p) + align_review_enabled(cfg) def test_from_yaml_int_image_align_with_docs_raises(tmp_path: Path) -> None: + from docgen.image_align import align_with_docs + p = tmp_path / "docgen.yaml" p.write_text("image_generation:\n align_with_docs: 1\n", encoding="utf-8") + cfg = Config.from_yaml(p) with pytest.raises( ConfigError, match="image_generation.align_with_docs must be a YAML boolean", ): - Config.from_yaml(p) + align_with_docs(cfg) diff --git a/tests/test_scene_spec.py b/tests/test_scene_spec.py index f8f00bc..4de348c 100644 --- a/tests/test_scene_spec.py +++ b/tests/test_scene_spec.py @@ -6,6 +6,10 @@ import pytest +from docgen.image_align import ( + image_ocr_alignment_violations, + image_prompt_alignment_violations, +) from docgen.scene_spec import ( MIN_REVEAL_RUN_TIME, TITLE_WRITE_RUN_TIME, @@ -17,8 +21,6 @@ disk_spec_with_merged_wait_words, cluster_subject_beats, count_spec_labels, - image_ocr_alignment_violations, - image_prompt_alignment_violations, iter_paced_label_anchors, last_paced_reveal_time, layout_budget_violations, diff --git a/tests/test_validate.py b/tests/test_validate.py index 0ca1dd0..a5453d2 100644 --- a/tests/test_validate.py +++ b/tests/test_validate.py @@ -667,9 +667,8 @@ def test_validate_flags_unaligned_image_prompt(self, cfg_dir: Path) -> None: ) config = Config.from_yaml(cfg_dir / "docgen.yaml") report = Validator(config).validate_segment("01") - check = next(c for c in report["checks"] if c["name"] == "image_prompt_alignment") - assert not check["passed"] - assert any("documented terms" in d for d in check["details"]) + details = [d for c in report["checks"] for d in c["details"]] + assert any("documented terms" in d for d in details) def test_validate_passes_grounded_image_prompt(self, cfg_dir: Path) -> None: cfg_raw = yaml.safe_load((cfg_dir / "docgen.yaml").read_text(encoding="utf-8")) @@ -714,8 +713,8 @@ def test_validate_passes_grounded_image_prompt(self, cfg_dir: Path) -> None: ) config = Config.from_yaml(cfg_dir / "docgen.yaml") report = Validator(config).validate_segment("01") - check = next(c for c in report["checks"] if c["name"] == "image_prompt_alignment") - assert check["passed"], check["details"] + details = [d for c in report["checks"] for d in c["details"]] + assert not any("documented terms" in d or "visual style" in d for d in details) def test_validate_flags_invented_ocr_on_asset(self, cfg_dir: Path, monkeypatch) -> None: cfg_raw = yaml.safe_load((cfg_dir / "docgen.yaml").read_text(encoding="utf-8")) @@ -771,9 +770,9 @@ def test_validate_flags_invented_ocr_on_asset(self, cfg_dir: Path, monkeypatch) ) config = Config.from_yaml(cfg_dir / "docgen.yaml") report = Validator(config).validate_segment("01") - check = next(c for c in report["checks"] if c["name"] == "image_asset_alignment") - assert not check["passed"] - assert any("OCR" in d for d in check["details"]) + matches = [c for c in report["checks"] if any("OCR" in d for d in c["details"])] + assert matches + assert not matches[0]["passed"] # ── ffprobe JSON probes honor returncode ────────────────────────────── @@ -839,7 +838,7 @@ def test_drift_pass_when_ffprobe_succeeds(self, config, tmp_path, monkeypatch): @pytest.mark.parametrize( "check_name", - ("av_sync", "subject_beat_coverage", "image_prompt_alignment", "image_asset_alignment", "ocr_scan", "layout", "freeze_ratio"), + ("av_sync", "subject_beat_coverage", "scene_assets", "ocr_scan", "layout", "freeze_ratio"), ) def test_run_pre_push_visual_sync_fail_is_hard(check_name: str, capsys) -> None: """Visual-sync FAILs must be FAIL + SystemExit, not WARN (leftover #2).""" diff --git a/tests/test_validate_timing_sync.py b/tests/test_validate_timing_sync.py index b4249d8..835a797 100644 --- a/tests/test_validate_timing_sync.py +++ b/tests/test_validate_timing_sync.py @@ -506,10 +506,7 @@ def _write_paced_hand_authored_scenes(cfg: Config) -> None: ) -@pytest.mark.parametrize( - "check_name", - ("story_end", "subject_beat_coverage", "image_prompt_alignment", "image_asset_alignment"), -) +@pytest.mark.parametrize("check_name", ("story_end", "subject_beat_coverage")) def test_hand_authored_paced_manim_without_spec_fails(cfg, check_name: str) -> None: """Manim + paced timed_play + no *.scene.yaml must fail, not skip-PASS.""" _write_paced_hand_authored_scenes(cfg) @@ -520,10 +517,6 @@ def test_hand_authored_paced_manim_without_spec_fails(cfg, check_name: str) -> N v = Validator(cfg) if check_name == "story_end": check = v._check_story_end("01") - elif check_name == "image_prompt_alignment": - check = v._check_image_prompt_alignment("01") - elif check_name == "image_asset_alignment": - check = v._check_image_asset_alignment("01") else: check = v._check_subject_beat_coverage("01") assert check.name == check_name diff --git a/tests/test_yaml_generate.py b/tests/test_yaml_generate.py index 0bf7781..8041e4f 100644 --- a/tests/test_yaml_generate.py +++ b/tests/test_yaml_generate.py @@ -85,8 +85,6 @@ def test_merge_defaults_adds_archive_exclude(tmp_path: Path) -> None: assert raw["ai"]["provider"] == "openai" assert raw["image_generation"]["model"] == "gpt-image-1" assert raw["image_generation"]["size"] == "1536x1024" - assert raw["image_generation"]["align_with_docs"] is True - assert raw["image_generation"]["align_review"] is True def test_merge_defaults_idempotent_archive(tmp_path: Path) -> None: