From de94d4dba3dbaf9ece4b955bc50cd563df127892 Mon Sep 17 00:00:00 2001 From: calesthio Date: Fri, 10 Apr 2026 16:42:39 -0700 Subject: [PATCH] Documentary Montage hardening plus governance fixes --- .env.example | 1 + README.md | 44 ++- pipeline_defs/documentary-montage.yaml | 4 +- remotion-composer/src/Root.tsx | 22 ++ remotion-composer/src/TitledVideo.tsx | 263 ++++++++++++++++++ remotion-composer/titled_video_props.json | 8 + schemas/artifacts/edit_decisions.schema.json | 2 +- schemas/artifacts/proposal_packet.schema.json | 2 +- schemas/artifacts/render_report.schema.json | 2 +- .../documentary-montage/asset-director.md | 9 +- .../documentary-montage/compose-director.md | 61 ++-- .../documentary-montage/edit-director.md | 6 + .../documentary-montage/executive-producer.md | 2 +- .../documentary-montage/idea-director.md | 5 +- .../documentary-montage/scene-director.md | 4 +- tests/tools/test_documentary_governance.py | 139 +++++++++ tests/tools/test_stock_source_adapters.py | 29 ++ tools/tool_registry.py | 10 + tools/video/corpus_builder.py | 65 ++++- tools/video/stock_sources/__init__.py | 107 +++++-- tools/video/stock_sources/archive_org.py | 7 + tools/video/stock_sources/base.py | 5 +- tools/video/stock_sources/nasa.py | 8 + tools/video/stock_sources/pexels.py | 8 + tools/video/stock_sources/unsplash.py | 184 ++++++++++++ tools/video/stock_sources/wikimedia.py | 196 +++++++++++++ tools/video/video_compose.py | 12 +- 27 files changed, 1107 insertions(+), 98 deletions(-) create mode 100644 remotion-composer/src/TitledVideo.tsx create mode 100644 remotion-composer/titled_video_props.json create mode 100644 tests/tools/test_documentary_governance.py create mode 100644 tests/tools/test_stock_source_adapters.py create mode 100644 tools/video/stock_sources/unsplash.py create mode 100644 tools/video/stock_sources/wikimedia.py diff --git a/.env.example b/.env.example index 2debf53..884c01c 100644 --- a/.env.example +++ b/.env.example @@ -28,6 +28,7 @@ MODAL_LTX2_ENDPOINT_URL= # Modal self-hosted LTX-2 endpoint (optional) # --- Stock Media --- PEXELS_API_KEY= # Pexels stock footage/images (free) PIXABAY_API_KEY= # Pixabay stock footage/images (free) +UNSPLASH_ACCESS_KEY= # Unsplash stock images (free developer key) # --- Analysis --- HF_TOKEN= # HuggingFace token — enables speaker diarization in transcriber diff --git a/README.md b/README.md index ab99259..77f4197 100644 --- a/README.md +++ b/README.md @@ -31,6 +31,8 @@ Turn your AI coding assistant into a full video production studio. Describe what you want in plain language — your agent handles research, scripting, asset generation, editing, and final composition. +**Important distinction:** OpenMontage can make image-based videos, but it can also make a real **video video** for free/open-source workflows: the agent builds a corpus from free stock footage and open archives, retrieves actual motion clips, edits them into a timeline, and renders a finished piece. That is not the usual "animate a handful of stills and call it video" trick. +
@@ -117,6 +119,12 @@ Open the project in your AI coding assistant and tell it what you want: "Make a 60-second animated explainer about how neural networks learn" ``` +Or if you want the real-footage path: + +```text +"Make a 75-second documentary montage about city life in the rain. Use real footage only, no narration, elegiac tone, with music." +``` + That's it. The agent researches your topic with live web search, generates AI images, writes and narrates the script with voice direction, finds royalty-free background music automatically, burns in word-level subtitles, and renders the final video. Before you see anything, the system runs a multi-point self-review — ffprobe validation, frame sampling, audio level analysis, delivery promise verification, and subtitle checks. Every provider selection is scored across 7 dimensions with an auditable decision log. Every creative decision gets your approval. > **No `make`?** Run manually: `pip install -r requirements.txt && cd remotion-composer && npm install && cd .. && pip install piper-tts && cp .env.example .env` @@ -149,8 +157,9 @@ This repo is built for agentic operation. If you're an OpenClaw-style agent, her FAL_KEY=your-key # FLUX images + Google Veo, Kling, MiniMax video + Recraft images # Free stock media: -PEXELS_API_KEY=your-key # Free — stock footage and images -PIXABAY_API_KEY=your-key # Free — stock footage and images +PEXELS_API_KEY=your-key # Free stock footage and images +PIXABAY_API_KEY=your-key # Free stock footage and images +UNSPLASH_ACCESS_KEY=your-key # Free stock images # Music: SUNO_API_KEY=your-key # Full songs, instrumentals, any genre @@ -183,17 +192,23 @@ VIDEO_GEN_LOCAL_MODEL=wan2.1-1.3b # or wan2.1-14b, hunyuan-1.5, ltx2-local, cog ## What You Get With Zero API Keys -You don't need any API keys to make real videos. Out of the box, `make setup` gives you: +You don't need paid API keys to make real videos. Out of the box, `make setup` gives you: | Capability | Free Tool | What It Does | |-----------|-----------|-------------| | **Narration** | Piper TTS | Free offline text-to-speech — real human-sounding narration | -| **Visuals** | Pexels + Pixabay | Free stock images and footage (API keys are free to get) | +| **Open footage** | Archive.org + NASA + Wikimedia Commons | Free/open archival footage, educational media, and documentary texture | +| **Extra stock** | Pexels + Unsplash + Pixabay | Free stock footage/images (developer keys are free to get) | | **Composition** | Remotion | Turns still images into animated video with spring physics, transitions, typography, and TikTok-style captions | | **Post-production** | FFmpeg | Encoding, subtitle burn-in, audio mixing, color grading | | **Subtitles** | Built-in | Auto-generated captions with word-level timing | -**The zero-key path:** Piper narrates your script, stock images provide the visuals, and Remotion animates everything into a polished video with transitions, text overlays, and synced captions. Add API keys later to unlock AI-generated images, video clips, premium voices, and music. +**Two free-ish paths:** + +- **Image-based video:** Piper narrates your script, images provide the visuals, and Remotion animates them into a polished edit. +- **Real-footage video:** the documentary montage pipeline builds a CLIP-searchable corpus from Archive.org, NASA, Wikimedia Commons, and optional free-key sources like Pexels and Unsplash, then cuts together actual motion footage into a finished video. + +If you want the second one, prompt for a **documentary montage**, **tone poem**, or **stock-footage collage**, and explicitly say **use real footage only**. --- @@ -217,6 +232,14 @@ Copy any of these into your AI coding assistant after setup. Each one runs a ful > "Make a data-driven explainer about coffee consumption around the world" +### Free real-footage documentary path + +> "Make a 90-second documentary montage about what a city feels like at 4am. Use real footage only, no narration, elegiac tone." + +> "Create a 60-second Adam-Curtis-style archival collage about 1950s consumer optimism. Prefer Archive.org and Wikimedia footage." + +> "Cut together a dreamlike montage about coming home in the rain using real stock footage only. Music yes, narration no." + ### With an image/video provider configured (~$0.15–$1.50) > "Create a 30-second Ghibli-style animated video of a magical floating library in the clouds at golden hour" @@ -248,7 +271,7 @@ Each pipeline is a complete production workflow, from idea to finished video. | **Avatar Spokesperson** | Avatar-driven presenter videos | Corporate comms, training, announcements | | **Cinematic** | Trailer, teaser, and mood-driven edits | Brand films, teasers, promotional content | | **Clip Factory** | Batch of ranked short-form clips from one long source | Repurposing long content for social media | -| **Documentary Montage** | Thematic montage cut from a CLIP-indexed corpus of free stock footage (Pexels, Archive.org, NASA) | Video essays, mood pieces, retrieval-first B-roll edits | +| **Documentary Montage** | Thematic montage cut from a CLIP-indexed corpus of free stock footage and open archives (Pexels, Archive.org, NASA, Wikimedia, Unsplash) | Video essays, mood pieces, retrieval-first B-roll edits, real-footage videos without paid generation APIs | | **Hybrid** | Source footage + AI-generated support visuals | Enhancing existing footage with graphics | | **Localization & Dub** | Subtitle, dub, and translate existing video | Multi-language distribution | | **Podcast Repurpose** | Podcast highlights to video | Podcast marketing, audiogram videos | @@ -271,12 +294,15 @@ Each stage has a dedicated **director skill** — a markdown instruction file th Most AI video tools give you a single clip from a prompt. OpenMontage gives you an **end-to-end production pipeline** — the same structured process a real production team follows, automated by your AI agent. +Most "free AI video" stacks quietly mean "animate still images." OpenMontage can do that too, but it can also build a finished video from **real footage** pulled from free/open sources, ranked semantically, edited intentionally, and rendered as a proper timeline. + Edit your own talking-head footage. Generate a fully animated explainer from scratch. Cut a 2-hour podcast into a dozen social clips. Translate and dub your content into 10 languages. Build a cinematic brand teaser from stock footage and AI-generated scenes. **If a production team can make it, OpenMontage can orchestrate it.** - **12 production pipelines** — explainers, talking heads, screen demos, cinematic trailers, animations, podcasts, localization, documentary montages, and more - **52 production tools** — spanning video generation, image creation, text-to-speech, music, audio mixing, subtitles, enhancement, and analysis - **400+ agent skills** — production skills, pipeline directors, creative techniques, quality checklists, and deep technology knowledge packs that teach the agent how to use every tool like an expert - **Reference-driven creation** — paste a video you like and the agent turns it into a grounded, differentiated production plan instead of forcing you to invent the perfect prompt from scratch +- **Real-footage documentary creation without paid video models** — build actual edited videos from free/open motion footage and archival sources, not just Ken Burns over images - **Live web research built in** — before writing a single word of script, the agent runs 15-25+ web searches across YouTube, Reddit, news sites, and academic sources to ground your video in real, current data - **Both free/local AND cloud providers** — every capability supports open-source local alternatives alongside premium APIs. Use what you have. - **No vendor lock-in** — swap providers freely. The scored selector ranks every provider across 7 dimensions (task fit, output quality, control, reliability, cost efficiency, latency, continuity) and picks the best match automatically. @@ -371,7 +397,7 @@ Each tool declares which Layer 3 skills it relies on. The agent reads Layer 1 to > **Full setup guide with pricing and free tiers:** [`docs/PROVIDERS.md`](docs/PROVIDERS.md)
-Video Generation — 13 providers +Video Generation — 14 providers | Provider | Type | Notes | |----------|------|-------| @@ -388,11 +414,12 @@ Each tool declares which Layer 3 skills it relies on. The agent reads Layer 1 to | **LTX-Video** | Local GPU / Modal | Free locally, or self-hosted cloud | | **Pexels** | Stock | Free stock footage | | **Pixabay** | Stock | Free stock footage | +| **Wikimedia Commons** | Stock | Free/open stock footage and archival video |
-Image Generation — 9 tools/providers +Image Generation — 10 tools/providers | Provider | Type | Notes | |----------|------|-------| @@ -404,6 +431,7 @@ Each tool declares which Layer 3 skills it relies on. The agent reads Layer 1 to | **Local Diffusion** | Local GPU | Stable Diffusion, free | | **Pexels** | Stock | Free stock images | | **Pixabay** | Stock | Free stock images | +| **Unsplash** | Stock | Free stock images | | **ManimCE** | Local | Mathematical animations |
diff --git a/pipeline_defs/documentary-montage.yaml b/pipeline_defs/documentary-montage.yaml index 25bc3f4..4102b81 100644 --- a/pipeline_defs/documentary-montage.yaml +++ b/pipeline_defs/documentary-montage.yaml @@ -2,7 +2,8 @@ name: documentary-montage version: "1.0" description: > Retrieval-first thematic montage pipeline. Builds a semantic corpus of real-world - footage from Pexels, Archive.org (Prelinger et al.), and NASA, then uses CLIP-based + footage from Pexels, Archive.org (Prelinger et al.), NASA, Wikimedia Commons, and + Unsplash, then uses CLIP-based retrieval to fill slot descriptions from a thematic brief. The edit arranges clips by narrative beat with music sync and uniform color grade across mixed-era footage. Inspired by Adam Curtis / Chris Marker / Errol Morris tone poems. @@ -120,6 +121,7 @@ stages: checkpoint_required: true human_approval_default: true review_focus: + - renderer_family is locked to documentary-montage - Hero slots hold longest; mid-sequence cutaways shortest - No two adjacent cuts share subject AND scale - Transition vocabulary is at most 4 distinct values diff --git a/remotion-composer/src/Root.tsx b/remotion-composer/src/Root.tsx index cc78026..2ce866f 100644 --- a/remotion-composer/src/Root.tsx +++ b/remotion-composer/src/Root.tsx @@ -6,6 +6,10 @@ import { } from "./CinematicRenderer"; import { signalFromTomorrowWithMusicFixture } from "./cinematic/fixtures"; import { TalkingHead, TalkingHeadProps } from "./TalkingHead"; +import { + TitledVideo, + calculateTitledVideoMetadata, +} from "./TitledVideo"; // --------------------------------------------------------------------------- // Theme System — prevents every video from looking like dark fintech @@ -181,6 +185,24 @@ export const Root: React.FC = () => { highlightColor: "#22D3EE", }} /> + ); }; diff --git a/remotion-composer/src/TitledVideo.tsx b/remotion-composer/src/TitledVideo.tsx new file mode 100644 index 0000000..baaa0b8 --- /dev/null +++ b/remotion-composer/src/TitledVideo.tsx @@ -0,0 +1,263 @@ +import { + AbsoluteFill, + CalculateMetadataFunction, + OffthreadVideo, + Sequence, + interpolate, + spring, + staticFile, + useCurrentFrame, + useVideoConfig, +} from "remotion"; +import { getVideoMetadata } from "@remotion/media-utils"; +import { loadFont } from "@remotion/google-fonts/PlayfairDisplay"; + +// Editorial serif for the tagline — Playfair Display at its boldest weight. +// Loaded once at module scope so every render reuses the same font face. +const { fontFamily } = loadFont("normal", { + weights: ["400", "700", "900"], + subsets: ["latin"], +}); + +export interface TitledVideoProps { + videoSrc: string; + tagline: string; + // When the tagline starts animating in, in seconds from the start of the video. + taglineInSeconds: number; + // When the tagline should be fully gone. If omitted, it holds to the end of the clip. + taglineOutSeconds?: number; + // Pixels from the top of the frame where the tagline sits. Upper-third by default. + topPx?: number; + // Optional override for the font size. + fontSize?: number; + // Accent color used for the underline and the glow halo. + accentColor?: string; +} + +// Resolve asset path — handle URLs, absolute paths, and public/ relative paths. +// Mirrors the helper in Explainer.tsx so absolute Windows/Unix paths work. +function resolveAsset(src: string): string { + if ( + src.startsWith("http://") || + src.startsWith("https://") || + src.startsWith("data:") + ) { + return src; + } + const clean = src.replace(/^file:\/\/\/?/, ""); + if (clean.startsWith("/") || /^[A-Za-z]:[\\/]/.test(clean)) { + return `file:///${clean.replace(/\\/g, "/")}`; + } + return staticFile(clean); +} + +// --------------------------------------------------------------------------- +// EditorialTagline — big bold serif, upper-third, drawn underline, warm glow. +// Letter-by-letter spring entrance. Designed to feel like a printed headline, +// not a subtitle. +// --------------------------------------------------------------------------- +const EditorialTagline: React.FC<{ + text: string; + topPx: number; + fontSize: number; + accentColor: string; +}> = ({ text, topPx, fontSize, accentColor }) => { + const frame = useCurrentFrame(); + const { fps, durationInFrames } = useVideoConfig(); + + const chars = text.split(""); + + // Hold the title until the last 8 frames, then ease out. + const fadeOut = interpolate( + frame, + [durationInFrames - 10, durationInFrames - 2], + [1, 0], + { extrapolateLeft: "clamp", extrapolateRight: "clamp" } + ); + + // Underline sweeps in after ~12 frames so the eye lands on the words first. + const underline = spring({ + frame: frame - 12, + fps, + config: { damping: 18, stiffness: 70 }, + }); + + // Glow pulses gently on a long period so the shine reads without flickering. + const glowPulse = + 0.6 + 0.4 * Math.sin((frame / fps) * Math.PI * 0.6); + + // Width target for the underline — measured in CSS px. + // We let it grow up to 70% of a max line width so it visually underscores + // the phrase no matter how many characters the tagline has. + const estimatedTextWidth = Math.min( + chars.length * fontSize * 0.48, + 1600 + ); + + return ( + + {/* Soft top-of-frame darkening so the bright type sits on a scrim + without covering the whole video. Gradient fades out by ~35% */} + +
+ {/* Tagline text */} +
+ {chars.map((char, i) => { + const delay = i * 1.6; + const charSpring = spring({ + frame: frame - delay, + fps, + config: { damping: 14, stiffness: 140 }, + }); + const dy = interpolate(charSpring, [0, 1], [38, 0]); + return ( + + {char} + + ); + })} +
+ + {/* Animated underline — draws in from the center outward */} +
+
+ + ); +}; + +// --------------------------------------------------------------------------- +// TitledVideo composition — plays a prerendered clip full-bleed and overlays +// the editorial tagline during a window of the timeline. +// --------------------------------------------------------------------------- +export const TitledVideo: React.FC = ({ + videoSrc, + tagline, + taglineInSeconds, + taglineOutSeconds, + topPx = 150, + fontSize = 148, + accentColor = "#F5C470", +}) => { + const { fps, durationInFrames } = useVideoConfig(); + + const inFrame = Math.max(0, Math.round(taglineInSeconds * fps)); + const endFrame = + taglineOutSeconds !== undefined + ? Math.round(taglineOutSeconds * fps) + : durationInFrames; + const overlayFrames = Math.max(1, endFrame - inFrame); + + return ( + + {/* Full-bleed background video — no fades, no vignette, no color shift. + The source is already color-graded final.mp4 with music baked in; + we play it through untouched, audio included. */} + + + {/* Tagline overlay in its own Sequence so it mounts exactly on the + fade-in frame and carries its own local frame counter. */} + + + + + ); +}; + +// --------------------------------------------------------------------------- +// calculateMetadata — probe the source clip so the composition runs at the +// clip's native duration. Falls back to 60s at 30fps if probing fails. +// --------------------------------------------------------------------------- +export const calculateTitledVideoMetadata: CalculateMetadataFunction< + TitledVideoProps +> = async ({ props }) => { + try { + const meta = await getVideoMetadata(resolveAsset(props.videoSrc)); + return { + durationInFrames: Math.max(1, Math.round(meta.durationInSeconds * 30)), + fps: 30, + width: 1920, + height: 1080, + }; + } catch { + return { + durationInFrames: 30 * 60, + fps: 30, + width: 1920, + height: 1080, + }; + } +}; diff --git a/remotion-composer/titled_video_props.json b/remotion-composer/titled_video_props.json new file mode 100644 index 0000000..85106da --- /dev/null +++ b/remotion-composer/titled_video_props.json @@ -0,0 +1,8 @@ +{ + "videoSrc": "walk-home/final.mp4", + "tagline": "home is a verb.", + "taglineInSeconds": 53.5, + "topPx": 150, + "fontSize": 148, + "accentColor": "#F5C470" +} diff --git a/schemas/artifacts/edit_decisions.schema.json b/schemas/artifacts/edit_decisions.schema.json index d287632..5e93b74 100644 --- a/schemas/artifacts/edit_decisions.schema.json +++ b/schemas/artifacts/edit_decisions.schema.json @@ -191,7 +191,7 @@ }, "renderer_family": { "type": "string", - "enum": ["explainer-data", "explainer-teacher", "cinematic-trailer", "product-reveal", "screen-demo", "presenter", "animation-first"], + "enum": ["explainer-data", "explainer-teacher", "cinematic-trailer", "documentary-montage", "product-reveal", "screen-demo", "presenter", "animation-first"], "description": "Locked at proposal stage — determines which Remotion composition renders this video" }, "slideshow_risk_score": { diff --git a/schemas/artifacts/proposal_packet.schema.json b/schemas/artifacts/proposal_packet.schema.json index 8985084..90a70e3 100644 --- a/schemas/artifacts/proposal_packet.schema.json +++ b/schemas/artifacts/proposal_packet.schema.json @@ -147,7 +147,7 @@ }, "renderer_family": { "type": "string", - "enum": ["explainer-data", "explainer-teacher", "cinematic-trailer", "product-reveal", "screen-demo", "presenter", "animation-first"], + "enum": ["explainer-data", "explainer-teacher", "cinematic-trailer", "documentary-montage", "product-reveal", "screen-demo", "presenter", "animation-first"], "description": "Locked at proposal stage — compose cannot change without logging a decision" }, "music_source": { diff --git a/schemas/artifacts/render_report.schema.json b/schemas/artifacts/render_report.schema.json index 33d12dd..e3602cc 100644 --- a/schemas/artifacts/render_report.schema.json +++ b/schemas/artifacts/render_report.schema.json @@ -39,7 +39,7 @@ "render_grammar": { "type": "string", "description": "Which renderer family was used for this render", - "enum": ["explainer-data", "explainer-teacher", "cinematic-trailer", "product-reveal", "screen-demo", "presenter", "animation-first"] + "enum": ["explainer-data", "explainer-teacher", "cinematic-trailer", "documentary-montage", "product-reveal", "screen-demo", "presenter", "animation-first"] }, "slideshow_risk_score": { "type": "object", diff --git a/skills/pipelines/documentary-montage/asset-director.md b/skills/pipelines/documentary-montage/asset-director.md index 7c3fe59..5e0c9dc 100644 --- a/skills/pipelines/documentary-montage/asset-director.md +++ b/skills/pipelines/documentary-montage/asset-director.md @@ -6,7 +6,7 @@ The shot list exists. You now have to actually go out and find the clips that fill each slot. This is a two-step operation: 1. **Build the corpus** — fan the scene director's queries out across - Pexels / Archive.org / NASA and download/embed the candidates. + Pexels / Archive.org / NASA / Wikimedia / Unsplash and download/embed the candidates. 2. **Pick per slot** — run CLIP retrieval against the corpus with each slot description and choose one winner per slot. @@ -71,7 +71,7 @@ corpus_builder.execute({ {"query": "taxi heavy rain yellow", "kind": "video", "per_source": 6}, # ... one entry per unique slot query ], - "sources": ["pexels", "archive_org"], # from preferred_sources union + "sources": ["pexels", "archive_org", "wikimedia"], # from preferred_sources union "filters": { "min_duration": 3, "max_duration": 40, @@ -85,6 +85,9 @@ corpus_builder.execute({ **Rules for the fan-out:** +- If the brief pins a source and `corpus_builder.source_provider_menu` + says that source is unavailable, STOP and surface it. Do not silently + drop to the remaining sources. - Budget the corpus for 8-12x the slot count. A 15-slot montage wants ~150 candidates so retrieval has real choices. - `per_source` of 4-8 per query is usually enough. Pushing to 20+ @@ -94,6 +97,8 @@ corpus_builder.execute({ is slow — don't interleave it with the modern Pexels batch. - If any slot has `nasa` in `preferred_sources`, run ONE small `nasa`-only batch. NASA is slow and its results are niche. +- `unsplash` is image-only. Use it as a support source, not the + backbone of a motion-led documentary cut. ### 3. Sanity-Check The Corpus Before Retrieval diff --git a/skills/pipelines/documentary-montage/compose-director.md b/skills/pipelines/documentary-montage/compose-director.md index 1e20b13..fbe9af2 100644 --- a/skills/pipelines/documentary-montage/compose-director.md +++ b/skills/pipelines/documentary-montage/compose-director.md @@ -16,7 +16,7 @@ The output is a single mp4 plus a `render_report` artifact. | Schema | `schemas/artifacts/render_report.schema.json` | Artifact validation | | Prior artifact | `state.artifacts["edit"]["edit_decisions"]` | Cuts, transitions, music, metadata hints | | Prior artifact | `state.artifacts["assets"]["asset_manifest"]` | File paths, durations, providers | -| Tool | `video_compose` (FFmpeg + Remotion) | Primary render engine | +| Tool | `video_compose` (Remotion-first + FFmpeg fallback) | Primary render engine | | Tool | `audio_mixer` | Music fade, silence window, L-cuts | | Tool (optional) | `color_grade` | Uniform LUT across mixed-era clips | | Tool (optional) | `video_trimmer`, `video_stitch` | Lower-level helpers if needed | @@ -47,11 +47,16 @@ If the brief said "no narration" and a narration track somehow appeared in the edit, STOP and ask. Do not render over a contract violation. -Also confirm the render engine you intend to use is actually -available — `video_compose` in FFmpeg-only mode is fine for this -pipeline (the whole piece is footage-led, no Remotion scenes needed -unless the user asked for title cards). FFmpeg alone can render this -pipeline end to end. +Also confirm that `edit_decisions.renderer_family` is locked to +`documentary-montage` and that the chosen render engine preserves that +decision. For this repo's governance model, `video_compose` is +Remotion-first on `operation="render"`, even for footage-led pieces. + +- If Remotion is available, use the normal `render` path and keep the + approved renderer family. +- If Remotion is unavailable, do NOT quietly drop to FFmpeg. Surface + the engine change and get approval before using a lower-level + FFmpeg-only path. ### 1. Resolve The Canvas @@ -83,32 +88,8 @@ For a pipeline this simple, the cleanest path is: video_compose.execute({ "operation": "render", "output_path": "projects//renders/final.mp4", - "canvas": { "width": 1920, "height": 1080, "fps": 24 }, - "cuts": [ - { - "source": "", - "in": 1.2, - "out": 5.2, - "scale": "fit_canvas_center_crop", - "transition_in": "fade", - "transition_in_duration": 0.8, - }, - # ... more cuts - ], - "audio": { - "music_path": "", - "music_volume": 0.7, - "music_fade_in": 1.0, - "music_fade_out": 4.0, - "silence_windows": [{"start": 54.0, "end": 56.0}], - "sfx_layers": [ - {"source": "", "start": 20.8, "duration": 1.2, "volume": 0.6} - ], - }, - "frame_treatment": { - "lut_path": "styles/luts/warm_film_100.cube", - "letterbox": "2.35:1" - } + "edit_decisions": edit_decisions_with_renderer_family, + "asset_manifest": asset_manifest, }) ``` @@ -116,6 +97,9 @@ The exact field names come from the live `video_compose` schema at render time — consult the tool's `agent_skills` if available before writing the call. Do not invent parameters. +`edit_decisions_with_renderer_family` means the normal edit artifact +with `renderer_family = "documentary-montage"` preserved intact. + ### 3. Apply Grade Via LUT, Not Per Clip Read `edit_decisions.metadata.grade_profile`. Map it to a LUT file: @@ -163,8 +147,8 @@ Recommended encoder settings for doc montage: | Audio codec | `aac` | Universal | | Audio bitrate | `192k` | Music-bed friendly | -If the source clips are 30fps and the canvas is 24fps, let FFmpeg -drop frames evenly — don't blend. Motion interpolation on +If the source clips are 30fps and the canvas is 24fps, let the render +pipeline drop frames evenly — don't blend. Motion interpolation on mixed-source footage looks awful. ### 6. Post-Render Verification @@ -210,7 +194,7 @@ Record verifications in `render_report.verification_notes`. "Silence window 54-56s confirmed (music -60dB)", "Last frame fades to black at 89.0s" ], - "render_grammar": "cinematic-trailer", + "render_grammar": "documentary-montage", "metadata": { "pipeline": "documentary-montage", "canvas": { "width": 1920, "height": 1080 }, @@ -247,10 +231,9 @@ Record verifications in `render_report.verification_notes`. - **Per-clip color grading.** One LUT across the whole piece. Do not try to balance each clip individually — it takes 10x the time and makes the register LESS consistent, not more. -- **Quiet render engine swap.** If `video_compose` routes through - Remotion for some reason and the aesthetic changes, stop and - surface. This pipeline is FFmpeg-friendly and shouldn't need - Remotion unless the user asked for title cards. +- **Quiet FFmpeg downgrade.** If Remotion is blocked and you route to + FFmpeg without surfacing it, you've changed the approved render path. + Stop and surface that downgrade before rendering. - **Overriding edit decisions at render time.** If you find yourself adjusting volumes, fades, or trims in the render call, you're editing during compose. Go back to the edit stage, fix the diff --git a/skills/pipelines/documentary-montage/edit-director.md b/skills/pipelines/documentary-montage/edit-director.md index a6fbc96..f9f68f7 100644 --- a/skills/pipelines/documentary-montage/edit-director.md +++ b/skills/pipelines/documentary-montage/edit-director.md @@ -34,6 +34,10 @@ The enemy is "slideshow" — a sequence of clips played back-to-back with the same hold length and no sound design. If it feels like a slideshow, the edit has failed, regardless of how good the clips are. +This stage also locks the render grammar. For documentary montage, +set `renderer_family` to `documentary-montage` so compose stays on the +approved Remotion-first path. + ## Process ### 0. Guardrails — No Silent Major Changes @@ -228,6 +232,7 @@ Canonical shape for this pipeline: ```json { "version": "1.0", + "renderer_family": "documentary-montage", "cuts": [ { "id": "cut_01", @@ -284,6 +289,7 @@ Canonical shape for this pipeline: - `sum(out - in for cut in cuts)` is within ±10% of `brief.duration_seconds`. +- `renderer_family = "documentary-montage"` is present and unchanged. - Hero slots have the longest holds. - No two adjacent cuts share subject AND scale. - The transition vocabulary is at most 4 distinct values. diff --git a/skills/pipelines/documentary-montage/executive-producer.md b/skills/pipelines/documentary-montage/executive-producer.md index 86a70ab..52dd46b 100644 --- a/skills/pipelines/documentary-montage/executive-producer.md +++ b/skills/pipelines/documentary-montage/executive-producer.md @@ -55,7 +55,7 @@ before starting the stage. | Tool | Role | |------|------| -| `corpus_builder` | Fans out across Pexels/Archive.org/NASA, downloads + embeds + indexes | +| `corpus_builder` | Fans out across Pexels/Archive.org/NASA/Wikimedia/Unsplash, downloads + embeds + indexes | | `clip_search` | Ranks clips for a slot, finds similar sets, diversifies selections | | `video_compose` / Remotion | Renders the final timeline | diff --git a/skills/pipelines/documentary-montage/idea-director.md b/skills/pipelines/documentary-montage/idea-director.md index d340750..47a7031 100644 --- a/skills/pipelines/documentary-montage/idea-director.md +++ b/skills/pipelines/documentary-montage/idea-director.md @@ -113,8 +113,9 @@ open for the scene director to decide per slot. - Tone is ONE register from the fixed list. - Duration and shape are concrete numbers / enum values. - Music source is named OR the brief explicitly says "no music". -- Sources list is non-empty and at least one is `available` per the - tool registry. +- Sources list is non-empty and at least one requested source is + `available` per `corpus_builder.source_provider_menu` surfaced in + preflight. ## Common Pitfalls diff --git a/skills/pipelines/documentary-montage/scene-director.md b/skills/pipelines/documentary-montage/scene-director.md index 50c0730..16520a9 100644 --- a/skills/pipelines/documentary-montage/scene-director.md +++ b/skills/pipelines/documentary-montage/scene-director.md @@ -5,7 +5,7 @@ The brief exists. You now have to turn a thematic question into a concrete list of SLOTS the retrieval layer can fill. Each slot is an intention ("a silhouette at a doorway at dusk") plus the queries that -will find it in the real world (Pexels/Archive.org/NASA). +will find it in the real world (Pexels/Archive.org/NASA/Wikimedia/Unsplash). This is the most creative stage in the pipeline. Retrieval is only as good as the slot descriptions you write. @@ -147,6 +147,8 @@ based on what footage lives where: | `pexels` | Modern HD footage, clean shots, people, cities, nature | Default for modern/any era | | `archive_org` | Prelinger home movies, mid-century educational film, 1940s-1980s texture | Vintage, wry, dreamlike, anything nostalgic | | `nasa` | Earth-from-orbit, astronomy, flight, scale imagery | Reverent, anything about scale, space, planet, flight | +| `wikimedia` | Commons photos and CC video, civic/documentary/public-event coverage | Public spaces, landmarks, protests, city texture, educational footage | +| `unsplash` | Polished editorial stills, lifestyle, product-adjacent photography | Modern still-image support shots when motion footage is thin | If `era_mix = "vintage"`, bias slots toward `archive_org` and write queries in period-appropriate vocabulary ("commuter", "housewife", diff --git a/tests/tools/test_documentary_governance.py b/tests/tools/test_documentary_governance.py new file mode 100644 index 0000000..9f91d83 --- /dev/null +++ b/tests/tools/test_documentary_governance.py @@ -0,0 +1,139 @@ +from __future__ import annotations + +from pathlib import Path + +from tools.base_tool import ToolStatus +from tools.tool_registry import ToolRegistry +from tools.video.corpus_builder import CorpusBuilder +from tools.video.video_compose import VideoCompose + + +class _DummySource: + def __init__(self, name: str, available: bool) -> None: + self.name = name + self._available = available + + def is_available(self) -> bool: + return self._available + + def search(self, query: str, filters): # pragma: no cover - protocol stub + return [] + + def download(self, candidate, out_path: Path): # pragma: no cover - protocol stub + return out_path + + +def test_corpus_builder_reports_source_level_discoverability(monkeypatch): + import tools.video.stock_sources as stock_sources + + monkeypatch.setattr( + stock_sources, + "all_sources", + lambda: [_DummySource("pexels", False), _DummySource("archive_org", True)], + ) + monkeypatch.setattr( + stock_sources, + "available_sources", + lambda: [_DummySource("archive_org", True)], + ) + monkeypatch.setattr( + stock_sources, + "source_catalog", + lambda: [ + {"name": "pexels", "status": "unavailable"}, + {"name": "archive_org", "status": "available"}, + ], + ) + monkeypatch.setattr( + stock_sources, + "source_summary", + lambda: { + "configured": 1, + "total": 2, + "available_source_names": ["archive_org"], + "unavailable_source_names": ["pexels"], + }, + ) + + tool = CorpusBuilder() + assert tool.get_status() == ToolStatus.DEGRADED + + info = tool.get_info() + assert info["source_provider_summary"]["configured"] == 1 + assert info["source_provider_summary"]["total"] == 2 + assert {entry["name"] for entry in info["source_provider_menu"]} == { + "pexels", + "archive_org", + } + + +def test_corpus_builder_rejects_unavailable_pinned_sources(monkeypatch, tmp_path): + import tools.video.stock_sources as stock_sources + + sources = { + "pexels": _DummySource("pexels", False), + "archive_org": _DummySource("archive_org", True), + } + + monkeypatch.setattr(stock_sources, "all_sources", lambda: list(sources.values())) + monkeypatch.setattr( + stock_sources, + "available_sources", + lambda: [sources["archive_org"]], + ) + monkeypatch.setattr(stock_sources, "get_source", lambda name: sources[name]) + monkeypatch.setattr( + stock_sources, + "source_summary", + lambda: { + "configured": 1, + "total": 2, + "available_source_names": ["archive_org"], + "unavailable_source_names": ["pexels"], + }, + ) + + result = CorpusBuilder().execute({ + "corpus_dir": str(tmp_path / "corpus"), + "queries": [{"query": "rain at night"}], + "sources": ["pexels"], + }) + + assert not result.success + assert "Requested stock sources are unavailable" in result.error + assert "archive_org" in result.error + + +def test_documentary_renderer_family_maps_to_remotion(): + assert VideoCompose._get_composition_id("documentary-montage") == "CinematicRenderer" + + +def test_provider_menu_preserves_tool_discovery_metadata(monkeypatch): + import tools.video.stock_sources as stock_sources + + monkeypatch.setattr(stock_sources, "all_sources", lambda: [_DummySource("archive_org", True)]) + monkeypatch.setattr(stock_sources, "available_sources", lambda: [_DummySource("archive_org", True)]) + monkeypatch.setattr( + stock_sources, + "source_catalog", + lambda: [{"name": "archive_org", "status": "available"}], + ) + monkeypatch.setattr( + stock_sources, + "source_summary", + lambda: { + "configured": 1, + "total": 1, + "available_source_names": ["archive_org"], + "unavailable_source_names": [], + }, + ) + + registry = ToolRegistry() + registry.register(CorpusBuilder()) + menu = registry.provider_menu() + entry = menu["corpus_population"]["available"][0] + + assert entry["name"] == "corpus_builder" + assert entry["source_provider_summary"]["configured"] == 1 + assert entry["source_provider_menu"][0]["name"] == "archive_org" diff --git a/tests/tools/test_stock_source_adapters.py b/tests/tools/test_stock_source_adapters.py new file mode 100644 index 0000000..dbffb04 --- /dev/null +++ b/tests/tools/test_stock_source_adapters.py @@ -0,0 +1,29 @@ +from tools.video.stock_sources import all_sources +from tools.video.stock_sources.unsplash import _build_download_url, _orientation_for_unsplash +from tools.video.stock_sources.wikimedia import _build_search_query, _kind_from_mime, _meta_value + + +def test_stock_source_autodiscovery_includes_new_sources(): + names = {source.name for source in all_sources()} + assert "wikimedia" in names + assert "unsplash" in names + + +def test_wikimedia_search_query_respects_kind(): + assert _build_search_query("rain city", "video").startswith("filetype:video") + assert _build_search_query("rain city", "image").startswith("filetype:image") + assert _build_search_query("rain city", "any") == "rain city" + + +def test_wikimedia_kind_and_metadata_helpers(): + assert _kind_from_mime("video/webm", "File:foo.webm") == "video" + assert _kind_from_mime("image/jpeg", "File:foo.jpg") == "image" + assert _meta_value({"Artist": {"value": "Test User"}}, "Artist") == "Test User" + + +def test_unsplash_helpers_preserve_query_params(): + assert _orientation_for_unsplash("square") == "squarish" + url = _build_download_url("https://images.unsplash.com/photo-123?ixid=abc", 1920) + assert "ixid=abc" in url + assert "w=1920" in url + assert "fm=jpg" in url diff --git a/tools/tool_registry.py b/tools/tool_registry.py index 7882b07..9202901 100644 --- a/tools/tool_registry.py +++ b/tools/tool_registry.py @@ -225,6 +225,7 @@ class ToolRegistry: if cap not in menu: menu[cap] = {"available": [], "unavailable": [], "total": 0, "configured": 0} + info = tool.get_info() status = tool.get_status() entry = { "name": tool.name, @@ -234,6 +235,15 @@ class ToolRegistry: "install_instructions": tool.install_instructions, "status": status.value, } + for extra_key in ( + "source_provider_menu", + "source_provider_summary", + "render_engines", + "remotion_note", + "provider_matrix", + ): + if extra_key in info: + entry[extra_key] = info[extra_key] if status == ToolStatus.AVAILABLE: menu[cap]["available"].append(entry) diff --git a/tools/video/corpus_builder.py b/tools/video/corpus_builder.py index 28f7717..4bf6b90 100644 --- a/tools/video/corpus_builder.py +++ b/tools/video/corpus_builder.py @@ -94,7 +94,8 @@ class CorpusBuilder(BaseTool): "pip install opencv-python numpy requests pillow transformers torch\n" "At least one stock source must be configured:\n" " PEXELS_API_KEY for Pexels (free at https://www.pexels.com/api/)\n" - " archive.org and nasa work without API keys" + " UNSPLASH_ACCESS_KEY for Unsplash (see https://unsplash.com/documentation)\n" + " archive.org, nasa, and wikimedia work without API keys" ) agent_skills = [] @@ -201,10 +202,32 @@ class CorpusBuilder(BaseTool): def get_status(self) -> ToolStatus: try: - from tools.video.stock_sources import available_sources + from tools.video.stock_sources import all_sources, available_sources except Exception: return ToolStatus.UNAVAILABLE - return ToolStatus.AVAILABLE if available_sources() else ToolStatus.UNAVAILABLE + total = len(all_sources()) + available = len(available_sources()) + if available == 0: + return ToolStatus.UNAVAILABLE + if available < total: + return ToolStatus.DEGRADED + return ToolStatus.AVAILABLE + + def get_info(self) -> dict[str, Any]: + info = super().get_info() + try: + from tools.video.stock_sources import source_catalog, source_summary + info["source_provider_menu"] = source_catalog() + info["source_provider_summary"] = source_summary() + except Exception: + info["source_provider_menu"] = [] + info["source_provider_summary"] = { + "configured": 0, + "total": 0, + "available_source_names": [], + "unavailable_source_names": [], + } + return info def estimate_cost(self, inputs: dict[str, Any]) -> float: return 0.0 # all sources are free-tier @@ -219,8 +242,10 @@ class CorpusBuilder(BaseTool): from lib.corpus import Corpus from tools.video.stock_sources import ( SearchFilters, + all_sources, available_sources, get_source, + source_summary, ) corpus_dir = Path(inputs["corpus_dir"]) @@ -232,17 +257,36 @@ class CorpusBuilder(BaseTool): thumbs_per_video = int(inputs.get("thumbs_per_video", 5)) # Resolve sources. If the caller passed an explicit list we - # respect it even if some are unavailable — their search - # calls will simply fail and be logged. + # must not silently degrade: pinned-but-unavailable sources + # are a provider substitution the agent needs to surface. if source_names: requested: list = [] + unavailable_requested: list[str] = [] + known_sources = {src.name: src for src in all_sources()} for name in source_names: - try: - s = get_source(name) - except KeyError as e: - return ToolResult(success=False, error=str(e)) + s = known_sources.get(name) + if s is None: + try: + s = get_source(name) + except KeyError as e: + return ToolResult(success=False, error=str(e)) if s.is_available(): requested.append(s) + else: + unavailable_requested.append(name) + if unavailable_requested: + summary = source_summary() + return ToolResult( + success=False, + error=( + "Requested stock sources are unavailable: " + f"{', '.join(unavailable_requested)}. " + "Available now: " + f"{', '.join(summary['available_source_names']) or 'none'}. " + "Check corpus_builder.source_provider_menu during preflight " + "before rerunning." + ), + ) sources = requested else: sources = available_sources() @@ -344,6 +388,9 @@ class CorpusBuilder(BaseTool): "per_source_counts": per_source_counts, "added_ids": added_ids, "total_corpus_size": len(corp), + "requested_sources": source_names or [], + "resolved_sources": [s.name for s in sources], + "source_provider_summary": source_summary(), "errors": errors[:25], # cap log noise }, cost_usd=0.0, diff --git a/tools/video/stock_sources/__init__.py b/tools/video/stock_sources/__init__.py index e4b3bf9..b61fd1c 100644 --- a/tools/video/stock_sources/__init__.py +++ b/tools/video/stock_sources/__init__.py @@ -5,44 +5,69 @@ NASA, ...) used by `corpus_builder` to populate the local clip corpus. See `base.py` for the protocol contract and the "adding a new source" checklist. -Registry --------- -`all_sources()` returns one instance of every adapter known to the -package, in a stable order. `available_sources()` filters to the ones -whose `is_available()` returns True right now. The corpus builder -queries the latter during fan-out; it never imports adapters directly. +Discovery +--------- +Adapters are auto-discovered from this package. `all_sources()` returns +one instance of every concrete adapter class found under +`tools.video.stock_sources`, ordered by each class's optional +`priority` and then by name. `available_sources()` filters to the ones +whose `is_available()` returns True right now. -To register a new adapter: add its class to the `_REGISTRY` tuple -below. Order matters only as a tiebreak when two sources return -matching clips. +This keeps source discoverability aligned with the rest of the repo: +adding a new adapter file is enough to make it visible to +`corpus_builder` and to preflight metadata. """ from __future__ import annotations -from .archive_org import ArchiveOrgSource +import importlib +import inspect +import pkgutil + from .base import Candidate, SearchFilters, StockSource -from .nasa import NasaSource -from .pexels import PexelsSource __all__ = [ "Candidate", "SearchFilters", "StockSource", - "PexelsSource", - "ArchiveOrgSource", - "NasaSource", "all_sources", "available_sources", "get_source", + "source_catalog", + "source_summary", ] -# Explicit, ordered list of every adapter class the package exposes. -# The corpus builder iterates this (filtered by availability) during -# fan-out. Order matters only as a tiebreak for identical clip ids. -_REGISTRY: tuple[type, ...] = ( - PexelsSource, - ArchiveOrgSource, - NasaSource, -) +def _is_source_adapter_class(cls: type) -> bool: + """Return True for concrete source adapters in this package.""" + return ( + inspect.isclass(cls) + and cls.__module__.startswith(f"{__name__}.") + and cls.__module__ != f"{__name__}.base" + and isinstance(getattr(cls, "name", None), str) + and bool(getattr(cls, "name", None)) + and callable(getattr(cls, "is_available", None)) + and callable(getattr(cls, "search", None)) + and callable(getattr(cls, "download", None)) + ) + + +def _source_classes() -> list[type]: + """Auto-discover stock source classes under this package.""" + discovered: dict[str, type] = {} + for module_info in pkgutil.iter_modules(__path__, f"{__name__}."): + if module_info.ispkg or module_info.name.endswith(".base"): + continue + module = importlib.import_module(module_info.name) + for _, cls in inspect.getmembers(module, inspect.isclass): + if not _is_source_adapter_class(cls): + continue + discovered[getattr(cls, "name")] = cls + return sorted( + discovered.values(), + key=lambda cls: ( + int(getattr(cls, "priority", 100)), + getattr(cls, "display_name", getattr(cls, "name")).lower(), + ), + ) def all_sources() -> list[StockSource]: @@ -53,7 +78,7 @@ def all_sources() -> list[StockSource]: want to show the user what sources exist regardless of whether their credentials are configured. """ - return [cls() for cls in _REGISTRY] + return [cls() for cls in _source_classes()] def available_sources() -> list[StockSource]: @@ -67,6 +92,40 @@ def available_sources() -> list[StockSource]: return [s for s in all_sources() if s.is_available()] +def source_catalog() -> list[dict[str, object]]: + """Return discoverability metadata for every stock source.""" + catalog: list[dict[str, object]] = [] + for source in all_sources(): + cls = source.__class__ + available = bool(source.is_available()) + catalog.append({ + "name": source.name, + "display_name": getattr(cls, "display_name", source.name), + "provider": getattr(cls, "provider", source.name), + "status": "available" if available else "unavailable", + "install_instructions": getattr( + cls, + "install_instructions", + "See the source adapter docs for setup details.", + ), + "supports": getattr(cls, "supports", {}), + }) + return catalog + + +def source_summary() -> dict[str, object]: + """Summarize source availability for preflight and tool contracts.""" + catalog = source_catalog() + available = [entry["name"] for entry in catalog if entry["status"] == "available"] + unavailable = [entry["name"] for entry in catalog if entry["status"] != "available"] + return { + "configured": len(available), + "total": len(catalog), + "available_source_names": available, + "unavailable_source_names": unavailable, + } + + def get_source(name: str) -> StockSource: """Look up a single adapter by its `name` attribute. diff --git a/tools/video/stock_sources/archive_org.py b/tools/video/stock_sources/archive_org.py index 3ca9ca3..a009176 100644 --- a/tools/video/stock_sources/archive_org.py +++ b/tools/video/stock_sources/archive_org.py @@ -70,6 +70,13 @@ class ArchiveOrgSource: """ name = "archive_org" + display_name = "Archive.org" + provider = "archive_org" + priority = 20 + install_instructions = ( + "No setup required. Archive.org is available without API keys." + ) + supports = {"video": True, "image": False} def is_available(self) -> bool: # No API key, no config. As long as the network is up, we're diff --git a/tools/video/stock_sources/base.py b/tools/video/stock_sources/base.py index 1589105..f1632f6 100644 --- a/tools/video/stock_sources/base.py +++ b/tools/video/stock_sources/base.py @@ -29,8 +29,9 @@ Adding a new source field the corpus might later want to display or attribute (creator, licence, source URL, description/tags for the text channel of the CLIP fused ranking). -3. Register the class so `corpus_builder` can discover it. The - registration pattern lives alongside the first concrete adapter. +3. Give the class a stable `name` attribute and optional discoverability + metadata such as `display_name`, `install_instructions`, `supports`, + and `priority`. The package auto-discovers concrete adapters. """ from __future__ import annotations diff --git a/tools/video/stock_sources/nasa.py b/tools/video/stock_sources/nasa.py index 282d91b..cf6c22d 100644 --- a/tools/video/stock_sources/nasa.py +++ b/tools/video/stock_sources/nasa.py @@ -55,6 +55,14 @@ class NasaSource: """ name = "nasa" + display_name = "NASA" + provider = "nasa" + priority = 30 + install_instructions = ( + "No setup required. NASA media search works without an API key; " + "NASA_API_KEY is optional for higher rate limits." + ) + supports = {"video": True, "image": True} def is_available(self) -> bool: # The public API is unauthenticated; as long as the network is diff --git a/tools/video/stock_sources/pexels.py b/tools/video/stock_sources/pexels.py index 7386a15..7d9f2f7 100644 --- a/tools/video/stock_sources/pexels.py +++ b/tools/video/stock_sources/pexels.py @@ -36,6 +36,14 @@ class PexelsSource: """ name = "pexels" + display_name = "Pexels" + provider = "pexels" + priority = 10 + install_instructions = ( + "Set PEXELS_API_KEY in .env to enable Pexels stock search " + "(free key at https://www.pexels.com/api/)." + ) + supports = {"video": True, "image": True} def is_available(self) -> bool: return bool(os.environ.get("PEXELS_API_KEY")) diff --git a/tools/video/stock_sources/unsplash.py b/tools/video/stock_sources/unsplash.py new file mode 100644 index 0000000..17b92c2 --- /dev/null +++ b/tools/video/stock_sources/unsplash.py @@ -0,0 +1,184 @@ +"""Unsplash stock photo adapter. + +Unsplash is image-only in this pipeline. It widens the corpus for +modern, polished, lifestyle, and product-adjacent scenes where a +high-quality still can still be valuable to the edit. +""" +from __future__ import annotations + +import os +from pathlib import Path +from typing import Any +from urllib.parse import parse_qsl, urlencode, urlparse, urlunparse + +from .base import Candidate, SearchFilters + + +_SEARCH_URL = "https://api.unsplash.com/search/photos" +_UNSPLASH_LICENSE = "Unsplash License (use returned hotlinked image URLs)" +_USER_AGENT = "OpenMontageBot/0.1 (https://github.com/calesthio/OpenMontage)" + + +class UnsplashSource: + """Adapter for Unsplash photo search.""" + + name = "unsplash" + display_name = "Unsplash" + provider = "unsplash" + priority = 18 + install_instructions = ( + "Set UNSPLASH_ACCESS_KEY in .env to enable Unsplash image search " + "(see https://unsplash.com/documentation)." + ) + supports = {"video": False, "image": True} + + def is_available(self) -> bool: + return bool(os.environ.get("UNSPLASH_ACCESS_KEY")) + + def search(self, query: str, filters: SearchFilters) -> list[Candidate]: + import requests # lazy + + kind = (filters.kind or "video").lower() + if kind == "video": + return [] + + params: dict[str, Any] = { + "query": query, + "page": max(1, filters.page), + "per_page": max(1, min(filters.per_page, 30)), + "content_filter": "high", + } + orientation = _orientation_for_unsplash(filters.orientation) + if orientation: + params["orientation"] = orientation + + r = requests.get( + _SEARCH_URL, + params=params, + headers=self._headers(), + timeout=30, + ) + r.raise_for_status() + data = r.json() + results = data.get("results") or [] + + out: list[Candidate] = [] + for photo in results: + cand = _photo_to_candidate(photo, filters) + if cand is not None: + out.append(cand) + return out + + def download(self, candidate: Candidate, out_path: Path) -> Path: + import requests # lazy + + if not candidate.download_url: + raise ValueError(f"Candidate {candidate.clip_id} has no download_url") + + out_path = Path(out_path) + out_path.parent.mkdir(parents=True, exist_ok=True) + + with requests.get( + candidate.download_url, + stream=True, + timeout=180, + headers={"User-Agent": _USER_AGENT}, + ) as r: + r.raise_for_status() + with open(out_path, "wb") as f: + for chunk in r.iter_content(chunk_size=1 << 16): + if chunk: + f.write(chunk) + return out_path + + def _headers(self) -> dict[str, str]: + key = os.environ.get("UNSPLASH_ACCESS_KEY") + if not key: + raise RuntimeError( + "UNSPLASH_ACCESS_KEY not set. Create an app at " + "https://unsplash.com/documentation and add the access key to .env." + ) + return { + "Authorization": f"Client-ID {key}", + "Accept-Version": "v1", + "User-Agent": _USER_AGENT, + } + + +def _photo_to_candidate(photo: dict[str, Any], filters: SearchFilters) -> Candidate | None: + width = int(photo.get("width") or 0) + height = int(photo.get("height") or 0) + if filters.min_width is not None and width and width < filters.min_width: + return None + if filters.orientation and not _matches_orientation(filters.orientation, width, height): + return None + + user = photo.get("user") or {} + links = photo.get("links") or {} + urls = photo.get("urls") or {} + raw_url = urls.get("raw") or urls.get("regular") or "" + if not raw_url: + return None + + description_parts = [ + photo.get("description") or "", + photo.get("alt_description") or "", + photo.get("slug") or "", + ] + source_tags = " ".join(part.strip() for part in description_parts if part).strip() + if len(source_tags) > 500: + source_tags = source_tags[:500] + + return Candidate( + source=UnsplashSource.name, + source_id=str(photo.get("id") or ""), + source_url=links.get("html", "") or "", + download_url=_build_download_url(raw_url, target_width=max(filters.min_width or 0, 1920)), + kind="image", + width=width, + height=height, + duration=0.0, + creator=user.get("name", "") or "", + license=_UNSPLASH_LICENSE, + source_tags=source_tags, + thumbnail_url=urls.get("small", "") or urls.get("thumb", "") or raw_url, + extra={ + "color": photo.get("color"), + "blur_hash": photo.get("blur_hash"), + "download_location": links.get("download_location"), + "photographer_url": user.get("links", {}).get("html"), + }, + ) + + +def _orientation_for_unsplash(orientation: str | None) -> str | None: + if orientation == "landscape": + return "landscape" + if orientation == "portrait": + return "portrait" + if orientation == "square": + return "squarish" + return None + + +def _matches_orientation(orientation: str, width: int, height: int) -> bool: + if not width or not height: + return True + if orientation == "landscape": + return width >= height + if orientation == "portrait": + return height > width + if orientation == "square": + return width == height + return True + + +def _build_download_url(raw_url: str, target_width: int) -> str: + parts = urlparse(raw_url) + params = dict(parse_qsl(parts.query, keep_blank_values=True)) + params.setdefault("fm", "jpg") + params.setdefault("q", "80") + if target_width > 0: + params["w"] = str(target_width) + params.setdefault("fit", "max") + return urlunparse(parts._replace(query=urlencode(params))) diff --git a/tools/video/stock_sources/wikimedia.py b/tools/video/stock_sources/wikimedia.py new file mode 100644 index 0000000..9ac881d --- /dev/null +++ b/tools/video/stock_sources/wikimedia.py @@ -0,0 +1,196 @@ +"""Wikimedia Commons stock media adapter. + +Provides image and video search over Wikimedia Commons using the +MediaWiki API. Commons is a uniquely useful documentary source because +it mixes public-domain historical imagery, recent CC-licensed videos, +and educational media under one searchable catalogue. +""" +from __future__ import annotations + +import html +import re +from pathlib import Path +from typing import Any + +from .base import Candidate, SearchFilters + + +_API_URL = "https://commons.wikimedia.org/w/api.php" +_USER_AGENT = "OpenMontageBot/0.1 (https://github.com/calesthio/OpenMontage)" +_COMMONS_LICENSE = "Wikimedia Commons (verify per-file license)" +_HTML_TAG_RE = re.compile(r"<[^>]+>") + + +class WikimediaSource: + """Adapter for Wikimedia Commons media search.""" + + name = "wikimedia" + display_name = "Wikimedia Commons" + provider = "wikimedia" + priority = 25 + install_instructions = ( + "No setup required. Wikimedia Commons media search works without API keys." + ) + supports = {"video": True, "image": True} + + def is_available(self) -> bool: + return True + + def search(self, query: str, filters: SearchFilters) -> list[Candidate]: + import requests # lazy + + params = { + "action": "query", + "format": "json", + "generator": "search", + "gsrsearch": _build_search_query(query, filters.kind), + "gsrnamespace": 6, + "gsrlimit": max(1, min(filters.per_page, 50)), + "gsroffset": max(0, (max(filters.page, 1) - 1) * max(1, min(filters.per_page, 50))), + "prop": "imageinfo|info", + "iiprop": "url|size|mime|extmetadata|mediatype", + "iiurlwidth": 640, + "inprop": "url", + } + + r = requests.get( + _API_URL, + params=params, + headers={"User-Agent": _USER_AGENT}, + timeout=30, + ) + r.raise_for_status() + data = r.json() + pages = list(((data.get("query") or {}).get("pages") or {}).values()) + pages.sort(key=lambda page: int(page.get("index", 0))) + + out: list[Candidate] = [] + for page in pages: + cand = _page_to_candidate(page, filters) + if cand is not None: + out.append(cand) + return out + + def download(self, candidate: Candidate, out_path: Path) -> Path: + import requests # lazy + + if not candidate.download_url: + raise ValueError(f"Candidate {candidate.clip_id} has no download_url") + + out_path = Path(out_path) + out_path.parent.mkdir(parents=True, exist_ok=True) + + with requests.get( + candidate.download_url, + stream=True, + timeout=300, + headers={"User-Agent": _USER_AGENT}, + ) as r: + r.raise_for_status() + with open(out_path, "wb") as f: + for chunk in r.iter_content(chunk_size=1 << 16): + if chunk: + f.write(chunk) + return out_path + + +def _build_search_query(query: str, kind: str) -> str: + user_query = query.strip() + kind = (kind or "video").lower() + if kind == "video": + return f"filetype:video {user_query}".strip() + if kind == "image": + return f"filetype:image {user_query}".strip() + return user_query + + +def _page_to_candidate(page: dict[str, Any], filters: SearchFilters) -> Candidate | None: + infos = page.get("imageinfo") or [] + if not infos: + return None + info = infos[0] + mime = (info.get("mime") or "").lower() + kind = _kind_from_mime(mime, page.get("title", "")) + + requested_kind = (filters.kind or "video").lower() + if requested_kind == "video" and kind != "video": + return None + if requested_kind == "image" and kind != "image": + return None + + width = int(info.get("width") or 0) + height = int(info.get("height") or 0) + duration = float(info.get("duration") or 0.0) + + if filters.min_width is not None and width and width < filters.min_width: + return None + if filters.min_duration is not None and duration and duration < filters.min_duration: + return None + if filters.max_duration is not None and duration and duration > filters.max_duration: + return None + if filters.orientation and not _matches_orientation(filters.orientation, width, height): + return None + + meta = info.get("extmetadata") or {} + object_name = _meta_value(meta, "ObjectName") + description = _meta_value(meta, "ImageDescription") + categories = _meta_value(meta, "Categories") + creator = _meta_value(meta, "Artist") + license_name = _meta_value(meta, "LicenseShortName") + usage_terms = _meta_value(meta, "UsageTerms") + source_tags = " ".join(part for part in (object_name, description, categories) if part).strip() + if len(source_tags) > 500: + source_tags = source_tags[:500] + + title = page.get("title", "") + page_id = str(page.get("pageid") or title.replace("File:", "", 1)) + source_url = info.get("descriptionurl") or page.get("canonicalurl") or "" + + return Candidate( + source=WikimediaSource.name, + source_id=page_id, + source_url=source_url, + download_url=info.get("url", "") or "", + kind=kind, + width=width, + height=height, + duration=duration, + creator=creator, + license=license_name or usage_terms or _COMMONS_LICENSE, + source_tags=source_tags, + thumbnail_url=info.get("thumburl", "") or info.get("url", "") or "", + extra={ + "mime": mime, + "title": title, + "mediatype": info.get("mediatype"), + "descriptionshorturl": info.get("descriptionshorturl"), + }, + ) + + +def _kind_from_mime(mime: str, title: str) -> str: + if mime.startswith("video/") or title.lower().endswith((".webm", ".ogv", ".ogg")): + return "video" + return "image" + + +def _matches_orientation(orientation: str, width: int, height: int) -> bool: + if not width or not height: + return True + if orientation == "landscape": + return width >= height + if orientation == "portrait": + return height > width + if orientation == "square": + return width == height + return True + + +def _meta_value(meta: dict[str, Any], key: str) -> str: + raw = ((meta.get(key) or {}).get("value")) or "" + if not raw: + return "" + text = html.unescape(str(raw)) + text = _HTML_TAG_RE.sub(" ", text) + text = re.sub(r"\s+", " ", text).strip() + return text diff --git a/tools/video/video_compose.py b/tools/video/video_compose.py index 7aa9100..d98b3b6 100644 --- a/tools/video/video_compose.py +++ b/tools/video/video_compose.py @@ -554,6 +554,7 @@ class VideoCompose(BaseTool): "explainer-data": "Explainer", "explainer-teacher": "Explainer", "cinematic-trailer": "CinematicRenderer", + "documentary-montage": "CinematicRenderer", "product-reveal": "Explainer", "screen-demo": "Explainer", "presenter": "TalkingHead", @@ -691,12 +692,11 @@ class VideoCompose(BaseTool): component types, transitions, and mixed content — all in a single React-based render pass. - Returns False (i.e. use FFmpeg) ONLY when ALL of these are true: - 1. Every cut source is a video file (no images, no component types) - 2. No cut requests animations or transitions - 3. No cut uses a Remotion scene type - 4. The edit_decisions explicitly set renderer_family to "ffmpeg-only" - OR Remotion is not available + Returns False (i.e. use FFmpeg) only when Remotion is not + available. For `operation="render"` the governance default is + Remotion-first: the renderer family was chosen earlier, and the + tool should preserve that decision instead of silently + downgrading to FFmpeg. This "Remotion-first" policy means mixed content (video clips + animated stills + text cards) is always composed in Remotion, which