From 0999eaddc791f1d876bb5f0a648fa454c84d5fae Mon Sep 17 00:00:00 2001 From: calesthio Date: Sun, 12 Apr 2026 14:23:14 -0700 Subject: [PATCH] docmontage: add direct_clip_search tool for fast provider-agnostic clip acquisition Adds a lightweight alternative to the corpus_builder + clip_search pipeline that skips CLIP embeddings, motion scores, and index files. Uses the same StockSource adapter protocol so it works with all providers (Pexels, Archive.org, NASA, Wikimedia, Unsplash). Asset-director skill now documents both fast path (direct search) and standard path (corpus + CLIP retrieval). Pipeline manifest updated to make corpus tools optional. --- pipeline_defs/documentary-montage.yaml | 18 +- .../documentary-montage/asset-director.md | 113 +++- tools/video/direct_clip_search.py | 503 ++++++++++++++++++ 3 files changed, 622 insertions(+), 12 deletions(-) create mode 100644 tools/video/direct_clip_search.py diff --git a/pipeline_defs/documentary-montage.yaml b/pipeline_defs/documentary-montage.yaml index c10df44..cbd75c1 100644 --- a/pipeline_defs/documentary-montage.yaml +++ b/pipeline_defs/documentary-montage.yaml @@ -90,28 +90,30 @@ stages: - brief produces: - asset_manifest - required_tools: + required_tools: [] # fast path needs only direct_clip_search; standard path needs corpus_builder + clip_search + optional_tools: + - direct_clip_search - corpus_builder - clip_search - optional_tools: - music_gen tools_available: + - direct_clip_search - corpus_builder - clip_search - music_gen checkpoint_required: true human_approval_default: false review_focus: - - Corpus size is at least 8x the slot count - - Every slot has exactly one picked clip with score >= 0.22 - - No clip_id is picked for two slots (exclude_ids enforced) - - diversify ran clean on the final timeline - - rejected_picks log has reasons for passes + - Every slot has exactly one picked clip + - No clip_id is picked for two slots - Provenance (provider, original_url, license) present on every asset + - "Standard path: corpus size >= 8x slot count, scores >= 0.22, diversify ran clean" + - "Fast path: thumbnails inspected, cross-act reuse noted, gaps filled" + - rejected_picks log has reasons for passes (standard path) success_criteria: - Schema-valid asset_manifest artifact - One video asset per scene slot - - metadata.corpus_stats present + - metadata.corpus_stats present (standard path) OR metadata.search_stats present (fast path) - name: edit skill: pipelines/documentary-montage/edit-director diff --git a/skills/pipelines/documentary-montage/asset-director.md b/skills/pipelines/documentary-montage/asset-director.md index 5e0c9dc..d543892 100644 --- a/skills/pipelines/documentary-montage/asset-director.md +++ b/skills/pipelines/documentary-montage/asset-director.md @@ -3,13 +3,45 @@ ## When To Use The shot list exists. You now have to actually go out and find the -clips that fill each slot. This is a two-step operation: +clips that fill each slot. There are two paths: + +### Standard Path: Corpus + CLIP Retrieval 1. **Build the corpus** — fan the scene director's queries out across Pexels / Archive.org / NASA / Wikimedia / Unsplash and download/embed the candidates. 2. **Pick per slot** — run CLIP retrieval against the corpus with each slot description and choose one winner per slot. +Best for: 50+ slot productions, automated diversification, hands-off +slot filling where CLIP similarity ranking matters. + +### Fast Path: Direct Search (Recommended for act-by-act production) + +1. **Search and download** — use `direct_clip_search` to fan out + across all available providers and download 2-3 clips per query. + No CLIP embeddings, no corpus index, no .npy files. +2. **Inspect thumbnails** — browse the extracted thumbnails (or use a + sub-agent) to verify visual matches against slot descriptions. +3. **Map clips to slots** — manually assign the best clip to each slot + based on visual inspection. + +Best for: act-by-act production with user review between acts, fast +iteration, productions under 30 slots per act. + +**Cross-act reuse:** When producing act-by-act, clips downloaded for +earlier acts can fill slots in later acts. Point the agent at +previously downloaded directories and reuse clips that match new slot +descriptions. This saved 40-50% of download time in production. + +**Parallel workflow:** While `direct_clip_search` runs in background, +simultaneously generate TTS narration, build audio mixes, create +subtitles, and search for music. This dramatically reduces total +production time. + +**Fallback:** If the fast path yields poor visual matches for specific +slots, use `corpus_builder` + `clip_search` for just those slots. +The two approaches are not mutually exclusive. + The output is an `asset_manifest` mapping every slot to exactly one clip with full provenance. @@ -20,8 +52,9 @@ clip with full provenance. | Schema | `schemas/artifacts/asset_manifest.schema.json` | Artifact validation | | Prior artifact | `state.artifacts["scene_plan"]["scene_plan"]` | Slot descriptions + queries + preferred_sources | | Prior artifact | `state.artifacts["idea"]["brief"]` | `era_mix`, `sources_allowed`, `music_plan` | -| Tool | `corpus_builder` | Populates the retrieval index | -| Tool | `clip_search` | Ranks clips against slot descriptions | +| Tool (fast path) | `direct_clip_search` | Lightweight multi-provider search + download | +| Tool (standard path) | `corpus_builder` | Populates the retrieval index with CLIP embeddings | +| Tool (standard path) | `clip_search` | Ranks clips against slot descriptions | | Tool (optional) | `music_gen`, user's `music_library/` | Score bed | ## Mental Model @@ -39,7 +72,79 @@ Three rules that follow from that: 3. **Pick per slot, not per clip.** Every clip only belongs to one slot in the final edit. Use `exclude_ids` to prevent double-use. -## Process +## Process — Fast Path (Direct Search) + +Use this when producing act-by-act with user review between acts, or +when the total slot count is under ~30. + +### F1. Run `direct_clip_search` In Background + +Fire off the search while you work on narration/audio/subtitles in +parallel: + +```python +direct_clip_search.execute({ + "output_dir": "projects//assets/video/raw_act2", + "queries": [ + {"query": "cesium atomic clock laboratory", "slot_id": "slot_01"}, + {"query": "laser beam laboratory optics", "slot_id": "slot_02"}, + {"query": "satellite dish night sky", "slot_id": "slot_03"}, + # ... one per slot + ], + "sources": ["pexels", "archive_org", "wikimedia"], # or omit for all available + "clips_per_query": 3, + "filters": { + "min_duration": 3, + "max_duration": 40, + "orientation": "landscape", + "min_width": 1280, + }, +}) +``` + +**Key parameters:** +- `clips_per_query=3` is the sweet spot. Enough choice, fast download. +- Omit `sources` to search all available providers automatically. +- Set `skip_existing=true` (default) to avoid re-downloading on retry. + +### F2. Inspect Thumbnails + +Browse `/thumbnails/` to verify each clip. Use a sub-agent +to read thumbnail images for visual confirmation if needed. + +For each slot, pick the best-matching clip from the downloaded set. + +### F3. Cross-Act Reuse + +When working on Acts 2-5, check clips from earlier acts before +downloading new ones. Many thematic overlaps exist across acts: + +- Laboratory footage (microscopes, lasers, scientists) +- Technology shots (servers, satellites, circuits) +- Nature/abstract footage (mountains, space, time-lapse) + +Point the agent at previous act directories and map existing clips +to new slots before running new searches. + +### F4. Fill Gaps With Targeted Searches + +If specific slots have no good match after the initial search: +1. Rewrite the query (more concrete nouns, different vocabulary). +2. Run `direct_clip_search` with just those queries. +3. If still no match, fall back to `corpus_builder` + `clip_search` + for those specific slots only. + +### F5. Record The Asset Manifest + +Same format as the standard path (see step 9 below). The `source_tool` +field should be `"direct_clip_search"` instead of `"corpus_builder"`. + +--- + +## Process — Standard Path (Corpus + CLIP Retrieval) + +Use this for large productions (50+ slots), when you need automated +CLIP-based ranking, or when the fast path yields poor matches. ### 1. Resolve The Corpus Directory diff --git a/tools/video/direct_clip_search.py b/tools/video/direct_clip_search.py new file mode 100644 index 0000000..53a118c --- /dev/null +++ b/tools/video/direct_clip_search.py @@ -0,0 +1,503 @@ +"""Direct clip search: lightweight provider-agnostic stock footage acquisition. + +This tool replaces the heavy corpus_builder → clip_search pipeline when +you already know what you want and just need clips downloaded fast. It +uses the same StockSource adapter protocol (Pexels, Archive.org, NASA, +Wikimedia, Unsplash, ...) but skips CLIP embeddings, motion scoring, +index.jsonl, and .npy files entirely. + +When to use this instead of corpus_builder +------------------------------------------ +- You have a shot list and know the queries for each slot. +- You want clips downloaded in minutes, not tens of minutes. +- You plan to inspect thumbnails yourself (or via a sub-agent) rather + than relying on CLIP similarity ranking. +- You are doing act-by-act production and can reuse clips across acts + by pointing at previously downloaded directories. + +When to use corpus_builder instead +---------------------------------- +- You need CLIP-based semantic ranking (clip_search.rank_for_slot). +- You have 50+ slots and want automated diversification. +- The visual match between query text and actual footage matters more + than speed. + +What it does per query +---------------------- +1. Fan out across all available (or specified) StockSource adapters. +2. Download up to `clips_per_query` clips per query. +3. Extract one thumbnail per clip via ffmpeg (for visual inspection). +4. Return full metadata: paths, durations, sources, thumbnails. + +No CLIP model. No embeddings. No corpus index. Just files on disk. +""" +from __future__ import annotations + +import subprocess +import time +import urllib.parse +from pathlib import Path +from typing import Any, Optional + +from tools.base_tool import ( + BaseTool, + Determinism, + ExecutionMode, + ResourceProfile, + RetryPolicy, + ToolResult, + ToolRuntime, + ToolStability, + ToolStatus, + ToolTier, +) + + +class DirectClipSearch(BaseTool): + name = "direct_clip_search" + version = "0.1.0" + tier = ToolTier.SOURCE + capability = "clip_acquisition" + provider = "openmontage" + stability = ToolStability.BETA + execution_mode = ExecutionMode.SYNC + determinism = Determinism.DETERMINISTIC + runtime = ToolRuntime.HYBRID # local disk + network APIs + + dependencies = [ + "python:requests", + ] + install_instructions = ( + "At least one stock source must be configured:\n" + " PEXELS_API_KEY for Pexels (free at https://www.pexels.com/api/)\n" + " UNSPLASH_ACCESS_KEY for Unsplash (see https://unsplash.com/documentation)\n" + " archive.org, nasa, and wikimedia work without API keys" + ) + agent_skills = [] + + capabilities = [ + "multi_source_search", + "clip_download", + "thumbnail_extraction", + ] + supports = { + "multi_source": True, + "video_and_image": True, + "provider_agnostic": True, + "cross_act_reuse": True, + } + best_for = [ + "act-by-act documentary production with manual clip selection", + "fast B-roll acquisition when you know what you need", + "downloading clips from multiple providers in one call", + "building clip libraries without CLIP embedding overhead", + ] + not_good_for = [ + "semantic similarity ranking (use corpus_builder + clip_search)", + "automated slot filling without human review", + ] + fallback_tools = ["corpus_builder", "pexels_video"] + + input_schema = { + "type": "object", + "required": ["output_dir", "queries"], + "properties": { + "output_dir": { + "type": "string", + "description": ( + "Directory where clips and thumbnails are saved. " + "e.g. projects/foo/assets/video/raw_act2" + ), + }, + "queries": { + "type": "array", + "minItems": 1, + "items": { + "type": "object", + "required": ["query"], + "properties": { + "query": { + "type": "string", + "description": "Search term for stock APIs", + }, + "slot_id": { + "type": "string", + "description": ( + "Optional slot reference (e.g. 'slot_03'). " + "Used to organize output and track provenance." + ), + }, + "kind": { + "type": "string", + "enum": ["video", "image", "any"], + "default": "video", + }, + }, + }, + }, + "sources": { + "type": "array", + "items": {"type": "string"}, + "description": ( + "Source adapter names to search (e.g. ['pexels','archive_org']). " + "Defaults to all available sources." + ), + }, + "clips_per_query": { + "type": "integer", + "default": 3, + "minimum": 1, + "maximum": 20, + "description": ( + "How many clips to download per query (across all sources). " + "Lower = faster. 2-3 is enough for manual selection." + ), + }, + "filters": { + "type": "object", + "properties": { + "min_duration": {"type": "number"}, + "max_duration": {"type": "number"}, + "orientation": { + "type": "string", + "enum": ["landscape", "portrait", "square"], + }, + "min_width": {"type": "integer"}, + }, + }, + "extract_thumbnails": { + "type": "boolean", + "default": True, + "description": ( + "Extract a mid-frame thumbnail from each video for visual " + "inspection. Uses ffmpeg, not CLIP." + ), + }, + "skip_existing": { + "type": "boolean", + "default": True, + "description": "Skip download if a file with the same clip_id already exists.", + }, + }, + } + + resource_profile = ResourceProfile( + cpu_cores=1, ram_mb=512, vram_mb=0, disk_mb=2000, network_required=True + ) + retry_policy = RetryPolicy(max_retries=1, retryable_errors=["timeout", "rate_limit"]) + side_effects = [ + "downloads clips to /clips/", + "extracts thumbnails to /thumbnails/", + "calls external stock APIs", + ] + user_visible_verification = [ + "Browse /thumbnails/ to visually verify clip matches", + "Play clips from /clips/ to check quality", + ] + + def get_status(self) -> ToolStatus: + try: + from tools.video.stock_sources import available_sources + except Exception: + return ToolStatus.UNAVAILABLE + if len(available_sources()) == 0: + return ToolStatus.UNAVAILABLE + return ToolStatus.AVAILABLE + + def get_info(self) -> dict[str, Any]: + info = super().get_info() + try: + from tools.video.stock_sources import source_catalog, source_summary + info["source_provider_menu"] = source_catalog() + info["source_provider_summary"] = source_summary() + except Exception: + info["source_provider_menu"] = [] + info["source_provider_summary"] = { + "configured": 0, + "total": 0, + "available_source_names": [], + "unavailable_source_names": [], + } + return info + + def estimate_cost(self, inputs: dict[str, Any]) -> float: + return 0.0 # all sources are free-tier + + # ------------------------------------------------------------------ + # Execute + # ------------------------------------------------------------------ + + def execute(self, inputs: dict[str, Any]) -> ToolResult: + start = time.time() + try: + from tools.video.stock_sources import ( + SearchFilters, + all_sources, + available_sources, + get_source, + source_summary, + ) + + output_dir = Path(inputs["output_dir"]) + queries: list[dict] = list(inputs["queries"]) + source_names: Optional[list[str]] = inputs.get("sources") + filters_in: dict = inputs.get("filters") or {} + clips_per_query = int(inputs.get("clips_per_query", 3)) + extract_thumbs = bool(inputs.get("extract_thumbnails", True)) + skip_existing = bool(inputs.get("skip_existing", True)) + + clips_dir = output_dir / "clips" + thumbs_dir = output_dir / "thumbnails" + clips_dir.mkdir(parents=True, exist_ok=True) + if extract_thumbs: + thumbs_dir.mkdir(parents=True, exist_ok=True) + + # --- Resolve sources --- + if source_names: + sources = [] + unavailable: list[str] = [] + known = {src.name: src for src in all_sources()} + for name in source_names: + s = known.get(name) + if s is None: + try: + s = get_source(name) + except KeyError: + return ToolResult( + success=False, + error=f"Unknown stock source: {name!r}. " + f"Available: {[src.name for src in all_sources()]}", + ) + if s.is_available(): + sources.append(s) + else: + unavailable.append(name) + if unavailable: + summary = source_summary() + return ToolResult( + success=False, + error=( + f"Requested sources unavailable: {', '.join(unavailable)}. " + f"Available: {', '.join(summary['available_source_names']) or 'none'}." + ), + ) + else: + sources = available_sources() + + if not sources: + return ToolResult( + success=False, + error="No stock sources available. " + self.install_instructions, + ) + + # --- Search and download --- + downloaded: list[dict] = [] + errors: list[dict] = [] + skipped = 0 + per_source_counts: dict[str, int] = {s.name: 0 for s in sources} + + for q_spec in queries: + query = q_spec["query"] + slot_id = q_spec.get("slot_id", "") + kind = q_spec.get("kind", "video") + collected_for_query = 0 + + filters = SearchFilters( + kind=kind, + per_page=max(clips_per_query * 2, 10), # fetch extra for filtering + min_duration=filters_in.get("min_duration"), + max_duration=filters_in.get("max_duration"), + orientation=filters_in.get("orientation"), + min_width=filters_in.get("min_width"), + ) + + for src in sources: + if collected_for_query >= clips_per_query: + break + + try: + candidates = src.search(query, filters) + except Exception as e: + errors.append({ + "phase": "search", + "source": src.name, + "query": query, + "error": f"{type(e).__name__}: {e}", + }) + continue + + for cand in candidates: + if collected_for_query >= clips_per_query: + break + + clip_id = cand.clip_id + ext = _guess_ext(cand) + clip_path = clips_dir / f"{clip_id}{ext}" + + # Skip if already downloaded + if skip_existing and clip_path.exists() and clip_path.stat().st_size > 1024: + skipped += 1 + # Still record it in results so the agent knows it's there + thumb_path = thumbs_dir / f"{clip_id}.jpg" + downloaded.append({ + "clip_id": clip_id, + "source": cand.source, + "source_id": cand.source_id, + "source_url": cand.source_url, + "query": query, + "slot_id": slot_id, + "kind": cand.kind, + "path": str(clip_path), + "thumbnail": str(thumb_path) if thumb_path.exists() else "", + "duration": cand.duration, + "width": cand.width, + "height": cand.height, + "creator": cand.creator, + "license": cand.license, + "source_tags": cand.source_tags, + "skipped_existing": True, + }) + collected_for_query += 1 + continue + + # Download + try: + src.download(cand, clip_path) + except Exception as e: + errors.append({ + "phase": "download", + "clip_id": clip_id, + "source": src.name, + "error": f"{type(e).__name__}: {e}", + }) + continue + + if not clip_path.exists() or clip_path.stat().st_size < 1024: + errors.append({ + "phase": "download", + "clip_id": clip_id, + "source": src.name, + "error": "Download produced empty or tiny file", + }) + try: + if clip_path.exists(): + clip_path.unlink() + except OSError: + pass + continue + + # Extract thumbnail + thumb_path_str = "" + if extract_thumbs and cand.kind == "video": + thumb_path = thumbs_dir / f"{clip_id}.jpg" + try: + _extract_mid_thumbnail(clip_path, thumb_path) + if thumb_path.exists(): + thumb_path_str = str(thumb_path) + except Exception: + pass # thumbnail failure is non-fatal + + per_source_counts[src.name] = per_source_counts.get(src.name, 0) + 1 + collected_for_query += 1 + + downloaded.append({ + "clip_id": clip_id, + "source": cand.source, + "source_id": cand.source_id, + "source_url": cand.source_url, + "query": query, + "slot_id": slot_id, + "kind": cand.kind, + "path": str(clip_path), + "thumbnail": thumb_path_str, + "duration": cand.duration, + "width": cand.width, + "height": cand.height, + "creator": cand.creator, + "license": cand.license, + "source_tags": cand.source_tags, + "skipped_existing": False, + }) + + elapsed = time.time() - start + + return ToolResult( + success=True, + data={ + "output_dir": str(output_dir), + "clips_downloaded": len([d for d in downloaded if not d.get("skipped_existing")]), + "clips_reused": skipped, + "total_clips": len(downloaded), + "per_source_counts": per_source_counts, + "queries_run": len(queries), + "resolved_sources": [s.name for s in sources], + "clips": downloaded, + "errors": errors[:25], + }, + cost_usd=0.0, + duration_seconds=round(elapsed, 2), + ) + + except Exception as e: + import traceback + return ToolResult( + success=False, + error=f"{type(e).__name__}: {e}\n{traceback.format_exc()[-800:]}", + ) + + +# ---------------------------------------------------------------------- +# Helpers +# ---------------------------------------------------------------------- + + +def _guess_ext(cand) -> str: + """Extract a sensible file extension from a candidate's URL.""" + known = {".mp4", ".mov", ".mkv", ".webm", ".ogv", ".m4v", + ".jpg", ".jpeg", ".png", ".tif", ".tiff"} + path = urllib.parse.urlparse(cand.download_url).path + ext = Path(path).suffix.lower() + if ext in known: + return ".jpg" if ext == ".jpeg" else ext + return ".mp4" if cand.kind == "video" else ".jpg" + + +def _extract_mid_thumbnail(video_path: Path, thumb_path: Path) -> None: + """Extract a single frame from the middle of the video via ffmpeg. + + This is deliberately simple — one frame, no CLIP, no motion score. + The agent or user inspects the thumbnail visually to decide if the + clip is a good match. + """ + thumb_path.parent.mkdir(parents=True, exist_ok=True) + + # Probe duration first + probe_cmd = [ + "ffprobe", "-v", "quiet", + "-show_entries", "format=duration", + "-of", "csv=p=0", + str(video_path), + ] + try: + result = subprocess.run( + probe_cmd, capture_output=True, text=True, timeout=10 + ) + duration = float(result.stdout.strip() or "0") + except (ValueError, subprocess.TimeoutExpired, FileNotFoundError): + duration = 0 + + # Seek to the middle (or 2 seconds in if duration unknown) + seek_time = max(0.5, duration / 2) if duration > 1 else 2.0 + + extract_cmd = [ + "ffmpeg", "-y", + "-ss", str(round(seek_time, 2)), + "-i", str(video_path), + "-frames:v", "1", + "-q:v", "3", + str(thumb_path), + ] + subprocess.run( + extract_cmd, capture_output=True, timeout=15, + creationflags=getattr(subprocess, "CREATE_NO_WINDOW", 0), + )