diff --git a/README.md b/README.md index 61e3ad1..75c8e7c 100644 --- a/README.md +++ b/README.md @@ -7,6 +7,7 @@

The first open-source, agentic video production system.

+ Paste A Video  ·  Quick Start  ·  Try These Prompts  ·  Pipelines  ·  @@ -29,6 +30,35 @@ Turn your AI coding assistant into a full video production studio. Describe what > **"SIGNAL FROM TOMORROW"** — a cinematic sci-fi trailer fully produced through OpenMontage: concept, script, scene plan, Veo-generated motion clips, soundtrack, and Remotion composition. +--- + +## Start From A Video You Already Love + +Most people are bad at prompting from scratch. They are much better at saying: "I want something like this." + +OpenMontage can start from a **YouTube video, Short, Reel, TikTok, or local clip** and turn it into a grounded production plan: + +1. **Paste a reference video** +2. **The agent analyzes transcript, pacing, scenes, keyframes, and style** +3. **You get 2-3 differentiated concepts, an honest tool path, cost estimates, and a sample before full production** + +```text +"Here's a YouTube Short I love. Make me something like this, but about quantum computing." +``` + +What you get back is not "best guess prompt spaghetti." You get: + +- **What it keeps** from the reference: pacing, hook style, structure, tone +- **What it changes**: topic, visual treatment, angle, narration approach +- **What it will cost** at your target duration, before asset generation starts +- **What it will actually look like** with your currently available tools + +**Why this matters:** the reference-video path is the easiest way into OpenMontage. You do not need to reverse-engineer prompts, name camera moves, or describe an aesthetic perfectly. Just point at something that works. + +**Guardrail:** OpenMontage does not aim for carbon copies. The agent is instructed to propose original, creatively differentiated variants of the reference. + +**Free-first workflow:** the reference analysis path is designed around local/free tooling for transcript extraction, scene detection, frame sampling, and structure analysis. Paid providers only enter once you approve a production direction. +

@@ -146,6 +176,14 @@ You don't need any API keys to make real videos. Out of the box, `make setup` gi Copy any of these into your AI coding assistant after setup. Each one runs a full production pipeline. +### Start from a reference video + +> "Here's a YouTube short I love. Make me something like this, but about CRISPR for high school students." + +> "Analyze this Reel and give me 3 original variants I could make for my own product launch." + +> "I like the pacing and hook in this video. Keep that energy, but turn it into a 45-second explainer about black holes." + ### Zero keys needed > "Make a 45-second animated explainer about why the sky is blue" @@ -212,6 +250,7 @@ Edit your own talking-head footage. Generate a fully animated explainer from scr - **11 production pipelines** — explainers, talking heads, screen demos, cinematic trailers, animations, podcasts, localization, and more - **49 production tools** — spanning video generation, image creation, text-to-speech, music, audio mixing, subtitles, enhancement, and analysis - **400+ agent skills** — production skills, pipeline directors, creative techniques, quality checklists, and deep technology knowledge packs that teach the agent how to use every tool like an expert +- **Reference-driven creation** — paste a video you like and the agent turns it into a grounded, differentiated production plan instead of forcing you to invent the perfect prompt from scratch - **Live web research built in** — before writing a single word of script, the agent runs 15-25+ web searches across YouTube, Reddit, news sites, and academic sources to ground your video in real, current data - **Both free/local AND cloud providers** — every capability supports open-source local alternatives alongside premium APIs. Use what you have. - **No vendor lock-in** — swap providers freely. The scored selector ranks every provider across 7 dimensions (task fit, output quality, control, reliability, cost efficiency, latency, continuity) and picks the best match automatically. diff --git a/lib/checkpoint.py b/lib/checkpoint.py index b4c3db7..8a07023 100644 --- a/lib/checkpoint.py +++ b/lib/checkpoint.py @@ -44,6 +44,7 @@ CANONICAL_STAGE_ARTIFACTS = { SUPPLEMENTARY_ARTIFACTS = { "source_media_review", # Required before first planning stage when user media exists "final_review", # Required by compose stage before presenting to user + "video_analysis_brief", # Reference-video grounding artifact carried alongside stages } diff --git a/lib/pipeline_loader.py b/lib/pipeline_loader.py index 63b8fed..727f679 100644 --- a/lib/pipeline_loader.py +++ b/lib/pipeline_loader.py @@ -56,18 +56,89 @@ def list_pipelines(defs_dir: Optional[Path] = None) -> list[str]: return [p.stem for p in defs_dir.glob("*.yaml")] -def get_stage_order(manifest: dict) -> list[str]: - """Extract the ordered list of stage names from a manifest.""" - return [stage["name"] for stage in manifest["stages"]] +def _condition_is_active(condition: Optional[str], context: Optional[dict[str, Any]]) -> bool: + """Evaluate a simple manifest condition against runtime context.""" + if not condition: + return True + if not context: + return False + return bool(context.get(condition)) + + +def get_reference_input_config(manifest: dict) -> dict[str, Any]: + """Return reference-input configuration, defaulting to disabled.""" + return manifest.get("reference_input", {}) or {} + + +def pipeline_supports_reference_input(manifest: dict) -> bool: + """Whether the manifest declares support for reference-video input.""" + return bool(get_reference_input_config(manifest).get("supported", False)) + + +def get_stage_sub_stages( + manifest: dict, + stage_name: str, + *, + context: Optional[dict[str, Any]] = None, + include_inactive: bool = True, +) -> list[dict[str, Any]]: + """Return sub-stage definitions for a stage. + + By default this returns all declared sub-stages so agents can inspect the + full workflow shape. Pass ``include_inactive=False`` with context to filter + to active sub-stages only. + """ + for stage in manifest["stages"]: + if stage["name"] != stage_name: + continue + sub_stages = list(stage.get("sub_stages", [])) + if include_inactive: + return sub_stages + return [ + sub_stage + for sub_stage in sub_stages + if _condition_is_active(sub_stage.get("condition"), context) + ] + return [] + + +def get_stage_order( + manifest: dict, + *, + include_sub_stages: bool = False, + context: Optional[dict[str, Any]] = None, +) -> list[str]: + """Extract the ordered list of stage names from a manifest. + + ``include_sub_stages=True`` exposes declarative sample/preview units to the + agent without turning them into mandatory checkpoint stages. Sub-stages are + emitted as ``.``. + """ + order: list[str] = [] + for stage in manifest["stages"]: + order.append(stage["name"]) + if not include_sub_stages: + continue + for sub_stage in get_stage_sub_stages( + manifest, + stage["name"], + context=context, + include_inactive=context is None, + ): + order.append(f"{stage['name']}.{sub_stage['name']}") + return order def get_required_tools(manifest: dict) -> set[str]: - """Collect all preferred + fallback + available tools across all stages.""" + """Collect tools across stages, sub-stages, and reference-input analysis.""" tools: set[str] = set() for stage in manifest["stages"]: tools.update(stage.get("preferred_tools", [])) tools.update(stage.get("fallback_tools", [])) tools.update(stage.get("tools_available", [])) + for sub_stage in stage.get("sub_stages", []): + tools.update(sub_stage.get("tools_available", [])) + tools.update(get_reference_input_config(manifest).get("analysis_tools", [])) return tools diff --git a/pipeline_defs/animated-explainer.yaml b/pipeline_defs/animated-explainer.yaml index bfd0a37..1ea2ac6 100644 --- a/pipeline_defs/animated-explainer.yaml +++ b/pipeline_defs/animated-explainer.yaml @@ -9,6 +9,17 @@ stability: production default_checkpoint_policy: guided +# Reference video input support +reference_input: + supported: true + analysis_depth: standard + analysis_tools: + - video_analyzer + - transcript_fetcher + - video_downloader + - scene_detect + - frame_sampler + extensions: custom_scripts: true custom_playbooks: true @@ -83,11 +94,22 @@ stages: - Cost estimate is itemized and honest - Quality/cost tradeoffs are clearly presented - Alternative production paths shown at different price points + - "Concept differentiation from reference (if reference-driven)" success_criteria: - Schema-valid proposal_packet with at least 3 concept_options - selected_concept references a valid concept_id - cost_estimate has itemized line_items with per-tool costs - approval.status is "approved" or "approved_with_changes" before proceeding + sub_stages: + - name: sample + description: "10-15 second preview clip for reference-driven productions" + condition: "video_analysis_brief_exists" + human_approval_default: true + tools_available: [tts_selector, image_selector, video_selector, video_compose, audio_mixer] + review_focus: + - "Sample represents the approved concept faithfully" + - "Audio levels and voice match approved direction" + - "Visual style matches approved playbook" # ── Production ────────────────────────────────────────────────── diff --git a/pipeline_defs/animation.yaml b/pipeline_defs/animation.yaml index 4e79905..0d1e9e8 100644 --- a/pipeline_defs/animation.yaml +++ b/pipeline_defs/animation.yaml @@ -10,6 +10,17 @@ category: animation stability: production default_checkpoint_policy: guided +# Reference video input support +reference_input: + supported: true + analysis_depth: standard + analysis_tools: + - video_analyzer + - transcript_fetcher + - video_downloader + - scene_detect + - frame_sampler + extensions: custom_scripts: true custom_playbooks: true @@ -89,6 +100,16 @@ stages: - selected_concept includes animation_mode and reuse_strategy - cost_estimate has itemized line_items with per-tool costs - approval.status is "approved" or "approved_with_changes" before proceeding + sub_stages: + - name: sample + description: "10-15 second animation preview for reference-driven productions" + condition: "video_analysis_brief_exists" + human_approval_default: true + tools_available: [tts_selector, image_selector, video_selector, video_compose, audio_mixer, math_animate, diagram_gen] + review_focus: + - "Sample represents the approved animation mode faithfully" + - "Visual identity matches approved playbook and style" + - "Pacing matches reference inspiration" # ── Production ────────────────────────────────────────────────── diff --git a/pipeline_defs/cinematic.yaml b/pipeline_defs/cinematic.yaml index 0e40fc4..a327f36 100644 --- a/pipeline_defs/cinematic.yaml +++ b/pipeline_defs/cinematic.yaml @@ -9,6 +9,17 @@ category: cinematic stability: production default_checkpoint_policy: guided +# Reference video input support +reference_input: + supported: true + analysis_depth: deep + analysis_tools: + - video_analyzer + - transcript_fetcher + - video_downloader + - scene_detect + - frame_sampler + orchestration: mode: executive-producer skill: pipelines/cinematic/executive-producer @@ -88,6 +99,16 @@ stages: - At least 3 concept options with different emotional arcs - Delivery promise present with motion_required flag - Cost estimate includes per-item breakdown + sub_stages: + - name: sample + description: "10-15 second cinematic preview for reference-driven productions" + condition: "video_analysis_brief_exists" + human_approval_default: true + tools_available: [tts_selector, image_selector, video_selector, video_compose, audio_mixer, color_grade] + review_focus: + - "Sample captures the approved emotional tone" + - "Color palette and lighting match approved direction" + - "Music mood matches approved cinematic treatment" - name: script skill: pipelines/cinematic/script-director diff --git a/schemas/artifacts/__init__.py b/schemas/artifacts/__init__.py index 41246a4..319cd6c 100644 --- a/schemas/artifacts/__init__.py +++ b/schemas/artifacts/__init__.py @@ -25,6 +25,7 @@ ARTIFACT_NAMES = [ "decision_log", "source_media_review", "final_review", + "video_analysis_brief", ] diff --git a/schemas/artifacts/video_analysis_brief.schema.json b/schemas/artifacts/video_analysis_brief.schema.json new file mode 100644 index 0000000..9003035 --- /dev/null +++ b/schemas/artifacts/video_analysis_brief.schema.json @@ -0,0 +1,223 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "title": "video_analysis_brief", + "description": "Structured analysis of a reference video — produced by VideoAnalyzer, enriched by agent vision. Used as grounding context for reference-driven pipeline productions.", + "type": "object", + "required": ["version", "source", "content_analysis", "structure_analysis"], + "properties": { + "version": { "const": "1.0" }, + "source": { + "type": "object", + "required": ["type", "duration_seconds"], + "properties": { + "type": { + "type": "string", + "enum": ["youtube", "shorts", "instagram", "tiktok", "local_file", "other_url"] + }, + "url": { "type": "string" }, + "local_path": { "type": "string" }, + "title": { "type": "string" }, + "duration_seconds": { "type": "number" }, + "resolution": { "type": "string" }, + "platform_metadata": { + "type": "object", + "properties": { + "uploader": { "type": "string" }, + "upload_date": { "type": "string" }, + "description": { "type": "string" }, + "view_count": { "type": "integer" }, + "like_count": { "type": "integer" } + } + } + } + }, + "content_analysis": { + "type": "object", + "required": ["summary", "topics", "target_audience"], + "properties": { + "summary": { + "type": "string", + "description": "2-3 sentence content summary" + }, + "topics": { + "type": "array", + "items": { "type": "string" } + }, + "key_claims": { + "type": "array", + "items": { "type": "string" } + }, + "target_audience": { "type": "string" }, + "tone": { + "type": "string", + "enum": ["educational", "entertaining", "cinematic", "corporate", + "casual", "dramatic", "inspirational", "humorous"] + }, + "hook_technique": { + "type": "string", + "description": "How the video opens to grab attention" + }, + "call_to_action": { "type": "string" } + } + }, + "structure_analysis": { + "type": "object", + "required": ["total_scenes", "scenes", "pacing_profile"], + "properties": { + "total_scenes": { "type": "integer" }, + "scenes": { + "type": "array", + "items": { + "type": "object", + "required": ["scene_index", "start_time", "end_time", "description"], + "properties": { + "scene_index": { "type": "integer" }, + "start_time": { "type": "number" }, + "end_time": { "type": "number" }, + "description": { "type": "string" }, + "narration_text": { "type": "string" }, + "visual_type": { + "type": "string", + "enum": ["talking_head", "b_roll", "screen_recording", "animation", + "text_card", "stock_footage", "diagram", "chart", + "product_shot", "transition", "other"] + }, + "shot_language": { + "type": "object", + "properties": { + "shot_size": { "type": "string" }, + "camera_movement": { "type": "string" }, + "lighting_key": { "type": "string" }, + "depth_of_field": { "type": "string" } + } + }, + "dominant_colors": { + "type": "array", + "items": { "type": "string" } + }, + "on_screen_text": { "type": "string" }, + "energy_level": { + "type": "string", + "enum": ["low", "medium", "high", "peak"] + } + } + } + }, + "pacing_profile": { + "type": "object", + "properties": { + "avg_scene_duration_seconds": { "type": "number" }, + "shortest_scene_seconds": { "type": "number" }, + "longest_scene_seconds": { "type": "number" }, + "cuts_per_minute": { "type": "number" }, + "pacing_style": { + "type": "string", + "enum": ["slow_contemplative", "steady_educational", + "dynamic_social", "rapid_fire", "variable"] + } + } + } + } + }, + "style_profile": { + "type": "object", + "properties": { + "color_palette": { + "type": "object", + "properties": { + "primary_colors": { + "type": "array", + "items": { "type": "string" } + }, + "accent_colors": { + "type": "array", + "items": { "type": "string" } + }, + "overall_mood": { "type": "string" } + } + }, + "typography_observed": { "type": "string" }, + "transition_types": { + "type": "array", + "items": { "type": "string" } + }, + "music_style": { "type": "string" }, + "narration_style": { + "type": "object", + "properties": { + "has_narration": { "type": "boolean" }, + "speaker_count": { "type": "integer" }, + "delivery_style": { "type": "string" }, + "words_per_minute": { "type": "number" } + } + }, + "subtitle_style": { "type": "string" }, + "production_quality": { + "type": "string", + "enum": ["amateur", "prosumer", "professional", "broadcast"] + }, + "closest_playbook": { "type": "string" }, + "playbook_delta": { "type": "string" } + } + }, + "narration_transcript": { + "type": "object", + "properties": { + "full_text": { "type": "string" }, + "segments": { + "type": "array", + "items": { + "type": "object", + "properties": { + "start": { "type": "number" }, + "end": { "type": "number" }, + "text": { "type": "string" }, + "speaker": { "type": "string" } + } + } + }, + "language": { "type": "string" }, + "word_count": { "type": "integer" } + } + }, + "replication_guidance": { + "type": "object", + "properties": { + "suggested_pipeline": { "type": "string" }, + "suggested_playbook": { "type": "string" }, + "key_elements_to_replicate": { + "type": "array", + "items": { "type": "string" }, + "description": "The 3-5 things that make this video work" + }, + "elements_requiring_custom_work": { + "type": "array", + "items": { "type": "string" } + }, + "estimated_complexity": { + "type": "string", + "enum": ["simple", "moderate", "complex", "beyond_current_capability"] + }, + "motion_required": { "type": "boolean" }, + "creative_differentiation_seeds": { + "type": "array", + "items": { "type": "string" }, + "description": "3-5 ways the output should DIFFER from the reference to avoid being a copy" + }, + "playbook_customizations": { "type": "object" } + } + }, + "keyframes": { + "type": "array", + "items": { + "type": "object", + "properties": { + "timestamp": { "type": "number" }, + "scene_index": { "type": "integer" }, + "path": { "type": "string" }, + "description": { "type": "string" } + } + } + } + } +} diff --git a/schemas/pipelines/pipeline_manifest.schema.json b/schemas/pipelines/pipeline_manifest.schema.json index e26abd5..ebf73b4 100644 --- a/schemas/pipelines/pipeline_manifest.schema.json +++ b/schemas/pipelines/pipeline_manifest.schema.json @@ -95,6 +95,22 @@ "success_criteria": { "type": "array", "items": { "type": "string" } + }, + "sub_stages": { + "type": "array", + "description": "Optional sub-stages within a stage (e.g., sample preview for reference-driven productions)", + "items": { + "type": "object", + "required": ["name"], + "properties": { + "name": { "type": "string" }, + "description": { "type": "string" }, + "condition": { "type": "string", "description": "When this sub-stage activates (e.g., video_analysis_brief_exists)" }, + "human_approval_default": { "type": "boolean", "default": true }, + "tools_available": { "type": "array", "items": { "type": "string" } }, + "review_focus": { "type": "array", "items": { "type": "string" } } + } + } } }, "additionalProperties": false @@ -106,6 +122,15 @@ "enum": ["guided", "manual_all", "auto_noncreative"], "default": "guided" }, + "reference_input": { + "type": "object", + "description": "Configuration for reference video input support", + "properties": { + "supported": { "type": "boolean", "default": false }, + "analysis_depth": { "type": "string", "enum": ["transcript_only", "standard", "deep"], "default": "standard" }, + "analysis_tools": { "type": "array", "items": { "type": "string" } } + } + }, "metadata": { "type": "object" }, "orchestration": { "type": "object", diff --git a/skills/meta/checkpoint-protocol.md b/skills/meta/checkpoint-protocol.md index b6e4984..688a65c 100644 --- a/skills/meta/checkpoint-protocol.md +++ b/skills/meta/checkpoint-protocol.md @@ -115,6 +115,40 @@ If a checkpoint exists with status `"awaiting_human"`: 2. Present the checkpoint data for review 3. Wait for approval before proceeding +### Sample Checkpoint (Reference-Driven Productions) + +When a production is reference-driven (VideoAnalysisBrief exists), there is an +additional checkpoint between proposal approval and full production: + +| Stage | checkpoint_required | human_approval_default | Notes | +|-------|--------------------|-----------------------|-------| +| `sample` | true | true | Always requires human approval | + +The sample checkpoint: +1. Presents: rendered sample clip (10-15 seconds) +2. Cost: sample cost vs. projected full-video cost +3. Action: approve (→ proceed to script), revise (→ re-generate sample), abort + +The sample checkpoint is NOT a pipeline stage — it's a sub-checkpoint within the +proposal stage. It does not produce a canonical artifact. It produces a rendered +preview clip stored at `projects//assets/sample/sample_v{N}.mp4`. + +**Presentation format:** +``` +## Sample Preview Ready + +**Sample clip:** [path to sample_v1.mp4] +- Duration: [X] seconds (hook + 1 middle scene) +- Voice: [TTS provider + voice name] +- Visuals: [description — AI images, Remotion animations, etc.] +- Music: [source] + +**Sample cost:** $[X.XX] +**Projected full video cost:** $[X.XX] + +Does this feel right? I can adjust: voice, visual style, pacing, music, colors. +``` + ## Key Principles 1. **Always checkpoint completed work.** Even if `checkpoint_required: false`, consider checkpointing anyway if the stage took significant time or cost. Losing work is worse than an extra file on disk. diff --git a/skills/meta/creative-intake.md b/skills/meta/creative-intake.md index 738ec71..0b6582e 100644 --- a/skills/meta/creative-intake.md +++ b/skills/meta/creative-intake.md @@ -55,6 +55,26 @@ The intake_brief is passed as context to the research-director, not as a formal artifact. It exists to prevent the research stage from inventing intent that the user never expressed. +## Handling Reference Video Input + +When the user provides a video URL or file as their starting point: + +1. **Read the video-reference-analyst skill** (`skills/meta/video-reference-analyst.md`) + and follow its protocol. Do not proceed with standard creative intake. + +2. The VideoAnalysisBrief replaces the need for most intake questions — it provides + tone, structure, pacing, audience signals, and style information directly from the + reference. + +3. The remaining intake questions are: + - What topic/subject for YOUR version? (if different from reference) + - How long? + - Narration yes/no? + - Budget ceiling? + +4. Do NOT ask "what should it feel like?" — the reference video IS the answer to that + question. Extract tone from the VideoAnalysisBrief instead. + ## What NOT To Do - Do not present a numbered survey. This is a conversation, not a form. diff --git a/skills/meta/onboarding.md b/skills/meta/onboarding.md index 16f9f3b..1343e20 100644 --- a/skills/meta/onboarding.md +++ b/skills/meta/onboarding.md @@ -105,12 +105,23 @@ Based on the user's tier, present **3 ready-to-use prompts** they can copy right > **Or:** "I recorded a founder update on my webcam — make it feel polished, confident, and premium without looking fake" *(Talking Head pipeline)* +**Reference-based prompts (all tiers):** + +> **Have a video you love?** Paste a YouTube link and say "make me something like this" +> — I'll analyze the style, pacing, and structure, then propose 2-3 creative variants +> you can choose from. Works with YouTube, Shorts, Instagram Reels, and TikTok. +> All analysis runs locally and free — no API keys needed. + +> **Got your own footage?** Drop in a video file and say "I want to make a video using +> this footage" — I'll transcribe it, detect scenes, and propose an edit plan. + **Rules for prompt suggestions:** - Present exactly 3 prompts. - The first prompt should be the most impressive thing their setup can produce. - Each prompt should target a different pipeline or style. - Include a brief note explaining what makes this prompt a good fit for their setup. - Use blockquote formatting so prompts are visually distinct and easy to copy. +- Always include the reference-based prompts above — they work at every tier. ### Step 5: Explain the Workflow (Briefly) diff --git a/skills/meta/reviewer.md b/skills/meta/reviewer.md index cd7fee0..95002bf 100644 --- a/skills/meta/reviewer.md +++ b/skills/meta/reviewer.md @@ -117,6 +117,39 @@ Structure your review as: | compose | Playability, duration accuracy, audio quality, pre-compose validation pass | | publish | SEO quality, metadata completeness, export packaging | +## Reference Alignment Review + +Run at **every stage** when a VideoAnalysisBrief exists (reference-driven production). + +### Checks: + +1. **Grounding check:** Does the output reference specific findings from the + VideoAnalysisBrief, or is it making things up about the reference? + - Proposal mentions "fast pacing" but reference pacing_style is "slow_contemplative" → **CRITICAL** + - Script claims reference has narration but VideoAnalysisBrief shows no narration → **CRITICAL** + +2. **Differentiation check:** Does each concept/scene have a clear creative + difference from the reference, or is it a copy? + - Proposal is a carbon copy of the reference (same topic, same structure, same treatment) → **CRITICAL** + - At least one element per concept MUST differ from the reference → **SUGGESTION** if weak + - Creative differentiation seeds from the brief should be reflected in proposals + +3. **Promise preservation:** Are the elements the user said they loved about the + reference still present in the output? + - User said "I love the pacing" but scene_plan has 2x longer scenes → **SUGGESTION** + - User said "keep the hook style" but script uses a different hook → **SUGGESTION** + +4. **Cost alignment:** Is the cost estimate still accurate, or has scope crept? + - If actual spend exceeds estimate by >30% without user re-approval → **CRITICAL** + - If new assets were added beyond the approved proposal → **SUGGESTION** + +### Severity: +- Factual errors about the reference video: **CRITICAL** +- Carbon copy with no differentiation: **CRITICAL** +- Weak differentiation (surface-level changes only): **SUGGESTION** +- User preference not honored: **SUGGESTION** +- Cost drift >30%: **CRITICAL** + ## Slideshow Risk Review Run at **scene_plan** and **edit** stages. Use `lib/slideshow_risk.py` to compute the score. diff --git a/skills/meta/video-reference-analyst.md b/skills/meta/video-reference-analyst.md new file mode 100644 index 0000000..cc25164 --- /dev/null +++ b/skills/meta/video-reference-analyst.md @@ -0,0 +1,243 @@ +# Video Reference Analyst — Meta Skill + +## When to Use + +When the user provides a video URL (YouTube, Shorts, Instagram, TikTok, or any URL) +or a local video file as a REFERENCE — meaning "make me something like this," not +"edit this footage." + +If the user says "edit this video" or "cut this into clips," route to the appropriate +footage-led pipeline (clip-factory, talking-head, hybrid) instead. This skill is for +REFERENCE-based production. + +## Detection Signals + +Trigger this skill when: +- User pastes a YouTube/Shorts/Instagram/TikTok URL +- User says "something like this," "inspired by," "in this style," "similar to" +- User uploads a video and says "I want one like this" +- User says "I saw this video and want to make something like it" + +Do NOT trigger when: +- User provides footage and says "edit this" or "cut this" → use source_media_review +- User provides audio and says "make a video for this" → standard pipeline +- User just wants a transcript → use TranscriptFetcher directly + +## Protocol + +### Step 1: Analyze the Reference + +Run VideoAnalyzer with `analysis_depth: "standard"`: + +```python +video_analyzer.execute({ + "source": "", + "analysis_depth": "standard", + "max_keyframes": 20 +}) +``` + +Read the resulting VideoAnalysisBrief. Before proceeding, present a summary to the +user. This is NOT a raw dump. It's a conversational interpretation: + +``` +"I've watched the video. Here's what I see: + +**Content:** [2-sentence summary of what the video is about] +**Style:** [1 sentence — pacing, visual treatment, energy] +**Structure:** [X scenes over Y seconds, pacing style] +**What makes it work:** [2-3 specific things — the hook technique, the pacing, +the visual transitions, the narration style] + +Now let me check what I can do with your current setup..." +``` + +**Vision analysis:** After presenting the structural data, examine the extracted +keyframes yourself. You ARE a multimodal model — look at the keyframe images and +enrich the VideoAnalysisBrief with: +- Per-frame descriptions (subjects, text, composition, color) +- Cross-frame visual continuity and style consistency +- Genre classification and production quality assessment +- Color palette extraction (dominant colors across keyframes) +- Typography style if on-screen text is present +- Transition patterns visible between sequential keyframes + +Update the brief's `content_analysis`, `style_profile`, and `replication_guidance` +fields with your visual observations. This is where the analysis becomes truly +comprehensive — the tools provide structure; your vision provides understanding. + +### Step 2: Capability Audit + +Run standard preflight: + +```bash +python -c "from tools.tool_registry import registry; import json; registry.discover(); print(json.dumps(registry.support_envelope(), indent=2))" +python -c "from tools.tool_registry import registry; import json; registry.discover(); print(json.dumps(registry.provider_menu(), indent=2))" +python -c "from tools.tool_registry import registry; import json; registry.discover(); print(json.dumps(registry.capability_catalog(), indent=2))" +``` + +Map the reference video's requirements against available capabilities: + +``` +REFERENCE NEEDS YOUR CAPABILITIES GAP +───────────────────── ───────────────────── ────────── +Video clips (sci-fi) Video gen: 0/12 configured BLOCKED without key +Narration (deep male) TTS: ElevenLabs available READY +Background music Music: MusicGen available READY +Text animations Remotion: available READY +Fast-cut editing FFmpeg: available READY +``` + +Be honest about gaps. If video generation is needed but unavailable, say so clearly: + +``` +"This reference uses generated sci-fi footage. Right now you don't have any video +generation providers configured. Here are your options: + +• Add FAL_KEY to .env → unlocks Kling 3.0, MiniMax, Wan (best for cinematic/sci-fi) +• Add REPLICATE_API_TOKEN → unlocks LTX Video (good for short clips) +• Proceed without video gen → I'll use stock footage + Remotion animations instead + (different feel, but still works) + +Which would you prefer?" +``` + +Read install_instructions from the registry for each unavailable tool — do NOT +hardcode key names or setup URLs. + +### Step 3: Ask Critical Questions + +Before proposing, gather what the VideoAnalysisBrief doesn't tell you: + +1. "Do you want narration in your version, or visuals-only with music?" +2. "How long should your video be? The reference is [X] seconds." +3. "Is there a specific topic/subject you want, or should I riff on the + same theme as the reference?" +4. "Any elements from the reference you specifically love or hate?" + +Do NOT ask all at once. Lead with the most important gap. If the user's initial +message already answers some of these, skip those. + +### Step 4: Creative Proposals (2-3 variants) + +MANDATORY: The agent must NEVER propose a carbon copy. The reference is inspiration, +not a template. Each proposal must have clear creative differentiation. + +Use this structure for each variant: + +``` +## Option [A/B/C]: "[Title]" + +**Inspired by:** [what it keeps from the reference — pacing, structure, tone] +**Creative twist:** [what it changes — angle, subject, visual treatment, hook] + +**Visual plan:** +- Playbook: [closest match + customizations] +- Visual treatment: [how visuals will be created — which tools, which providers] +- Motion: [Remotion animations / video gen clips / stock + Ken Burns / etc.] + +**Audio plan:** +- Narration: [yes/no, which TTS provider, voice style] +- Music: [library track / generated / none] +- Sound design: [any special audio needs] + +**Duration:** [X seconds] +**Estimated cost:** $[X.XX] breakdown: +- Image generation: $X.XX (N images × $X.XX each via [provider]) +- Video generation: $X.XX (N clips × $X.XX each via [provider]) +- TTS narration: $X.XX (N words via [provider]) +- Music: $X.XX ([source]) +- Total: $X.XX + +**Honest assessment:** [What this will look like realistically — don't oversell] +``` + +**Differentiation patterns:** + +| Pattern | Example | +|---------|---------| +| **Same structure, different subject** | Reference: "How black holes work" → Ours: "How neutron stars work" with same pacing | +| **Same subject, different angle** | Reference: "Kubernetes explained" → Ours: "Kubernetes from a security engineer's POV" | +| **Same tone, different visual treatment** | Reference: stock footage + voiceover → Ours: animated motion graphics + voiceover | +| **Same content, different platform** | Reference: 10-min YouTube → Ours: 60-sec Shorts version with faster pacing | +| **Counter-take** | Reference: "Why AI will replace jobs" → Ours: "Why AI won't replace YOUR job" | + +**Cost transparency is mandatory.** Each concept must include: +- Itemized cost estimate at the user's requested duration +- Cost broken down by: image gen, video gen, TTS, music, total +- Provider names for each cost line +- Honest note about what the budget buys vs. doesn't buy + +**Recommendation:** Always recommend one option with a brief reason why. Don't leave +the user paralyzed with equal choices. + +### Step 5: Sample-First Production (MANDATORY) + +After the user picks a variant, ALWAYS say: + +``` +"Great choice. Before I commit to the full [X]-second video, I'll produce a +10-15 second sample first — the opening hook + one middle scene. This lets you +hear the voice, see the visual style, and feel the pacing before we go all-in. + +Estimated sample cost: $[X.XX] +Shall I proceed with the sample?" +``` + +The sample is NOT optional. Even if the user says "just do the whole thing," push +back gently: + +``` +"I'd really recommend the sample first — it's a tiny fraction of the cost and +lets us catch any style mismatches early. If you love it, I'll proceed to the +full video immediately." +``` + +Only skip the sample if the user insists after being advised. + +**Sample contents:** +- 1-2 representative scenes (the hook + one middle scene) +- Actual TTS narration with chosen voice +- Actual generated/stock visuals +- Music bed snippet +- Subtitle style preview + +**Sample checkpoint:** +Present the sample with: "Here's a preview. Does this feel right? Things I can +adjust: voice, visual style, pacing, music, colors." + +Iterate on sample feedback until approved. Store samples at: +`projects//assets/sample/sample_v{N}.mp4` + +### Step 6: Enter Pipeline + +After sample approval, enter the appropriate pipeline with: +- VideoAnalysisBrief as grounding context in the research/proposal stage +- User's chosen variant as the approved direction +- Sample feedback incorporated into the brief +- All creative differentiation decisions recorded in the decision_log + +The pipeline takes over from here. The VideoAnalysisBrief travels alongside the +standard artifacts, providing reference grounding at every stage. + +## Multiple Reference Videos + +When the user provides multiple reference URLs: + +1. Analyze each video separately (run VideoAnalyzer on each) +2. Present a comparative summary: "Video A does X well, Video B does Y well" +3. In proposals, note which elements are inspired by which reference +4. The VideoAnalysisBrief for the primary reference travels with the pipeline; + secondary references are noted in the research_brief + +## Error Handling + +| Failure | Action | +|---------|--------| +| URL download fails | Report error, suggest: try another URL, provide local file, or proceed without reference | +| No captions available | Download video, transcribe with Whisper locally | +| Scene detection fails | Fall back to uniform frame sampling | +| All analysis fails | Ask user to describe the reference video verbally, proceed with standard creative intake | + +Never silently skip analysis steps. If something fails, tell the user what happened +and what the impact is on the analysis quality. diff --git a/skills/pipelines/animation/proposal-director.md b/skills/pipelines/animation/proposal-director.md index 6382beb..4565f58 100644 --- a/skills/pipelines/animation/proposal-director.md +++ b/skills/pipelines/animation/proposal-director.md @@ -22,6 +22,31 @@ Animation proposals have a unique dimension: **animation mode selection**. Unlik ## Process +### Step 0: Check for Reference Video Context + +Before starting proposal work, check if a VideoAnalysisBrief exists for this project. + +**When a VideoAnalysisBrief is present — Reference-Aware Animation Concept Design:** + +**HARD RULE: No carbon copies.** Each concept option MUST: +1. Name at least ONE animation element it keeps from the reference (pacing, motion style, narrative structure) +2. Name at least ONE element it changes (animation mode, visual identity, topic angle) +3. Explain WHY the change makes the output more engaging or clearer + +**Animation differentiation patterns:** + +| Pattern | Example | +|---------|---------| +| **Same topic, different animation mode** | Reference: stock footage → Ours: Manim mathematical visualization | +| **Same style, different complexity** | Reference: simple diagrams → Ours: progressive build with layers | +| **Same pacing, different visual identity** | Reference: corporate blue → Ours: vibrant neon-on-black | +| **Same narrative, different interactivity** | Reference: linear → Ours: data-driven with animated charts | + +**Mandatory Sample Protocol:** After concept approval, produce a 10-15 second sample +to validate the animation style before full production. + +**When no VideoAnalysisBrief is present:** Skip this step and proceed normally. + ### Step 1: Absorb the Research (or Direct Brief) **If a `research_brief` artifact exists:** Read it thoroughly. Extract: diff --git a/skills/pipelines/animation/research-director.md b/skills/pipelines/animation/research-director.md index 5a8fc35..72e706c 100644 --- a/skills/pipelines/animation/research-director.md +++ b/skills/pipelines/animation/research-director.md @@ -18,6 +18,40 @@ Animation videos differ from general explainers: the research must cover both ** ## Process +### Step 0: Check for Reference Video Context + +Before starting research, check if a VideoAnalysisBrief exists for this project. If it +does, this is a reference-driven production — the user provided a video they want to +riff on. + +**When a VideoAnalysisBrief is present:** + +1. Read it thoroughly. Extract: + - `content_analysis.topics` — research these topics for accuracy + - `content_analysis.key_claims` — verify these claims via web search + - `style_profile` — note the animation style (motion type, color palette, transitions) + - `structure_analysis.pacing_profile` — understand the rhythm + - `replication_guidance.creative_differentiation_seeds` — these are your concept seeds + - `replication_guidance.key_elements_to_replicate` — preserve these in proposals + +2. Your research focus SHIFTS: + - Standard research: "What topic + animation technique fits?" + - Reference-driven research: "What animation approach would DIFFERENTIATE us from the + reference while keeping the elements the user loved?" + "What animation techniques + exist for this topic that the reference DIDN'T use?" + +3. In the research_brief, add a `reference_context` section: + - The reference's animation style and technique + - What animation modes it used (motion graphics, manim, illustrative, etc.) + - Alternative animation approaches we could try instead + - What the reference did well vs. where we can improve + +4. The `angles_discovered` should explicitly position against the reference: + - "The reference used X animation style. We could try Y which is [more engaging/clearer/ + more novel] because [technique research finding]." + +**When no VideoAnalysisBrief is present:** Skip this step and proceed normally. + ### Step 1: Scope the Research Before searching anything, establish boundaries: diff --git a/skills/pipelines/cinematic/proposal-director.md b/skills/pipelines/cinematic/proposal-director.md index 66d922c..48d0579 100644 --- a/skills/pipelines/cinematic/proposal-director.md +++ b/skills/pipelines/cinematic/proposal-director.md @@ -20,6 +20,32 @@ You are the **Proposal Director** for a cinematic video (trailers, brand films, ## Process +### Step 0: Check for Reference Video Context + +Before starting proposal work, check if a VideoAnalysisBrief exists for this project. + +**When a VideoAnalysisBrief is present — Reference-Aware Cinematic Concept Design:** + +**HARD RULE: No carbon copies.** Each concept option MUST: +1. Name at least ONE cinematic element it keeps from the reference (mood, pacing, color palette, shot language) +2. Name at least ONE element it changes (emotional arc, visual treatment, subject matter, sound design) +3. Explain WHY the change creates a different emotional impact + +**Cinematic differentiation patterns:** + +| Pattern | Example | +|---------|---------| +| **Same mood, different subject** | Reference: dark sci-fi mood → Ours: same darkness applied to deep ocean | +| **Same subject, different emotional arc** | Reference: tension→reveal → Ours: wonder→scale | +| **Same pacing, different visual language** | Reference: handheld raw → Ours: locked-off geometric | +| **Same color world, different lighting** | Reference: warm golden hour → Ours: warm but tungsten/interior | + +**Mandatory Sample Protocol:** After concept approval, produce a 10-15 second cinematic +sample BEFORE full production. This is critical for cinematic work — mood mismatches are +expensive to fix downstream. Present with visual + audio + music. + +**When no VideoAnalysisBrief is present:** Skip this step and proceed normally. + ### Step 1: Absorb the Research Read the `research_brief` thoroughly. Extract: diff --git a/skills/pipelines/cinematic/research-director.md b/skills/pipelines/cinematic/research-director.md index c653c9a..2add39e 100644 --- a/skills/pipelines/cinematic/research-director.md +++ b/skills/pipelines/cinematic/research-director.md @@ -18,6 +18,40 @@ Unlike explainer research (which focuses on facts, data, and content gaps), cine ## Process +### Step 0: Check for Reference Video Context + +Before starting research, check if a VideoAnalysisBrief exists for this project. If it +does, this is a reference-driven production — the user provided a video they want to +riff on. + +**When a VideoAnalysisBrief is present:** + +1. Read it thoroughly. Extract: + - `content_analysis.topics` — research these topics for accuracy + - `content_analysis.key_claims` — verify these claims via web search + - `style_profile` — note the cinematic language (color palette, camera movements, lighting) + - `structure_analysis.scenes` — understand the shot language and emotional arc + - `replication_guidance.creative_differentiation_seeds` — these are your concept seeds + - `replication_guidance.key_elements_to_replicate` — preserve these in proposals + +2. Your research focus SHIFTS: + - Standard research: "What visual/emotional language fits this subject?" + - Reference-driven research: "What cinematic approach would DIFFERENTIATE us from the + reference while keeping the elements the user loved?" + "What mood/tone territory + is adjacent but unexplored?" + +3. In the research_brief, add a `reference_context` section: + - The reference's cinematic language (shot types, pacing, color palette) + - What emotional territory it occupies + - Adjacent emotional territories we could explore instead + - How the reference's visual approach could be evolved or reinterpreted + +4. The `angles_discovered` should explicitly position against the reference: + - "The reference uses X mood/palette/pacing. We could try Y which creates + [different emotional impact] because [research finding]." + +**When no VideoAnalysisBrief is present:** Skip this step and proceed normally. + ### Step 1: Classify the Brief Before searching, extract from the user's request: diff --git a/skills/pipelines/explainer/proposal-director.md b/skills/pipelines/explainer/proposal-director.md index cb70865..7023b73 100644 --- a/skills/pipelines/explainer/proposal-director.md +++ b/skills/pipelines/explainer/proposal-director.md @@ -22,6 +22,36 @@ Think of yourself as a creative agency pitching to a client: you present concept ## Process +### Step 0: Check for Reference Video Context + +Before starting proposal work, check if a VideoAnalysisBrief exists for this project. + +**When a VideoAnalysisBrief is present — Reference-Aware Concept Design:** + +**HARD RULE: No carbon copies.** Each concept option MUST: +1. Name at least ONE element it keeps from the reference (pacing, structure, tone, hook style) +2. Name at least ONE element it changes (topic angle, visual treatment, narration approach) +3. Explain WHY the change makes the output better, not just different + +**Differentiation patterns:** + +| Pattern | Example | +|---------|---------| +| **Same structure, different subject** | Reference: "How black holes work" → Ours: "How neutron stars work" with same pacing | +| **Same subject, different angle** | Reference: "Kubernetes explained" → Ours: "Kubernetes from a security engineer's POV" | +| **Same tone, different visual treatment** | Reference: stock footage + voiceover → Ours: animated motion graphics + voiceover | +| **Same content, different platform** | Reference: 10-min YouTube → Ours: 60-sec Shorts version with faster pacing | +| **Counter-take** | Reference: "Why AI will replace jobs" → Ours: "Why AI won't replace YOUR job" | + +**Mandatory Sample Protocol:** After the user approves a concept, BEFORE entering the +script stage, produce a 10-15 second sample: +1. The opening hook (first 5-7 seconds) + one representative middle scene +2. Actual TTS voice, actual visual style, music bed snippet +3. Present with: "Here's a preview. Does this feel right?" +4. Iterate until approved, then proceed to full production + +**When no VideoAnalysisBrief is present:** Skip this step and proceed normally. + ### Step 1: Absorb the Research Read the `research_brief` thoroughly. Extract: diff --git a/skills/pipelines/explainer/research-director.md b/skills/pipelines/explainer/research-director.md index 102c8e9..34bbb17 100644 --- a/skills/pipelines/explainer/research-director.md +++ b/skills/pipelines/explainer/research-director.md @@ -18,6 +18,38 @@ This stage is what separates an OpenMontage video from generic AI slop. Without ## Process +### Step 0: Check for Reference Video Context + +Before starting research, check if a VideoAnalysisBrief exists for this project. If it +does, this is a reference-driven production — the user provided a video they want to +riff on. + +**When a VideoAnalysisBrief is present:** + +1. Read it thoroughly. Extract: + - `content_analysis.topics` — research these topics for accuracy + - `content_analysis.key_claims` — verify these claims via web search + - `style_profile` — note this for the proposal stage (do not research style) + - `replication_guidance.creative_differentiation_seeds` — these are your concept seeds + - `replication_guidance.key_elements_to_replicate` — preserve these in proposals + +2. Your research focus SHIFTS: + - Standard research: "What is interesting about this topic?" + - Reference-driven research: "What is interesting about this topic that the + reference video DIDN'T cover?" + "What would make our version DIFFERENT and BETTER?" + +3. In the research_brief, add a `reference_context` section: + - What the reference covered + - What it missed (your differentiation opportunity) + - What claims it made that you can verify or update + - How the landscape has changed since the reference was published + +4. The `angles_discovered` should explicitly position against the reference: + - "The reference took angle X. We could take angle Y which is [fresher/deeper/more + surprising] because [research finding]." + +**When no VideoAnalysisBrief is present:** Skip this step and proceed normally. + ### Step 1: Scope the Research Before searching anything, establish boundaries: diff --git a/tests/contracts/test_phase0_contracts.py b/tests/contracts/test_phase0_contracts.py index cd39355..ab64347 100644 --- a/tests/contracts/test_phase0_contracts.py +++ b/tests/contracts/test_phase0_contracts.py @@ -31,9 +31,11 @@ from lib.pipeline_loader import ( get_required_tools, get_stage_order, get_stage_skill, + get_stage_sub_stages, get_stage_review_focus, list_pipelines, load_pipeline, + pipeline_supports_reference_input, ) from tools.base_tool import BaseTool, ToolResult, ToolTier, ToolStatus from tools.tool_registry import ToolRegistry @@ -201,6 +203,49 @@ def sample_artifact(name: str) -> dict: } ], } + if name == "video_analysis_brief": + return { + "version": "1.0", + "source": { + "type": "youtube", + "url": "https://example.com/watch?v=abc123def45", + "title": "Reference Video", + "duration_seconds": 60, + }, + "content_analysis": { + "summary": "A fast explainer reference.", + "topics": ["quantum computing"], + "target_audience": "general", + }, + "structure_analysis": { + "total_scenes": 3, + "scenes": [ + { + "scene_index": 0, + "start_time": 0, + "end_time": 5, + "description": "Hook", + }, + { + "scene_index": 1, + "start_time": 5, + "end_time": 20, + "description": "Setup", + }, + { + "scene_index": 2, + "start_time": 20, + "end_time": 60, + "description": "Payoff", + }, + ], + "pacing_profile": { + "avg_scene_duration_seconds": 20, + "cuts_per_minute": 3, + "pacing_style": "steady_educational", + }, + }, + } raise KeyError(f"Unknown artifact sample: {name}") @@ -235,6 +280,9 @@ class TestSchemas: with pytest.raises(Exception): validate_artifact("brief", {"version": "1.0"}) + def test_video_analysis_brief_validates(self): + validate_artifact("video_analysis_brief", sample_artifact("video_analysis_brief")) + # ---- Checkpoint ---- @@ -289,6 +337,21 @@ class TestCheckpoint: {"research_brief": sample_artifact("research_brief")}, ) + def test_supplementary_video_analysis_brief_is_validated(self, tmp_path): + write_checkpoint( + tmp_path, + "proj", + "proposal", + "completed", + { + "proposal_packet": sample_artifact("proposal_packet"), + "video_analysis_brief": sample_artifact("video_analysis_brief"), + }, + ) + cp = read_checkpoint(tmp_path, "proj", "proposal") + assert cp is not None + assert "video_analysis_brief" in cp["artifacts"] + # ---- Pipeline manifests ---- @@ -302,6 +365,22 @@ class TestPipelineManifests: def test_framework_smoke_manifest_listed(self): assert "framework-smoke" in list_pipelines() + def test_reference_sub_stage_helpers(self): + manifest = load_pipeline("animated-explainer") + assert pipeline_supports_reference_input(manifest) is True + assert "video_analyzer" in get_required_tools(manifest) + + all_units = get_stage_order(manifest, include_sub_stages=True) + assert "proposal.sample" in all_units + + active_sub_stages = get_stage_sub_stages( + manifest, + "proposal", + context={"video_analysis_brief_exists": True}, + include_inactive=False, + ) + assert any(s["name"] == "sample" for s in active_sub_stages) + # ---- BaseTool ---- @@ -425,6 +504,34 @@ class TestCostTracker: t2 = CostTracker(cost_log_path=log_path) assert t2.budget_spent_usd == 0.08 + def test_reference_estimate_falls_back_when_scene_types_are_unclassified(self): + tracker = CostTracker(mode=BudgetMode.OBSERVE) + brief = { + "source": {"type": "shorts", "duration_seconds": 60}, + "structure_analysis": { + "total_scenes": 12, + "pacing_profile": {"pacing_style": "rapid_fire"}, + "scenes": [{"visual_type": "other"} for _ in range(12)], + }, + "narration_transcript": {"word_count": 180}, + "replication_guidance": {"motion_required": True, "suggested_pipeline": "animation"}, + } + plan = { + "video_generation": {"tool": "kling_fal", "cost_per_unit": 0.3, "clip_duration_seconds": 5}, + "image_generation": {"tool": "flux_fal", "cost_per_unit": 0.05}, + "tts": {"tool": "elevenlabs_tts", "cost_per_word": 0.00003}, + "music": {"tool": "music_gen", "cost_per_track": 0.1}, + } + + estimate = tracker.estimate_from_reference(brief, 60, plan) + + assert estimate["motion_ratio"] >= 0.6 + assert estimate["estimated_clips"] >= 7 + assert any( + "scene visual types have not been enriched yet" in note + for note in estimate["assumptions"] + ) + # ---- Pipeline Instruction Architecture ---- diff --git a/tools/analysis/frame_sampler.py b/tools/analysis/frame_sampler.py index 91f52e4..31ab9f8 100644 --- a/tools/analysis/frame_sampler.py +++ b/tools/analysis/frame_sampler.py @@ -44,6 +44,7 @@ class FrameSampler(BaseTool): "extract_frames_interval", "extract_frames_count", "extract_frames_timestamps", + "extract_frames_scene_guided", ] input_schema = { @@ -53,7 +54,7 @@ class FrameSampler(BaseTool): "input_path": {"type": "string"}, "strategy": { "type": "string", - "enum": ["interval", "count", "timestamps"], + "enum": ["interval", "count", "timestamps", "scene_guided"], }, "interval_seconds": { "type": "number", @@ -70,6 +71,23 @@ class FrameSampler(BaseTool): "items": {"type": "number"}, "description": "Specific timestamps in seconds (for timestamps strategy)", }, + "scene_boundaries": { + "type": "array", + "items": { + "type": "object", + "properties": { + "start_seconds": {"type": "number"}, + "end_seconds": {"type": "number"}, + }, + }, + "description": "Scene boundary list (for scene_guided strategy)", + }, + "max_frames": { + "type": "integer", + "minimum": 1, + "default": 20, + "description": "Max frames to extract (for scene_guided strategy)", + }, "output_dir": {"type": "string"}, "format": {"type": "string", "enum": ["png", "jpg"], "default": "jpg"}, "quality": {"type": "integer", "minimum": 1, "maximum": 31, "default": 2}, @@ -101,6 +119,8 @@ class FrameSampler(BaseTool): frames = self._extract_count(input_path, output_dir, fmt, quality, inputs) elif strategy == "timestamps": frames = self._extract_timestamps(input_path, output_dir, fmt, quality, inputs) + elif strategy == "scene_guided": + frames = self._extract_scene_guided(input_path, output_dir, fmt, quality, inputs) else: return ToolResult(success=False, error=f"Unknown strategy: {strategy}") except Exception as e: @@ -208,6 +228,54 @@ class FrameSampler(BaseTool): return frames + def _extract_scene_guided( + self, + input_path: Path, + output_dir: Path, + fmt: str, + quality: int, + inputs: dict, + ) -> list[dict]: + """Extract keyframes guided by scene boundaries. + + Extracts the first frame of each scene plus a midpoint frame for scenes + longer than 3 seconds. This captures all visual transitions with a + bounded, predictable number of frames — much better than uniform FPS. + """ + scene_boundaries = inputs.get("scene_boundaries", []) + max_frames = inputs.get("max_frames", 20) + + if not scene_boundaries: + # No scene data — fall back to count-based + return self._extract_count(input_path, output_dir, fmt, quality, { + "count": min(max_frames, 15), + }) + + # Compute timestamps: first frame + midpoint for long scenes + timestamps = [] + for scene in scene_boundaries: + start = scene.get("start_seconds", 0) + end = scene.get("end_seconds", 0) + duration = end - start + + # First frame of scene (offset slightly to avoid black frames) + timestamps.append(start + 0.1) + + # Midpoint for scenes > 3 seconds + if duration > 3.0: + timestamps.append(start + duration / 2) + + # Deduplicate, sort, limit + timestamps = sorted(set(round(t, 3) for t in timestamps)) + if len(timestamps) > max_frames: + step = len(timestamps) / max_frames + timestamps = [timestamps[int(i * step)] for i in range(max_frames)] + + # Extract via timestamps strategy + return self._extract_timestamps( + input_path, output_dir, fmt, quality, {"timestamps": timestamps} + ) + def _get_duration(self, input_path: Path) -> float: """Get video duration in seconds via ffprobe.""" cmd = [ diff --git a/tools/analysis/transcript_fetcher.py b/tools/analysis/transcript_fetcher.py new file mode 100644 index 0000000..d4868d9 --- /dev/null +++ b/tools/analysis/transcript_fetcher.py @@ -0,0 +1,216 @@ +"""YouTube transcript fetcher tool wrapping youtube-transcript-api. + +Extracts transcripts/captions from YouTube videos without downloading the video. +Instant, free, no API key needed. Falls back to yt-dlp subtitle download. +""" + +from __future__ import annotations + +import re +import time +from typing import Any + +from tools.base_tool import ( + BaseTool, + Determinism, + ExecutionMode, + ResourceProfile, + ToolResult, + ToolStability, + ToolStatus, + ToolTier, + ToolRuntime, +) + + +class TranscriptFetcher(BaseTool): + name = "transcript_fetcher" + version = "0.1.0" + tier = ToolTier.ANALYZE + capability = "analysis" + provider = "youtube-transcript-api" + stability = ToolStability.PRODUCTION + execution_mode = ExecutionMode.SYNC + determinism = Determinism.DETERMINISTIC + runtime = ToolRuntime.LOCAL + + dependencies = ["python:youtube_transcript_api"] + install_instructions = ( + "Install youtube-transcript-api: pip install youtube-transcript-api" + ) + agent_skills = [] + + capabilities = [ + "fetch_transcript", + "list_transcripts", + ] + + best_for = [ + "fast YouTube transcript extraction", + "caption-based analysis without video download", + "getting timestamped text from YouTube videos", + ] + + not_good_for = [ + "non-YouTube platforms (Instagram, TikTok)", + "videos without any captions", + "speaker diarization (use transcriber tool instead)", + ] + + input_schema = { + "type": "object", + "required": ["url_or_video_id"], + "properties": { + "url_or_video_id": { + "type": "string", + "description": "YouTube URL or video ID", + }, + "languages": { + "type": "array", + "items": {"type": "string"}, + "default": ["en"], + "description": "Preferred languages in priority order", + }, + "include_auto_generated": { + "type": "boolean", + "default": True, + "description": "Whether to include auto-generated captions", + }, + }, + } + + output_schema = { + "type": "object", + "properties": { + "transcript": { + "type": "array", + "items": { + "type": "object", + "properties": { + "text": {"type": "string"}, + "start": {"type": "number"}, + "duration": {"type": "number"}, + }, + }, + }, + "full_text": {"type": "string"}, + "language": {"type": "string"}, + "is_auto_generated": {"type": "boolean"}, + "word_count": {"type": "integer"}, + "source": {"type": "string"}, + }, + } + + resource_profile = ResourceProfile( + cpu_cores=1, ram_mb=256, vram_mb=0, disk_mb=10, + network_required=True, + ) + idempotency_key_fields = ["url_or_video_id", "languages"] + side_effects = [] + fallback = "transcriber" + user_visible_verification = [ + "Spot-check transcript accuracy against video audio", + ] + + def _extract_video_id(self, url_or_id: str) -> str: + """Extract YouTube video ID from URL or return as-is if already an ID.""" + # Already a bare ID (11 chars, alphanumeric + - _) + if re.match(r"^[A-Za-z0-9_-]{11}$", url_or_id): + return url_or_id + + # Standard YouTube URLs + patterns = [ + r"(?:youtube\.com/watch\?.*v=|youtu\.be/|youtube\.com/embed/|youtube\.com/shorts/)([A-Za-z0-9_-]{11})", + ] + for pattern in patterns: + match = re.search(pattern, url_or_id) + if match: + return match.group(1) + + # If nothing matched, try using the whole string as ID + return url_or_id.strip() + + def execute(self, inputs: dict[str, Any]) -> ToolResult: + video_id = self._extract_video_id(inputs["url_or_video_id"]) + languages = inputs.get("languages", ["en"]) + include_auto = inputs.get("include_auto_generated", True) + + start = time.time() + + try: + from youtube_transcript_api import YouTubeTranscriptApi + + ytt = YouTubeTranscriptApi() + + # Fetch transcript using the instance-based API (v1.0+) + transcript_result = ytt.fetch(video_id, languages=languages) + + # Build segments and full text from snippets + segments = [] + full_text_parts = [] + for snippet in transcript_result.snippets: + segments.append({ + "text": snippet.text, + "start": round(snippet.start, 3), + "duration": round(snippet.duration, 3), + }) + full_text_parts.append(snippet.text) + + full_text = " ".join(full_text_parts) + word_count = len(full_text.split()) + + # Get auto-generated status and language from the result + is_auto = getattr(transcript_result, "is_generated", False) + detected_lang = getattr(transcript_result, "language", languages[0]) + # If language is an object, get the code + if hasattr(detected_lang, "code"): + detected_lang = detected_lang.code + elif not isinstance(detected_lang, str): + detected_lang = languages[0] + + elapsed = time.time() - start + + return ToolResult( + success=True, + data={ + "transcript": segments, + "full_text": full_text, + "language": detected_lang, + "is_auto_generated": is_auto, + "word_count": word_count, + "source": "youtube_captions", + "video_id": video_id, + "segment_count": len(segments), + }, + duration_seconds=round(elapsed, 2), + ) + + except ImportError: + return ToolResult( + success=False, + error="youtube-transcript-api not installed. Run: pip install youtube-transcript-api", + ) + except Exception as e: + elapsed = time.time() - start + error_str = str(e) + + # Provide helpful error messages + if "Could not retrieve" in error_str or "TranscriptsDisabled" in error_str: + return ToolResult( + success=False, + error=( + f"No captions available for video {video_id}. " + "This video may not have captions enabled. " + "Fallback: download the video and use the transcriber tool " + "with Whisper for local transcription." + ), + data={"video_id": video_id, "fallback_suggested": "transcriber"}, + duration_seconds=round(elapsed, 2), + ) + + return ToolResult( + success=False, + error=f"Transcript fetch failed: {error_str}", + data={"video_id": video_id}, + duration_seconds=round(elapsed, 2), + ) diff --git a/tools/analysis/video_analyzer.py b/tools/analysis/video_analyzer.py new file mode 100644 index 0000000..26efd35 --- /dev/null +++ b/tools/analysis/video_analyzer.py @@ -0,0 +1,678 @@ +"""Video analyzer tool — comprehensive reference video analysis. + +Orchestrates multiple analysis tools to produce a VideoAnalysisBrief from a +video URL or local file. Runs entirely locally with zero API keys: yt-dlp for +download, youtube-transcript-api for captions, PySceneDetect/FFmpeg for scene +detection, FFmpeg for frame extraction, and faster-whisper for transcription. + +The agent's own vision model analyzes extracted keyframes — this tool provides +the structured data; the agent provides the visual interpretation. +""" + +from __future__ import annotations + +import json +import time +from pathlib import Path +from typing import Any + +from tools.base_tool import ( + BaseTool, + Determinism, + ExecutionMode, + ResourceProfile, + ToolResult, + ToolStability, + ToolStatus, + ToolTier, + ToolRuntime, +) + + +class VideoAnalyzer(BaseTool): + name = "video_analyzer" + version = "0.1.0" + tier = ToolTier.ANALYZE + capability = "analysis" + provider = "multi" + stability = ToolStability.BETA + execution_mode = ExecutionMode.SYNC + determinism = Determinism.DETERMINISTIC + runtime = ToolRuntime.LOCAL + + dependencies = ["cmd:ffmpeg"] + install_instructions = ( + "Core: FFmpeg is required (https://ffmpeg.org/download.html)\n" + "For URL downloads: pip install yt-dlp\n" + "For YouTube transcripts: pip install youtube-transcript-api\n" + "For local transcription: pip install faster-whisper\n" + "For scene detection: pip install scenedetect[opencv]\n" + "All dependencies are free and local — no API keys needed." + ) + agent_skills = ["video-understand", "ffmpeg"] + + capabilities = [ + "analyze_reference_video", + "extract_structure", + "extract_style", + "extract_transcript", + ] + + best_for = [ + "comprehensive video analysis", + "reference video understanding", + "style extraction from example video", + "understanding video structure and pacing", + ] + + not_good_for = [ + "editing or modifying video", + "generating new video content", + ] + + input_schema = { + "type": "object", + "required": ["source"], + "properties": { + "source": { + "type": "string", + "description": "Video file path or URL (YouTube, Shorts, Instagram, TikTok)", + }, + "analysis_depth": { + "type": "string", + "enum": ["transcript_only", "standard", "deep"], + "default": "standard", + "description": ( + "transcript_only: transcript + metadata only. " + "standard: + scene detection + keyframes + audio energy. " + "deep: + intra-scene sampling + detailed style extraction." + ), + }, + "max_keyframes": { + "type": "integer", + "default": 20, + "minimum": 1, + "maximum": 50, + "description": "Maximum keyframes to extract", + }, + "output_dir": { + "type": "string", + "description": "Directory for analysis outputs (default: auto-generated)", + }, + }, + } + + output_schema = { + "type": "object", + "description": "VideoAnalysisBrief artifact — see schemas/artifacts/video_analysis_brief.schema.json", + } + + resource_profile = ResourceProfile( + cpu_cores=2, ram_mb=2048, vram_mb=0, disk_mb=3000, + network_required=False, # Only needed for URL sources + ) + idempotency_key_fields = ["source", "analysis_depth"] + side_effects = [ + "downloads video to output_dir (if URL)", + "writes keyframe images to output_dir/keyframes/", + "writes analysis JSON to output_dir/video_analysis_brief.json", + ] + fallback_tools = [] + user_visible_verification = [ + "Review keyframe images for representative coverage", + "Check transcript accuracy against video", + "Verify scene boundaries look correct", + ] + + def _is_url(self, source: str) -> bool: + """Check if source is a URL vs local file.""" + return source.startswith(("http://", "https://", "www.")) + + def _detect_platform(self, source: str) -> str: + """Detect platform from URL.""" + if not self._is_url(source): + return "local_file" + s = source.lower() + if "youtube.com/shorts" in s: + return "shorts" + if "youtube.com" in s or "youtu.be" in s: + return "youtube" + if "instagram.com" in s: + return "instagram" + if "tiktok.com" in s: + return "tiktok" + return "other_url" + + def _is_youtube(self, platform: str) -> bool: + return platform in ("youtube", "shorts") + + def execute(self, inputs: dict[str, Any]) -> ToolResult: + source = inputs["source"] + depth = inputs.get("analysis_depth", "standard") + max_keyframes = inputs.get("max_keyframes", 20) + + # Setup output directory + if inputs.get("output_dir"): + output_dir = Path(inputs["output_dir"]) + else: + output_dir = Path("projects/_analysis") / f"analysis_{int(time.time())}" + output_dir.mkdir(parents=True, exist_ok=True) + + platform = self._detect_platform(source) + is_url = self._is_url(source) + start = time.time() + + # Initialize brief structure + brief = { + "version": "1.0", + "source": { + "type": platform, + "duration_seconds": 0, + }, + "content_analysis": { + "summary": "", + "topics": [], + "target_audience": "general", + }, + "structure_analysis": { + "total_scenes": 0, + "scenes": [], + "pacing_profile": {}, + }, + } + + if is_url: + brief["source"]["url"] = source + else: + brief["source"]["local_path"] = source + + # Track what succeeded and what failed + steps_completed = [] + steps_failed = [] + + # ─── STEP 1: Get metadata + download (if URL) ─── + video_path = None + audio_path = None + metadata = {} + + if is_url: + try: + from tools.analysis.video_downloader import VideoDownloader + downloader = VideoDownloader() + + if depth == "transcript_only" and self._is_youtube(platform): + # Only get metadata, skip video download + dl_result = downloader.execute({ + "url": source, + "output_dir": str(output_dir), + "format": "metadata_only", + }) + else: + dl_result = downloader.execute({ + "url": source, + "output_dir": str(output_dir), + "format": "video", + "max_resolution": "720p", + }) + + if dl_result.success: + metadata = dl_result.data.get("metadata", {}) + video_path = dl_result.data.get("video_path") + audio_path = dl_result.data.get("audio_path") + brief["source"]["title"] = metadata.get("title", "") + brief["source"]["duration_seconds"] = metadata.get("duration", 0) + brief["source"]["resolution"] = metadata.get("resolution", "") + brief["source"]["platform_metadata"] = { + "uploader": metadata.get("uploader", ""), + "upload_date": metadata.get("upload_date", ""), + "view_count": metadata.get("view_count", 0), + "like_count": metadata.get("like_count", 0), + "description": metadata.get("description", ""), + } + steps_completed.append("metadata") + if video_path: + steps_completed.append("download") + else: + steps_failed.append(f"download: {dl_result.error}") + except Exception as e: + steps_failed.append(f"download: {e}") + else: + # Local file + local_path = Path(source) + if not local_path.exists(): + return ToolResult( + success=False, + error=f"Local file not found: {source}", + ) + video_path = str(local_path) + # Get duration via ffprobe + try: + duration = self._get_duration(local_path) + brief["source"]["duration_seconds"] = duration + brief["source"]["title"] = local_path.stem + steps_completed.append("metadata") + except Exception as e: + steps_failed.append(f"metadata: {e}") + + # ─── STEP 2: Get transcript ─── + transcript_data = None + + # Try youtube-transcript-api first (instant, for YouTube) + if self._is_youtube(platform): + try: + from youtube_transcript_api import YouTubeTranscriptApi + + from tools.analysis.transcript_fetcher import TranscriptFetcher + fetcher = TranscriptFetcher() + + # Auto-detect available languages instead of hardcoding "en" + languages_to_try = ["en"] + try: + ytt = YouTubeTranscriptApi() + available = ytt.list(fetcher._extract_video_id(source)) + # Build priority list: manual first, then auto-generated + lang_codes = [] + for t in available: + code = t.language_code if hasattr(t, "language_code") else str(t) + if code not in lang_codes: + lang_codes.append(code) + if lang_codes: + languages_to_try = lang_codes + except Exception: + pass # Fall through to default ["en"] + + tf_result = fetcher.execute({ + "url_or_video_id": source, + "languages": languages_to_try, + "include_auto_generated": True, + }) + if tf_result.success: + transcript_data = tf_result.data + brief["narration_transcript"] = { + "full_text": transcript_data.get("full_text", ""), + "segments": transcript_data.get("transcript", []), + "language": transcript_data.get("language", "en"), + "word_count": transcript_data.get("word_count", 0), + } + steps_completed.append("transcript_youtube") + except Exception as e: + steps_failed.append(f"transcript_youtube: {e}") + + # Fallback: If transcript failed and we don't have audio yet, + # download the video to get audio for Whisper transcription + if transcript_data is None and audio_path is None and video_path is None and is_url: + try: + from tools.analysis.video_downloader import VideoDownloader + downloader = VideoDownloader() + dl_result = downloader.execute({ + "url": source, + "output_dir": str(output_dir), + "format": "video", + "max_resolution": "720p", + }) + if dl_result.success: + video_path = dl_result.data.get("video_path") + audio_path = dl_result.data.get("audio_path") + if video_path: + steps_completed.append("download_for_whisper") + # Also update metadata if we didn't have it + if not metadata: + metadata = dl_result.data.get("metadata", {}) + brief["source"]["title"] = metadata.get("title", "") + brief["source"]["duration_seconds"] = metadata.get("duration", 0) + except Exception as e: + steps_failed.append(f"download_for_whisper: {e}") + + # Fallback: Whisper transcription on audio + if transcript_data is None and audio_path: + try: + from tools.analysis.transcriber import Transcriber + transcriber = Transcriber() + # Let Whisper auto-detect language instead of assuming English + tr_inputs = { + "input_path": audio_path, + "model_size": "base", + "output_dir": str(output_dir), + } + # Only set language if we know it from transcript attempt + detected_lang = brief.get("narration_transcript", {}).get("language") + if detected_lang and detected_lang != "en": + tr_inputs["language"] = detected_lang + # else: let Whisper auto-detect + + tr_result = transcriber.execute(tr_inputs) + if tr_result.success: + segments = tr_result.data.get("segments", []) + full_text = " ".join(s.get("text", "") for s in segments) + brief["narration_transcript"] = { + "full_text": full_text, + "segments": [ + { + "start": s.get("start", 0), + "end": s.get("end", 0), + "text": s.get("text", ""), + } + for s in segments + ], + "language": tr_result.data.get("language", "en"), + "word_count": len(full_text.split()), + } + transcript_data = brief["narration_transcript"] + steps_completed.append("transcript_whisper") + except Exception as e: + steps_failed.append(f"transcript_whisper: {e}") + + # For transcript_only depth, we're done + if depth == "transcript_only": + brief["_analysis_meta"] = { + "depth": depth, + "steps_completed": steps_completed, + "steps_failed": steps_failed, + "duration_seconds": round(time.time() - start, 2), + } + self._save_brief(brief, output_dir) + return ToolResult( + success=True, + data=brief, + artifacts=[str(output_dir / "video_analysis_brief.json")], + duration_seconds=round(time.time() - start, 2), + ) + + # ─── STEP 3: Scene detection (standard + deep) ─── + scenes = [] + if video_path: + try: + from tools.analysis.scene_detect import SceneDetect + detector = SceneDetect() + sd_result = detector.execute({ + "input_path": video_path, + "method": "content", + "min_scene_length_seconds": 0.5, + "output_path": str(output_dir / "scenes.json"), + }) + if sd_result.success: + scenes = sd_result.data.get("scenes", []) + steps_completed.append("scene_detect") + except Exception as e: + steps_failed.append(f"scene_detect: {e}") + + # Build scene list for the brief + if scenes: + brief["structure_analysis"]["total_scenes"] = len(scenes) + brief_scenes = [] + for scene in scenes: + brief_scenes.append({ + "scene_index": scene.get("index", scene.get("scene_index", 0)), + "start_time": scene.get("start_seconds", 0), + "end_time": scene.get("end_seconds", 0), + "description": "", # Agent fills this via vision + "visual_type": "other", # Agent classifies via vision + "energy_level": "medium", + }) + brief["structure_analysis"]["scenes"] = brief_scenes + + # Compute pacing profile + durations = [ + s.get("end_seconds", 0) - s.get("start_seconds", 0) + for s in scenes + ] + total_duration = brief["source"]["duration_seconds"] or sum(durations) + if durations: + brief["structure_analysis"]["pacing_profile"] = { + "avg_scene_duration_seconds": round(sum(durations) / len(durations), 2), + "shortest_scene_seconds": round(min(durations), 2), + "longest_scene_seconds": round(max(durations), 2), + "cuts_per_minute": round(len(durations) / (total_duration / 60), 2) if total_duration > 0 else 0, + "pacing_style": self._classify_pacing(durations), + } + + # ─── STEP 4: Keyframe extraction (scene-guided) ─── + keyframes = [] + keyframe_dir = output_dir / "keyframes" + if video_path and scenes: + try: + # Extract keyframes at scene boundaries + midpoints + timestamps = self._compute_keyframe_timestamps(scenes, max_keyframes, depth) + + from tools.analysis.frame_sampler import FrameSampler + sampler = FrameSampler() + fs_result = sampler.execute({ + "input_path": video_path, + "strategy": "timestamps", + "timestamps": timestamps, + "output_dir": str(keyframe_dir), + "format": "jpg", + "quality": 2, + }) + if fs_result.success: + for frame in fs_result.data.get("frames", []): + # Map each frame to its scene + scene_idx = self._timestamp_to_scene( + frame["timestamp_seconds"], scenes + ) + keyframes.append({ + "timestamp": frame["timestamp_seconds"], + "scene_index": scene_idx, + "path": frame["path"], + "description": "", # Agent fills via vision + }) + steps_completed.append("keyframes") + except Exception as e: + steps_failed.append(f"keyframes: {e}") + elif video_path and not scenes: + # No scene detection — fall back to count-based extraction + try: + from tools.analysis.frame_sampler import FrameSampler + sampler = FrameSampler() + fs_result = sampler.execute({ + "input_path": video_path, + "strategy": "count", + "count": min(max_keyframes, 15), + "output_dir": str(keyframe_dir), + "format": "jpg", + "quality": 2, + }) + if fs_result.success: + for frame in fs_result.data.get("frames", []): + keyframes.append({ + "timestamp": frame["timestamp_seconds"], + "scene_index": 0, + "path": frame["path"], + "description": "", + }) + steps_completed.append("keyframes_uniform") + except Exception as e: + steps_failed.append(f"keyframes_uniform: {e}") + + brief["keyframes"] = keyframes + + # ─── STEP 5: Audio energy analysis ─── + if audio_path or video_path: + audio_source = audio_path or video_path + try: + from tools.analysis.audio_energy import AudioEnergy + energy = AudioEnergy() + ae_result = energy.execute({ + "input_path": audio_source, + "video_duration_seconds": brief["source"]["duration_seconds"], + }) + if ae_result.success: + # Store energy profile summary in style_profile + if "style_profile" not in brief: + brief["style_profile"] = {} + brief["style_profile"]["audio_energy_profile"] = { + "recommended_offset": ae_result.data.get("recommended_offset_seconds", 0), + "has_energy_data": True, + } + steps_completed.append("audio_energy") + except Exception as e: + steps_failed.append(f"audio_energy: {e}") + + # ─── STEP 6: Build replication guidance ─── + brief["replication_guidance"] = { + "suggested_pipeline": self._suggest_pipeline(brief), + "suggested_playbook": "flat-motion-graphics", + "key_elements_to_replicate": [], # Agent fills via analysis + "elements_requiring_custom_work": [], + "estimated_complexity": self._estimate_complexity(brief), + "motion_required": self._needs_motion(brief), + "creative_differentiation_seeds": [], # Agent fills + } + + # ─── STEP 7: Initialize style_profile ─── + if "style_profile" not in brief: + brief["style_profile"] = {} + + # Narration style from transcript + if transcript_data: + duration = brief["source"]["duration_seconds"] + wc = transcript_data.get("word_count", 0) if isinstance(transcript_data, dict) else brief.get("narration_transcript", {}).get("word_count", 0) + wpm = round(wc / (duration / 60), 1) if duration > 0 else 0 + brief["style_profile"]["narration_style"] = { + "has_narration": wc > 20, + "speaker_count": 1, # Agent refines via analysis + "delivery_style": "", # Agent fills + "words_per_minute": wpm, + } + + # Initialize remaining style fields for agent to fill + brief["style_profile"].setdefault("color_palette", { + "primary_colors": [], + "accent_colors": [], + "overall_mood": "", + }) + brief["style_profile"].setdefault("typography_observed", "") + brief["style_profile"].setdefault("transition_types", []) + brief["style_profile"].setdefault("music_style", "") + brief["style_profile"].setdefault("subtitle_style", "") + brief["style_profile"].setdefault("production_quality", "prosumer") + brief["style_profile"].setdefault("closest_playbook", "") + brief["style_profile"].setdefault("playbook_delta", "") + + # ─── Finalize ─── + brief["_analysis_meta"] = { + "depth": depth, + "steps_completed": steps_completed, + "steps_failed": steps_failed, + "keyframe_count": len(keyframes), + "scene_count": len(scenes), + "has_transcript": transcript_data is not None, + "duration_seconds": round(time.time() - start, 2), + } + + self._save_brief(brief, output_dir) + + elapsed = time.time() - start + artifacts = [str(output_dir / "video_analysis_brief.json")] + if keyframe_dir.exists(): + artifacts.append(str(keyframe_dir)) + + return ToolResult( + success=True, + data=brief, + artifacts=artifacts, + duration_seconds=round(elapsed, 2), + ) + + # ─── Helpers ─── + + def _get_duration(self, video_path: Path) -> float: + """Get video duration via ffprobe.""" + cmd = [ + "ffprobe", "-v", "quiet", + "-show_entries", "format=duration", + "-of", "json", + str(video_path), + ] + result = self.run_command(cmd) + data = json.loads(result.stdout) + return float(data.get("format", {}).get("duration", 0)) + + def _compute_keyframe_timestamps( + self, scenes: list[dict], max_frames: int, depth: str + ) -> list[float]: + """Compute optimal keyframe timestamps from scene boundaries.""" + timestamps = [] + + for scene in scenes: + start = scene.get("start_seconds", 0) + end = scene.get("end_seconds", 0) + duration = end - start + + # First frame of each scene + timestamps.append(start + 0.1) + + # Midpoint for scenes > 3 seconds + if duration > 3.0: + timestamps.append(start + duration / 2) + + # For deep analysis, add more intra-scene samples + if depth == "deep" and duration > 6.0: + timestamps.append(start + duration * 0.25) + timestamps.append(start + duration * 0.75) + + # Deduplicate, sort, and limit + timestamps = sorted(set(round(t, 3) for t in timestamps)) + if len(timestamps) > max_frames: + # Uniform subsample to max_frames + step = len(timestamps) / max_frames + timestamps = [timestamps[int(i * step)] for i in range(max_frames)] + + return timestamps + + def _timestamp_to_scene(self, ts: float, scenes: list[dict]) -> int: + """Map a timestamp to its scene index.""" + for scene in scenes: + start = scene.get("start_seconds", 0) + end = scene.get("end_seconds", 0) + if start <= ts <= end: + return scene.get("index", scene.get("scene_index", 0)) + return 0 + + def _classify_pacing(self, durations: list[float]) -> str: + """Classify pacing style from scene durations.""" + if not durations: + return "variable" + avg = sum(durations) / len(durations) + if avg > 10: + return "slow_contemplative" + if avg > 5: + return "steady_educational" + if avg > 2: + return "dynamic_social" + return "rapid_fire" + + def _suggest_pipeline(self, brief: dict) -> str: + """Suggest the best pipeline based on content analysis.""" + platform = brief["source"]["type"] + pacing = brief["structure_analysis"].get("pacing_profile", {}).get("pacing_style", "") + + if platform in ("shorts", "tiktok", "instagram"): + return "animation" # Short-form → animation pipeline works well + if pacing in ("slow_contemplative",): + return "cinematic" + return "animated-explainer" + + def _estimate_complexity(self, brief: dict) -> str: + """Estimate how complex it would be to recreate this style.""" + scenes = brief["structure_analysis"]["total_scenes"] + duration = brief["source"]["duration_seconds"] + + if duration > 300 or scenes > 30: + return "complex" + if duration > 120 or scenes > 15: + return "moderate" + return "simple" + + def _needs_motion(self, brief: dict) -> bool: + """Determine if motion (video gen or Remotion) is required.""" + pacing = brief["structure_analysis"].get("pacing_profile", {}).get("pacing_style", "") + return pacing in ("dynamic_social", "rapid_fire") + + def _save_brief(self, brief: dict, output_dir: Path) -> None: + """Save the VideoAnalysisBrief to disk.""" + out_path = output_dir / "video_analysis_brief.json" + # Remove non-serializable items + clean_brief = {k: v for k, v in brief.items()} + with open(out_path, "w", encoding="utf-8") as f: + json.dump(clean_brief, f, indent=2, default=str) diff --git a/tools/analysis/video_downloader.py b/tools/analysis/video_downloader.py new file mode 100644 index 0000000..2a983aa --- /dev/null +++ b/tools/analysis/video_downloader.py @@ -0,0 +1,355 @@ +"""Video downloader tool wrapping yt-dlp. + +Downloads video, audio, or subtitles from YouTube, Shorts, Instagram Reels, +TikTok, and 1000+ other sites. Designed for reference video analysis — downloads +at analysis quality (720p), not production quality. +""" + +from __future__ import annotations + +import json +import re +import time +from pathlib import Path +from typing import Any + +from tools.base_tool import ( + BaseTool, + Determinism, + ExecutionMode, + ResourceProfile, + ToolResult, + ToolStability, + ToolStatus, + ToolTier, + ToolRuntime, +) + + +class VideoDownloader(BaseTool): + name = "video_downloader" + version = "0.1.0" + tier = ToolTier.SOURCE + capability = "source_ingest" + provider = "yt-dlp" + stability = ToolStability.PRODUCTION + execution_mode = ExecutionMode.SYNC + determinism = Determinism.DETERMINISTIC + runtime = ToolRuntime.LOCAL + + dependencies = ["python:yt_dlp"] + install_instructions = ( + "Install yt-dlp: pip install yt-dlp\n" + "For YouTube support, also install Deno (JS runtime): " + "https://deno.land/#installation\n" + "Without Deno, YouTube downloads may fail but other platforms still work." + ) + agent_skills = ["video-download"] + + capabilities = [ + "download_video", + "download_audio", + "download_subtitles", + "extract_metadata", + ] + + best_for = [ + "downloading reference video from URL", + "extracting audio from online video", + "downloading subtitles from YouTube", + "getting video metadata without downloading", + ] + + not_good_for = [ + "downloading entire playlists", + "downloading DRM-protected content", + ] + + input_schema = { + "type": "object", + "required": ["url", "output_dir"], + "properties": { + "url": {"type": "string", "description": "Video URL to download"}, + "output_dir": {"type": "string", "description": "Directory for downloaded files"}, + "format": { + "type": "string", + "enum": ["video", "audio_only", "subtitles_only", "metadata_only"], + "default": "video", + "description": "What to download", + }, + "max_resolution": { + "type": "string", + "enum": ["360p", "480p", "720p", "1080p"], + "default": "720p", + "description": "Maximum video resolution (for analysis, 720p is sufficient)", + }, + "max_duration_seconds": { + "type": "integer", + "default": 600, + "description": "Reject videos longer than this (safety limit)", + }, + }, + } + + output_schema = { + "type": "object", + "properties": { + "video_path": {"type": ["string", "null"]}, + "audio_path": {"type": ["string", "null"]}, + "subtitle_path": {"type": ["string", "null"]}, + "metadata": { + "type": "object", + "properties": { + "title": {"type": "string"}, + "duration": {"type": "number"}, + "uploader": {"type": "string"}, + "upload_date": {"type": "string"}, + "description": {"type": "string"}, + "view_count": {"type": "integer"}, + "like_count": {"type": "integer"}, + }, + }, + "platform": {"type": "string"}, + }, + } + + resource_profile = ResourceProfile( + cpu_cores=1, ram_mb=512, vram_mb=0, disk_mb=2000, + network_required=True, + ) + idempotency_key_fields = ["url", "format", "max_resolution"] + side_effects = ["downloads media files to output_dir"] + resume_support_value = "from_start" + user_visible_verification = [ + "Check downloaded file plays correctly", + "Verify resolution matches requested max", + ] + + # --- Resolution mapping --- + _RES_MAP = { + "360p": 360, + "480p": 480, + "720p": 720, + "1080p": 1080, + } + + def _detect_platform(self, url: str) -> str: + """Detect platform from URL.""" + url_lower = url.lower() + if "youtube.com/shorts" in url_lower or "youtu.be" in url_lower and "/shorts" in url_lower: + return "shorts" + if "youtube.com" in url_lower or "youtu.be" in url_lower: + return "youtube" + if "instagram.com" in url_lower: + return "instagram" + if "tiktok.com" in url_lower: + return "tiktok" + if "vimeo.com" in url_lower: + return "vimeo" + if "twitter.com" in url_lower or "x.com" in url_lower: + return "twitter" + return "other_url" + + def _extract_metadata(self, url: str) -> dict: + """Extract metadata without downloading.""" + import yt_dlp + + ydl_opts = { + "quiet": True, + "no_warnings": True, + "skip_download": True, + } + try: + with yt_dlp.YoutubeDL(ydl_opts) as ydl: + info = ydl.extract_info(url, download=False) + if info is None: + return {"error": "No info extracted", "title": "", "duration": 0} + return { + "title": info.get("title", ""), + "duration": info.get("duration", 0), + "uploader": info.get("uploader", info.get("channel", "")), + "upload_date": info.get("upload_date", ""), + "description": (info.get("description", "") or "")[:500], + "view_count": info.get("view_count", 0), + "like_count": info.get("like_count", 0), + "resolution": f"{info.get('width', 0)}x{info.get('height', 0)}", + "fps": info.get("fps", 0), + } + except Exception as e: + return {"error": str(e), "title": "", "duration": 0} + + def execute(self, inputs: dict[str, Any]) -> ToolResult: + url = inputs["url"] + output_dir = Path(inputs["output_dir"]) + dl_format = inputs.get("format", "video") + max_res = inputs.get("max_resolution", "720p") + max_duration = inputs.get("max_duration_seconds", 600) + + output_dir.mkdir(parents=True, exist_ok=True) + platform = self._detect_platform(url) + start = time.time() + + # Step 1: Always get metadata first + metadata = self._extract_metadata(url) + + # Check duration limit + duration = metadata.get("duration", 0) + if duration and duration > max_duration: + return ToolResult( + success=False, + error=( + f"Video is {duration}s, exceeds max_duration_seconds={max_duration}. " + f"Increase the limit or use a shorter video." + ), + data={"metadata": metadata, "platform": platform}, + ) + + if dl_format == "metadata_only": + return ToolResult( + success=True, + data={ + "video_path": None, + "audio_path": None, + "subtitle_path": None, + "metadata": metadata, + "platform": platform, + }, + duration_seconds=round(time.time() - start, 2), + ) + + video_path = None + audio_path = None + subtitle_path = None + + try: + if dl_format == "video": + video_path, audio_path = self._download_video( + url, output_dir, max_res + ) + elif dl_format == "audio_only": + audio_path = self._download_audio(url, output_dir) + elif dl_format == "subtitles_only": + subtitle_path = self._download_subtitles(url, output_dir) + except Exception as e: + elapsed = time.time() - start + return ToolResult( + success=False, + error=f"Download failed: {e}", + data={"metadata": metadata, "platform": platform}, + duration_seconds=round(elapsed, 2), + ) + + elapsed = time.time() - start + artifacts = [p for p in [video_path, audio_path, subtitle_path] if p] + + return ToolResult( + success=True, + data={ + "video_path": video_path, + "audio_path": audio_path, + "subtitle_path": subtitle_path, + "metadata": metadata, + "platform": platform, + }, + artifacts=artifacts, + duration_seconds=round(elapsed, 2), + ) + + def _download_video( + self, url: str, output_dir: Path, max_res: str + ) -> tuple[str | None, str | None]: + """Download video + extract audio track.""" + import yt_dlp + + height = self._RES_MAP.get(max_res, 720) + video_out = str(output_dir / "reference_video.%(ext)s") + + ydl_opts = { + "format": f"bestvideo[height<={height}]+bestaudio/best[height<={height}]/best", + "merge_output_format": "mp4", + "outtmpl": video_out, + "noplaylist": True, + "quiet": True, + "no_warnings": True, + } + with yt_dlp.YoutubeDL(ydl_opts) as ydl: + ydl.download([url]) + + # Find the downloaded video file + video_path = self._find_downloaded(output_dir, "reference_video", ["mp4", "mkv", "webm"]) + + # Extract audio separately for transcription + audio_path = None + if video_path: + audio_out = output_dir / "reference_audio.wav" + try: + audio_cmd = [ + "ffmpeg", "-y", + "-i", video_path, + "-vn", + "-acodec", "pcm_s16le", + "-ar", "16000", + "-ac", "1", + str(audio_out), + ] + self.run_command(audio_cmd, timeout=120) + if audio_out.exists(): + audio_path = str(audio_out) + except Exception: + pass # Audio extraction is optional + + return video_path, audio_path + + def _download_audio(self, url: str, output_dir: Path) -> str | None: + """Download audio only.""" + import yt_dlp + + audio_out = str(output_dir / "reference_audio.%(ext)s") + ydl_opts = { + "format": "bestaudio/best", + "postprocessors": [{ + "key": "FFmpegExtractAudio", + "preferredcodec": "wav", + "preferredquality": "0", + }], + "outtmpl": audio_out, + "noplaylist": True, + "quiet": True, + "no_warnings": True, + } + with yt_dlp.YoutubeDL(ydl_opts) as ydl: + ydl.download([url]) + return self._find_downloaded(output_dir, "reference_audio", ["wav", "mp3", "m4a", "opus"]) + + def _download_subtitles(self, url: str, output_dir: Path) -> str | None: + """Download subtitles only.""" + import yt_dlp + + sub_out = str(output_dir / "reference_subs.%(ext)s") + ydl_opts = { + "writesubtitles": True, + "writeautomaticsub": True, + "subtitleslangs": ["en"], + "subtitlesformat": "srt", + "skip_download": True, + "outtmpl": sub_out, + "noplaylist": True, + "quiet": True, + "no_warnings": True, + } + try: + with yt_dlp.YoutubeDL(ydl_opts) as ydl: + ydl.download([url]) + except Exception: + pass + return self._find_downloaded(output_dir, "reference_subs", ["srt", "vtt", "ass"]) + + def _find_downloaded( + self, output_dir: Path, prefix: str, extensions: list[str] + ) -> str | None: + """Find a downloaded file by prefix and possible extensions.""" + for ext in extensions: + candidates = list(output_dir.glob(f"{prefix}*.{ext}")) + if candidates: + return str(candidates[0]) + return None diff --git a/tools/cost_tracker.py b/tools/cost_tracker.py index 71c2c71..ee324ef 100644 --- a/tools/cost_tracker.py +++ b/tools/cost_tracker.py @@ -173,6 +173,310 @@ class CostTracker: entry["timestamp"] = self._now() self._save() + # ---- Reference-driven estimation ---- + + def estimate_from_reference( + self, + video_analysis_brief: dict, + target_duration_seconds: int, + tool_plan: dict, + ) -> dict: + """Estimate production cost based on reference analysis + target duration. + + Args: + video_analysis_brief: The VideoAnalysisBrief artifact from video analysis + target_duration_seconds: How long the output video should be + tool_plan: Which tools will be used for each asset type, e.g.: + { + "image_generation": {"tool": "flux_fal", "cost_per_unit": 0.05}, + "video_generation": {"tool": "kling_fal", "cost_per_unit": 0.30, + "clip_duration_seconds": 5}, + "tts": {"tool": "elevenlabs_tts", "cost_per_word": 0.00003}, + "music": {"tool": "music_gen", "cost_per_track": 0.10}, + } + + Returns: + Itemized cost breakdown with line items, total, sample cost, and assumptions. + """ + structure = video_analysis_brief.get("structure_analysis", {}) + pacing = structure.get("pacing_profile", {}) + narration = video_analysis_brief.get("narration_transcript", {}) + ref_duration = video_analysis_brief.get("source", {}).get("duration_seconds", 60) + pacing_style = pacing.get("pacing_style", "steady_educational") + + # ── Scene count estimation ── + # Don't just scale linearly — use the PACING DENSITY from the reference. + # A music video with 8 scenes in 162s has ~3 cuts/min. + # Scaling to 60s should PRESERVE that cut rate, not reduce scene count. + ref_scenes = structure.get("total_scenes", 8) + if ref_duration > 0: + cuts_per_minute = ref_scenes / (ref_duration / 60) + else: + cuts_per_minute = 4.0 # default: moderate pacing + + # Apply pacing-aware minimums (a fast-cut video doesn't become a slideshow) + min_scenes_by_pacing = { + "rapid_fire": 10, + "dynamic_social": 8, + "steady_educational": 5, + "slow_contemplative": 3, + "variable": 6, + } + min_scenes = min_scenes_by_pacing.get(pacing_style, 5) + + # Scene count = max(pacing-density-based, minimum for style) + density_based_scenes = round(cuts_per_minute * (target_duration_seconds / 60)) + estimated_scenes = max(min_scenes, density_based_scenes) + + # ── Narration word count ── + ref_word_count = narration.get("word_count", 0) + if ref_duration > 0 and ref_word_count > 0: + actual_wpm = (ref_word_count / ref_duration) * 60 + else: + actual_wpm = 150 # default conversational pace + estimated_words = round(actual_wpm * (target_duration_seconds / 60)) + + # ── Motion ratio from reference ── + scenes_list = structure.get("scenes", []) + motion_ratio, motion_basis = self._estimate_motion_ratio( + video_analysis_brief=video_analysis_brief, + scenes_list=scenes_list, + pacing_style=pacing_style, + ) + + estimated_motion_scenes = ( + max(1, round(estimated_scenes * motion_ratio)) + if motion_ratio > 0 + else 0 + ) + estimated_still_scenes = estimated_scenes - estimated_motion_scenes + + # ── Video clip coverage ── + # Video gen tools produce clips of limited duration (typically 5-10s). + # A 60s video with motion needs enough clips to COVER the duration, + # not just 1 per scene. + vid_plan = tool_plan.get("video_generation", {}) + clip_duration = vid_plan.get("clip_duration_seconds", 5) if vid_plan else 5 + motion_seconds = target_duration_seconds * motion_ratio + clips_needed_for_coverage = max( + estimated_motion_scenes, + round(motion_seconds / clip_duration) + ) if vid_plan else 0 + + # ── Retry/waste buffer ── + # Not every generation succeeds or looks good. Add a buffer. + retry_multiplier = 1.3 # ~30% extra for retries and rejected outputs + + # ── Image count ── + # Images per scene depends on visual variety needs: + # - Explainer: 1-2 images per scene + # - Music video / cinematic: 2-3 images per scene (mood shifts, variety) + images_per_scene = 2.0 if pacing_style in ("dynamic_social", "rapid_fire") else 1.5 + estimated_images = max( + estimated_scenes, + round(estimated_scenes * images_per_scene) + ) + + # Build line items + line_items = [] + assumptions = [] + + assumptions.append( + f"{estimated_scenes} scenes (reference has {cuts_per_minute:.1f} cuts/min, " + f"pacing: {pacing_style})" + ) + assumptions.append(motion_basis) + + # Image generation + img_plan = tool_plan.get("image_generation", {}) + if img_plan: + img_count = round(estimated_images * retry_multiplier) + unit_cost = img_plan.get("cost_per_unit", 0.05) + line_items.append({ + "category": "image_generation", + "provider": img_plan.get("tool", "unknown"), + "quantity": img_count, + "unit_cost_usd": unit_cost, + "total_usd": round(img_count * unit_cost, 4), + "basis": ( + f"~{images_per_scene:.0f} images/scene x {estimated_scenes} scenes " + f"+ {round((retry_multiplier - 1) * 100)}% retry buffer" + ), + }) + + # Video generation + if vid_plan and clips_needed_for_coverage > 0: + clip_count = round(clips_needed_for_coverage * retry_multiplier) + unit_cost = vid_plan.get("cost_per_unit", 0.30) + line_items.append({ + "category": "video_generation", + "provider": vid_plan.get("tool", "unknown"), + "quantity": clip_count, + "unit_cost_usd": unit_cost, + "total_usd": round(clip_count * unit_cost, 4), + "basis": ( + f"{motion_seconds:.0f}s of motion / {clip_duration}s clips = " + f"{clips_needed_for_coverage} clips + retry buffer" + ), + }) + assumptions.append( + f"{round(motion_ratio * 100)}% motion ratio → " + f"{motion_seconds:.0f}s needs {clips_needed_for_coverage} clips " + f"({clip_duration}s each)" + ) + + # TTS narration + tts_plan = tool_plan.get("tts", {}) + if tts_plan and estimated_words > 10: + cost_per_word = tts_plan.get("cost_per_word", 0.00003) + tts_cost = round(estimated_words * cost_per_word, 4) + line_items.append({ + "category": "tts_narration", + "provider": tts_plan.get("tool", "unknown"), + "quantity": estimated_words, + "unit_cost_usd": cost_per_word, + "total_usd": tts_cost, + "basis": f"Narration at {round(actual_wpm)} WPM = ~{estimated_words} words", + }) + assumptions.append( + f"Narration at {round(actual_wpm)} WPM = ~{estimated_words} words " + f"for {target_duration_seconds} seconds" + ) + + # Music + music_plan = tool_plan.get("music", {}) + if music_plan: + music_cost = music_plan.get("cost_per_track", 0.0) + line_items.append({ + "category": "music", + "provider": music_plan.get("tool", "unknown"), + "quantity": 1, + "unit_cost_usd": music_cost, + "total_usd": music_cost, + "basis": "1 background music track", + }) + + subtotal = round(sum(item["total_usd"] for item in line_items), 4) + + # ── Cost range instead of single number ── + # Low: everything works first try. High: retry buffer fully consumed. + low_total = round(subtotal / retry_multiplier, 4) + high_total = round(subtotal * 1.15, 4) # 15% above retry-buffered estimate + + # Sample cost: 2 scenes worth of assets (hook + 1 middle) + sample_scenes = 2 + sample_fraction = sample_scenes / max(estimated_scenes, 1) + sample_cost = round(subtotal * sample_fraction, 4) + + # Confidence based on how much data we have + if scenes_list and narration.get("word_count", 0) > 0: + confidence = "high" + elif scenes_list or narration.get("word_count", 0) > 0: + confidence = "medium" + else: + confidence = "low" + + return { + "line_items": line_items, + "total_usd": subtotal, + "total_range_usd": {"low": low_total, "high": high_total}, + "sample_cost_usd": sample_cost, + "confidence": confidence, + "assumptions": assumptions, + "estimated_scenes": estimated_scenes, + "estimated_images": estimated_images, + "estimated_clips": clips_needed_for_coverage, + "estimated_words": estimated_words, + "motion_ratio": round(motion_ratio, 2), + "cuts_per_minute": round(cuts_per_minute, 1), + "target_duration_seconds": target_duration_seconds, + } + + def _estimate_motion_ratio( + self, + *, + video_analysis_brief: dict, + scenes_list: list[dict[str, Any]], + pacing_style: str, + ) -> tuple[float, str]: + """Estimate how much of the target treatment truly needs motion.""" + motion_weights = { + "animation": 1.0, + "b_roll": 1.0, + "stock_footage": 1.0, + "product_shot": 0.9, + "transition": 0.6, + "screen_recording": 0.45, + "talking_head": 0.35, + "diagram": 0.25, + "chart": 0.25, + "text_card": 0.2, + } + classified_weights = [ + motion_weights[visual_type] + for scene in scenes_list + if (visual_type := scene.get("visual_type")) in motion_weights + ] + if classified_weights: + ratio = sum(classified_weights) / len(classified_weights) + unknown_count = max(0, len(scenes_list) - len(classified_weights)) + if unknown_count: + fallback_ratio, _ = self._fallback_motion_ratio( + video_analysis_brief=video_analysis_brief, + pacing_style=pacing_style, + ) + ratio = ( + (sum(classified_weights) + fallback_ratio * unknown_count) + / len(scenes_list) + ) + basis = ( + "motion ratio blended from classified scene types and " + "reference-style fallback for unclassified scenes" + ) + else: + basis = "motion ratio derived from classified scene types" + return round(min(max(ratio, 0.0), 0.95), 2), basis + + return self._fallback_motion_ratio( + video_analysis_brief=video_analysis_brief, + pacing_style=pacing_style, + ) + + def _fallback_motion_ratio( + self, + *, + video_analysis_brief: dict, + pacing_style: str, + ) -> tuple[float, str]: + """Fallback heuristic for motion ratio before scene vision enrichment.""" + source_type = video_analysis_brief.get("source", {}).get("type", "") + replication = video_analysis_brief.get("replication_guidance", {}) + motion_required = bool(replication.get("motion_required")) + suggested_pipeline = replication.get("suggested_pipeline", "") + + base_by_pacing = { + "rapid_fire": 0.8, + "dynamic_social": 0.65, + "steady_educational": 0.35, + "slow_contemplative": 0.2, + "variable": 0.5, + } + ratio = base_by_pacing.get(pacing_style, 0.5) + + if source_type in ("shorts", "instagram", "tiktok"): + ratio = max(ratio, 0.7) + if motion_required: + ratio = max(ratio, 0.6) + if suggested_pipeline == "cinematic": + ratio = max(ratio, 0.55) + + ratio = round(min(max(ratio, 0.1), 0.95), 2) + basis = ( + "motion ratio inferred from pacing/style because scene visual types " + "have not been enriched yet" + ) + return ratio, basis + # ---- Persistence ---- def _save(self) -> None: