Talking-head pipeline: 8 new tools, Remotion TalkingHead composition, and skill rewrites

New tools: face_tracker, visual_qa, eye_enhance, auto_reframe, remotion_caption_burn, showcase_card, silence_cutter. Updated audio_mixer with segmented_music operation and subtitle_gen with ASR corrections. Registered TalkingHead composition in Root.tsx. Rewrote compose/edit/scene director skills for full enhancement chain, Remotion captions, multi-clip assembly, and visual QA. Gitignore cleanup: exclude test demo-props, downloaded music, and generated images.
This commit is contained in:
calesthio
2026-04-01 10:00:15 -07:00
parent 237af7fb5c
commit 358b8647f5
17 changed files with 3467 additions and 37 deletions
+7
View File
@@ -44,6 +44,10 @@ skills-lock.json
# Generated images (logo drafts — final logo lives in assets/)
logo_*.png
generated_image.png
# Downloaded music files (Pixabay, Freesound — not source code)
pixabay_music_*
# OS
.DS_Store
@@ -57,3 +61,6 @@ remotion-composer/out/
remotion-composer/public/*
# But keep demo props (shipped with the project for zero-key demos)
!remotion-composer/public/demo-props/
# Ignore test/scratch compositions in demo-props (keep only curated demos)
remotion-composer/public/demo-props/test-*
remotion-composer/public/demo-props/talking-head-*
Binary file not shown.

Before

Width:  |  Height:  |  Size: 295 KiB

+29 -2
View File
@@ -78,8 +78,12 @@ stages:
required_tools: []
optional_tools:
- frame_sampler # representative frame extraction
- face_tracker # face position analysis for reframing decisions
- silence_cutter # silence detection for jump cut planning
tools_available:
- frame_sampler
- face_tracker
- silence_cutter
checkpoint_required: true
human_approval_default: true
review_focus:
@@ -126,7 +130,12 @@ stages:
- script
produces:
- edit_decisions
tools_available: []
optional_tools:
- silence_cutter # auto jump cuts on silent segments
- video_trimmer # speed control (0.5x2x), cut, concat
tools_available:
- silence_cutter
- video_trimmer
checkpoint_required: true
human_approval_default: false
review_focus:
@@ -149,17 +158,35 @@ stages:
- render_report
required_tools:
- video_compose # rendering
- audio_mixer # audio layering
- audio_mixer # audio layering + segmented music
optional_tools:
- face_enhance # face enhancement
- eye_enhance # eye brightening, dark circle removal
- color_grade # color correction
- audio_enhance # audio cleanup
- auto_reframe # aspect ratio conversion with face tracking
- face_tracker # face position data for reframing
- silence_cutter # auto jump cuts
- video_trimmer # speed control (0.5x2x)
- remotion_caption_burn # animated word-by-word captions via Remotion
- showcase_card # letterboxed showcase cards with typography
- video_stitch # multi-clip assembly with transitions
- visual_qa # frame extraction + quality validation
tools_available:
- video_compose
- audio_mixer
- face_enhance
- eye_enhance
- color_grade
- audio_enhance
- auto_reframe
- face_tracker
- silence_cutter
- video_trimmer
- remotion_caption_burn
- showcase_card
- video_stitch
- visual_qa
checkpoint_required: true
human_approval_default: false
review_focus:
+16
View File
@@ -5,6 +5,7 @@ import {
calculateCinematicMetadata,
} from "./CinematicRenderer";
import { signalFromTomorrowWithMusicFixture } from "./cinematic/fixtures";
import { TalkingHead, TalkingHeadProps } from "./TalkingHead";
const calculateMetadata: CalculateMetadataFunction<ExplainerProps> = async ({
props,
@@ -61,6 +62,21 @@ export const Root: React.FC = () => {
defaultProps={signalFromTomorrowWithMusicFixture}
calculateMetadata={calculateCinematicMetadata}
/>
<Composition
id="TalkingHead"
component={TalkingHead}
durationInFrames={30 * 30}
fps={30}
width={1080}
height={1920}
defaultProps={{
videoSrc: "",
captions: [],
wordsPerPage: 4,
fontSize: 52,
highlightColor: "#22D3EE",
}}
/>
</>
);
};
+35
View File
@@ -0,0 +1,35 @@
import { AbsoluteFill, OffthreadVideo } from "remotion";
import { CaptionOverlay, WordCaption } from "./components/CaptionOverlay";
export interface TalkingHeadProps {
videoSrc: string;
captions: WordCaption[];
wordsPerPage?: number;
fontSize?: number;
highlightColor?: string;
}
export const TalkingHead: React.FC<TalkingHeadProps> = ({
videoSrc,
captions,
wordsPerPage = 4,
fontSize = 52,
highlightColor = "#22D3EE",
}) => {
return (
<AbsoluteFill style={{ backgroundColor: "#000" }}>
<OffthreadVideo
src={videoSrc}
style={{ width: "100%", height: "100%", objectFit: "cover" }}
/>
<CaptionOverlay
words={captions}
wordsPerPage={wordsPerPage}
fontSize={fontSize}
highlightColor={highlightColor}
backgroundColor="rgba(0, 0, 0, 0.65)"
color="#FFFFFF"
/>
</AbsoluteFill>
);
};
+184 -20
View File
@@ -18,51 +18,215 @@ You have edit decisions and an asset manifest. Your job is to render the final t
### Step 1: Run Enhancement Chain
Apply video enhancements in order:
1. **Face enhancement** (if face_enhance tool available) — sharpen faces
2. **Color grading** (if color_grade tool available) — apply a profile
3. **Audio enhancement** (if audio_enhance tool available) — noise reduction, normalization
1. **Face enhancement** (if `face_enhance` tool available) — apply `talking_head_standard` preset
2. **Eye enhancement** (if `eye_enhance` tool available) — under-eye dark circle removal + eye brightening
3. **Color grading** (if `color_grade` tool available) — apply a profile
4. **Audio enhancement** (if `audio_enhance` tool available) — noise reduction, normalization
Each step is optional — check tool availability first.
### Step 2: Burn Subtitles
**Eye enhancement** — removes under-eye dark circles and brightens eyes using MediaPipe Face Mesh landmark detection:
```
eye_enhance.execute({
"input_path": "<face_enhanced_video>",
"output_path": "<project>/assets/video/eye_enhanced.mp4",
"operations": ["dark_circles", "brighten_eyes"],
"dark_circle_intensity": 0.4, # 0-1, subtle is better
"eye_brighten_intensity": 0.3,
})
```
**Important:** Keep intensities low (0.2-0.5). Over-processing makes eyes look unnatural. Always compare before/after.
### Step 1b: Speed Adjustment (if requested)
If the user wants the video sped up or slowed down, use `video_trimmer`:
```
video_trimmer.execute({
"operation": "speed",
"input_path": "<enhanced_video>",
"output_path": "<project>/assets/video/speed_adjusted.mp4",
"speed_factor": 1.25 # 0.5x (slow), 1.25x, 1.5x, 2x (fast)
})
```
Common speed factors:
| Factor | Use Case |
|--------|----------|
| `0.5` | Slow-mo for dramatic effect |
| `1.0` | Normal (no change) |
| `1.25` | Slightly faster — tighter pacing without sounding unnatural |
| `1.5` | Noticeably faster — good for recaps or condensed content |
| `2.0` | Double speed — time-lapse effect |
Apply speed AFTER enhancements, BEFORE reframing.
### Step 2: Auto-Reframe (if target platform requires it)
If the target platform requires a different aspect ratio (e.g. Instagram Reels = 9:16), use `auto_reframe`:
```
auto_reframe.execute({
"input_path": "<enhanced_video>",
"output_path": "<project>/renders/reframed.mp4",
"target_aspect": "portrait", # 9:16 for Reels/TikTok/Shorts
"smoothing_window": 15, # smooth camera pan
"face_padding": 0.4, # 40% padding around face
})
```
**Aspect ratio presets:**
| Preset | Ratio | Platform |
|--------|-------|----------|
| `portrait` | 9:16 | Instagram Reels, TikTok, YouTube Shorts |
| `square` | 1:1 | Instagram Feed |
| `landscape` | 16:9 | YouTube, LinkedIn |
| `vertical_4_5` | 4:5 | Instagram portrait post |
The tool automatically runs face detection and keeps the speaker centered. If MediaPipe is not installed, falls back to center-crop.
**Important:** Run auto_reframe AFTER face_enhance and color_grade but BEFORE burning subtitles. Subtitles need to be positioned for the final aspect ratio.
### Step 3: Burn Subtitles
**Preferred: Remotion captions** (if `remotion_caption_burn` tool available):
```
remotion_caption_burn.execute({
"input_path": "<reframed_or_enhanced_video>",
"output_path": "<project>/assets/video/captioned.mp4",
"segments": <transcript_segments_from_asset_manifest>,
"corrections": {"cloud": "Claude", "co-pilot": "Copilot"},
"words_per_page": 4,
"font_size": 52,
"highlight_color": "#22D3EE",
})
```
Remotion renders animated word-by-word captions at the bottom of the frame with active word highlighting. Captions are positioned away from the face.
**Fallback: FFmpeg subtitles** (if Remotion unavailable):
Use `video_compose` with `burn_subtitles` operation:
- Input: enhanced video (or raw if no enhancements)
- Input: reframed video (or enhanced video if no reframe needed)
- Subtitle file from asset manifest
- Style from playbook
- For vertical (9:16) output: position subtitles in the lower 20% of frame with `MarginV=100`
- **Never** position subtitles in the center of the frame — they will occlude the face
### Step 3: Mix Audio
### Step 3b: Build Showcase Cards (if multi-clip reel)
Use `audio_mixer` to:
- Layer original audio with any background music
If the output is a reel with showcase clips, use `showcase_card` for each:
```
showcase_card.execute({
"input_path": "<showcase_video>",
"output_path": "<project>/assets/video/sc_<name>.mp4",
"title": "VIDEO TITLE",
"subtitle": "Description | Style | Cost: $0.15",
"background_color": "0x0A0F1A",
})
```
This creates letterboxed 9:16 cards with typography.
### Step 4: Assemble Multi-Clip (if applicable)
If the output has multiple segments (e.g. talking head + showcase clips), use `video_stitch`:
```
video_stitch.execute({
"operation": "stitch",
"clips": ["intro.mp4", "showcase1.mp4", ..., "outro.mp4"],
"output_path": "<project>/renders/assembled.mp4",
"transition": "crossfade", # or "fade" for fade-through-black
"transition_duration": 0.5,
})
```
**Transition guidance:**
- `crossfade` (fade): smooth blend between talking head and showcase
- `fade` (fade-through-black): brief dip to black between showcase clips
- Mix transition types: use `crossfade` for talk→showcase, `fade` between showcases
### Step 5: Mix Audio
Use `audio_mixer` to layer background music:
**For multi-clip reels** — use `segmented_music` to play music only during talking head sections:
```
audio_mixer.execute({
"operation": "segmented_music",
"video_path": "<assembled_video>",
"music_path": "<bg_music>",
"music_volume": 0.20,
"segments": [
{"start": 0, "end": 17.0}, # intro speech
{"start": 167.0, "end": 175.0} # outro speech
],
"fade_duration": 0.5,
"output_path": "<project>/renders/final.mp4",
})
```
**For single talking-head videos** — use `duck` or `full_mix`:
- Layer original audio with background music
- Apply ducking if music is present
- Normalize final levels
### Step 4: Final Encode
### Step 6: Final Encode
Use `video_compose` with `encode` operation:
- Apply target media profile (youtube_landscape, tiktok, etc.)
- Apply target media profile (youtube_landscape, tiktok, instagram_reels, etc.)
- Two-pass encoding for quality
### Step 5: Verify Output
### Step 7: Visual QA
- Check file exists and is playable
- Verify duration matches expectations
- Check audio is present
Use `visual_qa` to verify the output before declaring success:
```
visual_qa.execute({
"operation": "review",
"input_path": "<final_video>",
"timestamps": [3.0, 10.0, 25.0, 50.0, 100.0, 170.0],
})
```
Then **read each extracted frame** to verify:
- Captions are visible and positioned at the bottom (not on the face)
- Face enhancement is applied (skin looks smooth, not over-processed)
- Transitions are clean (no artifacts at transition points)
- Showcase cards have readable typography
### Step 6: Build Render Report
Also run probe validation:
```
visual_qa.execute({
"operation": "probe",
"input_path": "<final_video>",
"expected": {
"width": 1080, "height": 1920,
"has_audio": true,
"pixel_format": "yuv420p"
},
})
```
Document output: path, format, resolution, duration, file size.
And check audio levels:
```
visual_qa.execute({
"operation": "audio_levels",
"input_path": "<final_video>",
"timestamps": [5.0, 50.0, 170.0],
})
```
Verify: speech sections have higher volume than showcase sections (confirms music placement).
### Step 7: Self-Evaluate
### Step 8: Build Render Report
Document output: path, format, resolution, duration, file size, QA results.
### Step 9: Self-Evaluate
| Criterion | Question |
|-----------|----------|
| **Playability** | Does the video play without errors? |
| **Quality** | Are enhancements applied correctly? |
| **Audio** | Is speech clear with balanced levels? |
| **Subtitles** | Are subtitles visible and synced? |
| **Framing** | If reframed — is the face centered? No important content cropped? |
| **Audio** | Is speech clear with balanced levels? Music only during intended segments? |
| **Subtitles** | Are captions visible at the bottom? Not occluding the face? Word highlighting working? |
| **Transitions** | Are transitions clean? Correct type (crossfade vs fadeblack)? |
| **Showcase** | Are showcase cards properly letterboxed with readable typography? |
### Step 8: Submit
### Step 10: Submit
Validate the render_report against the schema and persist via checkpoint.
+31 -7
View File
@@ -14,39 +14,63 @@ You have a scene plan and asset manifest. Your job is to assemble the edit decis
## Process
### Step 1: Define Primary Cut
### Step 1: Apply Silence Cuts (if planned)
If the scene plan includes silence removal, run `silence_cutter` before defining cuts:
```
silence_cutter.execute({
"input_path": "<raw_footage>",
"mode": "remove", # or "speed_up" for less jarring result
"silence_threshold_db": -35,
"min_silence_duration": 0.5,
"padding_seconds": 0.08, # prevents clipped words
"output_path": "<project>/assets/video/footage_cut.mp4"
})
```
**Choosing the mode:**
- `remove` — Hard jump cuts. Best for fast-paced social content (Reels, TikTok, Shorts)
- `speed_up` — Fast-forwards through silence at 6x. Less jarring for longer-form content (YouTube, LinkedIn)
Present the result to the user: "Removed X seconds of silence (Y%) — output is now Z seconds."
Use the cut footage as the source for all subsequent steps.
### Step 2: Define Primary Cut
For talking-head, the primary cut is usually the full footage (or trimmed segments). Create cuts that:
- Reference the raw footage as source
- Reference the raw footage (or silence-cut footage) as source
- Use timestamps from the script sections
- Apply any trim decisions (cut dead air, false starts)
### Step 2: Configure Subtitles
### Step 3: Configure Subtitles
- Enable subtitles with playbook-compatible styling
- Reference the subtitle asset from the manifest
- Set position (usually bottom-center)
### Step 3: Configure Audio
### Step 4: Configure Audio
- Set narration to the raw footage audio
- If background music is desired, configure ducking
- Set music volume per playbook
### Step 4: Plan Enhancements
### Step 5: Plan Enhancements
If the scene plan includes overlays:
- Add overlay cuts for text cards, lower thirds
- Time them to match speech content
### Step 5: Self-Evaluate
### Step 6: Self-Evaluate
| Criterion | Question |
|-----------|----------|
| **Coverage** | Do cuts span the full intended duration? |
| **Silence** | Were silence cuts applied if planned? What % was removed? |
| **Subtitles** | Are subtitles enabled and styled? |
| **Audio** | Is audio configuration complete? |
### Step 6: Submit
### Step 7: Submit
Validate the edit_decisions against the schema and persist via checkpoint.
@@ -11,28 +11,70 @@ You have a script (from transcription) and raw footage. Your job is to create a
| Schema | `schemas/artifacts/scene_plan.schema.json` | Artifact validation |
| Prior artifacts | Script, Brief | Section timing and context |
| Tools | `frame_sampler` (optional) | Extract representative frames |
| Tools | `face_tracker` (optional) | Analyze speaker face position for reframing |
| Tools | `silence_cutter` (optional) | Detect silence for jump cut planning |
## Process
### Step 1: Plan Base Scenes
### Step 1: Analyze Footage (if tools available)
**Face tracking** — If `face_tracker` is available, run it on the raw footage:
```
face_tracker.execute({
"input_path": "<raw_footage>",
"sample_fps": 5
})
```
This outputs per-frame face bounding boxes. Use this data to:
- Decide if reframing is needed (e.g. speaker is off-center for vertical crop)
- Identify sections where the speaker moves significantly (needs dynamic crop)
- Note face position for auto_reframe in the compose stage
**Silence detection** — If `silence_cutter` is available, run in `mark` mode:
```
silence_cutter.execute({
"input_path": "<raw_footage>",
"mode": "mark",
"silence_threshold_db": -35,
"min_silence_duration": 0.5
})
```
This outputs silence/speech segment timestamps. Use this to:
- Plan which segments should be jump-cut or sped up
- Identify dead air, false starts, and long pauses
- Estimate the final video duration after cuts
- Present the user with a summary: "Found X seconds of silence across Y segments — recommend removing?"
### Step 2: Plan Base Scenes
For talking-head, the base is simple: one scene per script section, all type `talking_head`. The raw footage IS the scene.
### Step 2: Plan Enhancement Scenes
### Step 3: Plan Enhancement Scenes
Based on script enhancement cues, plan overlay scenes:
- Text cards for key terms or statistics
- Lower thirds for speaker identification
- B-roll suggestions for topic illustrations
### Step 3: Build Scene Plan
### Step 4: Plan Reframing & Cuts
If the target platform requires a different aspect ratio (e.g. Instagram Reels = 9:16):
- Note `auto_reframe` should be applied in the compose stage
- Record the target aspect ratio in the scene plan
- If face tracking data shows significant speaker movement, note that dynamic crop is needed
If silence detection found segments to cut:
- Record the recommended cut mode (`remove` or `speed_up`) in the scene plan
- Note padding preferences (default 0.08s to avoid clipping words)
### Step 5: Build Scene Plan
Create a scene per section with:
- Type: `talking_head` (primary)
- Timing from script sections
- Required assets: subtitle file, any overlay images
### Step 4: Self-Evaluate
### Step 6: Self-Evaluate
| Criterion | Question |
|-----------|----------|
@@ -40,6 +82,6 @@ Create a scene per section with:
| **Enhancement** | Are overlay opportunities identified? |
| **Feasibility** | Can all required assets be generated? |
### Step 5: Submit
### Step 7: Submit
Validate the scene_plan against the schema and persist via checkpoint.
+314
View File
@@ -0,0 +1,314 @@
"""Face tracking tool using MediaPipe Face Mesh.
Tracks face bounding boxes, landmarks, and head pose across video frames.
Outputs per-frame face data as JSON — used by auto_reframe, face_enhance,
and other tools that need to know where the speaker's face is.
Falls back to OpenCV Haar cascade if MediaPipe is not installed.
"""
from __future__ import annotations
import json
import time
from pathlib import Path
from typing import Any
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
ToolResult,
ToolStability,
ToolStatus,
ToolTier,
)
class FaceTracker(BaseTool):
name = "face_tracker"
version = "0.1.0"
tier = ToolTier.CORE
capability = "analysis"
provider = "mediapipe"
stability = ToolStability.EXPERIMENTAL
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
dependencies = ["cmd:ffmpeg"]
install_instructions = (
"For best results install MediaPipe:\n"
"pip install mediapipe opencv-python\n\n"
"Falls back to OpenCV Haar cascade (ships with opencv-python)."
)
agent_skills = ["ffmpeg"]
capabilities = [
"face_detection",
"face_tracking",
"face_bounding_box",
"head_pose_estimation",
]
input_schema = {
"type": "object",
"required": ["input_path"],
"properties": {
"input_path": {"type": "string"},
"output_path": {
"type": "string",
"description": "Path for face tracking JSON output",
},
"sample_fps": {
"type": "number",
"default": 5,
"description": "Frames per second to sample (lower = faster, less precise)",
},
"min_detection_confidence": {
"type": "number",
"default": 0.5,
"minimum": 0.0,
"maximum": 1.0,
},
},
}
output_schema = {
"type": "object",
"properties": {
"frame_count": {"type": "integer"},
"face_detected_count": {"type": "integer"},
"video_width": {"type": "integer"},
"video_height": {"type": "integer"},
"fps": {"type": "number"},
"duration_seconds": {"type": "number"},
"faces": {
"type": "array",
"items": {
"type": "object",
"properties": {
"frame_index": {"type": "integer"},
"timestamp_seconds": {"type": "number"},
"bbox": {
"type": "object",
"properties": {
"x": {"type": "number"},
"y": {"type": "number"},
"width": {"type": "number"},
"height": {"type": "number"},
},
},
},
},
},
},
}
resource_profile = ResourceProfile(cpu_cores=2, ram_mb=1024, vram_mb=0, disk_mb=100)
idempotency_key_fields = ["input_path", "sample_fps", "min_detection_confidence"]
side_effects = ["writes face tracking JSON to output_path"]
user_visible_verification = [
"Spot-check bounding boxes against video frames",
]
fallback_tools = []
def _has_mediapipe(self) -> bool:
try:
import mediapipe # noqa: F401
return True
except ImportError:
return False
def _has_opencv(self) -> bool:
try:
import cv2 # noqa: F401
return True
except ImportError:
return False
def get_status(self) -> ToolStatus:
if self._has_mediapipe() and self._has_opencv():
return ToolStatus.AVAILABLE
if self._has_opencv():
return ToolStatus.DEGRADED
return ToolStatus.UNAVAILABLE
def execute(self, inputs: dict[str, Any]) -> ToolResult:
input_path = Path(inputs["input_path"])
if not input_path.exists():
return ToolResult(success=False, error=f"Input not found: {input_path}")
if not self._has_opencv():
return ToolResult(
success=False,
error="opencv-python is required. Install: pip install opencv-python",
)
output_path = Path(
inputs.get("output_path", str(input_path.with_suffix(".faces.json")))
)
output_path.parent.mkdir(parents=True, exist_ok=True)
sample_fps = inputs.get("sample_fps", 5)
confidence = inputs.get("min_detection_confidence", 0.5)
start = time.time()
if self._has_mediapipe():
result_data = self._track_mediapipe(input_path, sample_fps, confidence)
else:
result_data = self._track_opencv(input_path, sample_fps)
elapsed = time.time() - start
output_path.write_text(json.dumps(result_data, indent=2), encoding="utf-8")
return ToolResult(
success=True,
data={
"output": str(output_path),
"video_width": result_data["video_width"],
"video_height": result_data["video_height"],
"fps": result_data["fps"],
"duration_seconds": result_data["duration_seconds"],
"frames_sampled": result_data["frame_count"],
"faces_detected": result_data["face_detected_count"],
"method": "mediapipe" if self._has_mediapipe() else "opencv_haar",
},
artifacts=[str(output_path)],
duration_seconds=round(elapsed, 2),
)
def _track_mediapipe(
self, input_path: Path, sample_fps: float, confidence: float
) -> dict:
import cv2
import mediapipe as mp
mp_face = mp.solutions.face_detection
cap = cv2.VideoCapture(str(input_path))
video_fps = cap.get(cv2.CAP_PROP_FPS) or 30.0
video_w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))
video_h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
duration = total_frames / video_fps if video_fps > 0 else 0
# Calculate frame sampling interval
sample_interval = max(1, int(video_fps / sample_fps))
faces_data: list[dict] = []
frame_idx = 0
sampled = 0
with mp_face.FaceDetection(
model_selection=1, # 1 = full range (up to 5m), 0 = short range (up to 2m)
min_detection_confidence=confidence,
) as detector:
while cap.isOpened():
ret, frame = cap.read()
if not ret:
break
if frame_idx % sample_interval == 0:
sampled += 1
rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
results = detector.process(rgb)
if results.detections:
# Use the highest-confidence detection
det = max(
results.detections,
key=lambda d: d.score[0],
)
bbox = det.location_data.relative_bounding_box
faces_data.append({
"frame_index": frame_idx,
"timestamp_seconds": round(frame_idx / video_fps, 3),
"confidence": round(det.score[0], 3),
"bbox": {
"x": round(bbox.xmin, 4),
"y": round(bbox.ymin, 4),
"width": round(bbox.width, 4),
"height": round(bbox.height, 4),
},
})
frame_idx += 1
cap.release()
return {
"video_width": video_w,
"video_height": video_h,
"fps": round(video_fps, 2),
"duration_seconds": round(duration, 3),
"frame_count": sampled,
"face_detected_count": len(faces_data),
"faces": faces_data,
}
def _track_opencv(self, input_path: Path, sample_fps: float) -> dict:
"""Fallback: OpenCV Haar cascade face detection."""
import cv2
cascade_path = cv2.data.haarcascades + "haarcascade_frontalface_default.xml"
cascade = cv2.CascadeClassifier(cascade_path)
cap = cv2.VideoCapture(str(input_path))
video_fps = cap.get(cv2.CAP_PROP_FPS) or 30.0
video_w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))
video_h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
duration = total_frames / video_fps if video_fps > 0 else 0
sample_interval = max(1, int(video_fps / sample_fps))
faces_data: list[dict] = []
frame_idx = 0
sampled = 0
while cap.isOpened():
ret, frame = cap.read()
if not ret:
break
if frame_idx % sample_interval == 0:
sampled += 1
gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)
detected = cascade.detectMultiScale(
gray, scaleFactor=1.1, minNeighbors=5, minSize=(60, 60)
)
if len(detected) > 0:
# Pick largest face
areas = [w * h for (_, _, w, h) in detected]
best_idx = areas.index(max(areas))
x, y, w, h = detected[best_idx]
faces_data.append({
"frame_index": frame_idx,
"timestamp_seconds": round(frame_idx / video_fps, 3),
"confidence": 0.0, # Haar doesn't provide confidence
"bbox": {
"x": round(x / video_w, 4),
"y": round(y / video_h, 4),
"width": round(w / video_w, 4),
"height": round(h / video_h, 4),
},
})
frame_idx += 1
cap.release()
return {
"video_width": video_w,
"video_height": video_h,
"fps": round(video_fps, 2),
"duration_seconds": round(duration, 3),
"frame_count": sampled,
"face_detected_count": len(faces_data),
"faces": faces_data,
}
+343
View File
@@ -0,0 +1,343 @@
"""Visual QA tool for automated video quality checks.
Extracts frames at specified timestamps and runs basic quality checks:
- File existence, resolution, duration, codec validation
- Frame extraction for visual inspection by the agent
- Caption occlusion check (compares brightness in face vs caption zones)
- Transition verification (frame similarity at transition points)
Returns frame paths so the agent can visually inspect them.
"""
from __future__ import annotations
import time
from pathlib import Path
from typing import Any
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
ToolResult,
ToolStability,
ToolTier,
)
class VisualQA(BaseTool):
name = "visual_qa"
version = "0.1.0"
tier = ToolTier.CORE
capability = "analysis"
provider = "ffmpeg"
stability = ToolStability.EXPERIMENTAL
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
dependencies = ["cmd:ffmpeg", "cmd:ffprobe"]
install_instructions = "Install FFmpeg: https://ffmpeg.org/download.html"
agent_skills = ["ffmpeg"]
capabilities = [
"extract_review_frames",
"probe_video",
"check_audio_levels",
]
input_schema = {
"type": "object",
"required": ["operation", "input_path"],
"properties": {
"operation": {
"type": "string",
"enum": ["review", "probe", "audio_levels"],
"description": (
"review: extract frames at timestamps for visual inspection. "
"probe: get video metadata (duration, resolution, codecs). "
"audio_levels: check audio volume at specified timestamps."
),
},
"input_path": {
"type": "string",
"description": "Path to the video file to inspect.",
},
"timestamps": {
"type": "array",
"items": {"type": "number"},
"description": (
"Timestamps (in seconds) at which to extract frames or "
"check audio levels."
),
},
"output_dir": {
"type": "string",
"description": (
"Directory to save extracted frames. Defaults to a "
"'review_frames' subdirectory next to the input file."
),
},
"checks": {
"type": "array",
"items": {
"type": "string",
"enum": [
"resolution",
"duration",
"audio_present",
"pixel_format",
"file_size",
],
},
"description": "Specific checks to run (probe operation).",
},
"expected": {
"type": "object",
"description": (
"Expected values for validation. "
"Keys: width, height, min_duration, max_duration, "
"pixel_format, has_audio."
),
"properties": {
"width": {"type": "integer"},
"height": {"type": "integer"},
"min_duration": {"type": "number"},
"max_duration": {"type": "number"},
"pixel_format": {"type": "string"},
"has_audio": {"type": "boolean"},
},
},
},
}
resource_profile = ResourceProfile(cpu_cores=1, ram_mb=512, vram_mb=0, disk_mb=200)
idempotency_key_fields = ["operation", "input_path", "timestamps"]
side_effects = ["writes frame images to output_dir"]
user_visible_verification = [
"Visually inspect extracted frames for quality issues",
]
def execute(self, inputs: dict[str, Any]) -> ToolResult:
operation = inputs["operation"]
input_path = inputs["input_path"]
if not Path(input_path).exists():
return ToolResult(success=False, error=f"Input not found: {input_path}")
start = time.time()
try:
if operation == "review":
result = self._review(inputs)
elif operation == "probe":
result = self._probe(inputs)
elif operation == "audio_levels":
result = self._audio_levels(inputs)
else:
return ToolResult(success=False, error=f"Unknown operation: {operation}")
except Exception as e:
return ToolResult(success=False, error=str(e))
result.duration_seconds = round(time.time() - start, 2)
return result
def _review(self, inputs: dict[str, Any]) -> ToolResult:
"""Extract frames at specified timestamps for visual review."""
input_path = inputs["input_path"]
timestamps = inputs.get("timestamps", [])
if not timestamps:
# Auto-generate timestamps: start, 25%, 50%, 75%, end-1s
dur = self._get_duration(input_path)
timestamps = [
1.0,
dur * 0.25,
dur * 0.50,
dur * 0.75,
max(dur - 1.0, 0),
]
output_dir = inputs.get("output_dir")
if not output_dir:
output_dir = str(Path(input_path).parent / "review_frames")
Path(output_dir).mkdir(parents=True, exist_ok=True)
frames = []
for ts in timestamps:
ts_label = f"{ts:.1f}".replace(".", "_")
frame_path = str(Path(output_dir) / f"frame_{ts_label}s.jpg")
cmd = [
"ffmpeg", "-y",
"-ss", str(ts),
"-i", input_path,
"-frames:v", "1",
"-q:v", "2",
frame_path,
]
try:
self.run_command(cmd)
if Path(frame_path).exists():
frames.append({
"timestamp": ts,
"path": frame_path,
})
except Exception:
frames.append({
"timestamp": ts,
"path": None,
"error": f"Failed to extract frame at {ts}s",
})
return ToolResult(
success=True,
data={
"operation": "review",
"input": input_path,
"frame_count": len([f for f in frames if f.get("path")]),
"frames": frames,
},
artifacts=[f["path"] for f in frames if f.get("path")],
)
def _probe(self, inputs: dict[str, Any]) -> ToolResult:
"""Probe video metadata and optionally validate against expectations."""
input_path = inputs["input_path"]
expected = inputs.get("expected", {})
# Get comprehensive probe data
cmd = [
"ffprobe", "-v", "error",
"-show_entries",
"format=duration,size:stream=width,height,codec_name,pix_fmt,"
"r_frame_rate,sample_rate,channels,codec_type",
"-of", "json",
input_path,
]
import json
probe_out = self.run_command(cmd, capture=True)
probe_data = json.loads(probe_out)
# Extract key info
video_stream = None
audio_stream = None
for s in probe_data.get("streams", []):
if s.get("codec_type") == "video" and not video_stream:
video_stream = s
elif s.get("codec_type") == "audio" and not audio_stream:
audio_stream = s
info = {
"duration": float(probe_data.get("format", {}).get("duration", 0)),
"file_size_mb": round(
int(probe_data.get("format", {}).get("size", 0)) / 1048576, 1
),
"has_audio": audio_stream is not None,
}
if video_stream:
info.update({
"width": video_stream.get("width"),
"height": video_stream.get("height"),
"pixel_format": video_stream.get("pix_fmt"),
"video_codec": video_stream.get("codec_name"),
"frame_rate": video_stream.get("r_frame_rate"),
})
if audio_stream:
info.update({
"audio_codec": audio_stream.get("codec_name"),
"sample_rate": audio_stream.get("sample_rate"),
"channels": audio_stream.get("channels"),
})
# Validate against expectations
issues = []
if "width" in expected and info.get("width") != expected["width"]:
issues.append(f"Width: expected {expected['width']}, got {info.get('width')}")
if "height" in expected and info.get("height") != expected["height"]:
issues.append(f"Height: expected {expected['height']}, got {info.get('height')}")
if "min_duration" in expected and info["duration"] < expected["min_duration"]:
issues.append(
f"Duration too short: {info['duration']:.1f}s < {expected['min_duration']}s"
)
if "max_duration" in expected and info["duration"] > expected["max_duration"]:
issues.append(
f"Duration too long: {info['duration']:.1f}s > {expected['max_duration']}s"
)
if "pixel_format" in expected and info.get("pixel_format") != expected["pixel_format"]:
issues.append(
f"Pixel format: expected {expected['pixel_format']}, got {info.get('pixel_format')}"
)
if "has_audio" in expected and info["has_audio"] != expected["has_audio"]:
issues.append(
f"Audio: expected {'present' if expected['has_audio'] else 'absent'}, "
f"got {'present' if info['has_audio'] else 'absent'}"
)
info["validation_issues"] = issues
info["validation_passed"] = len(issues) == 0
return ToolResult(
success=True,
data={
"operation": "probe",
"input": input_path,
**info,
},
)
def _audio_levels(self, inputs: dict[str, Any]) -> ToolResult:
"""Check audio levels at specified timestamps."""
input_path = inputs["input_path"]
timestamps = inputs.get("timestamps", [])
if not timestamps:
dur = self._get_duration(input_path)
timestamps = [1.0, dur * 0.5, max(dur - 2.0, 0)]
levels = []
for ts in timestamps:
cmd = [
"ffmpeg", "-y",
"-ss", str(ts),
"-t", "3",
"-i", input_path,
"-vn", "-af", "volumedetect",
"-f", "null", "/dev/null",
]
try:
output = self.run_command(cmd, capture=True, stderr=True)
mean_vol = None
max_vol = None
for line in output.split("\n"):
if "mean_volume" in line:
mean_vol = float(line.split("mean_volume:")[1].strip().split()[0])
elif "max_volume" in line:
max_vol = float(line.split("max_volume:")[1].strip().split()[0])
levels.append({
"timestamp": ts,
"mean_volume_db": mean_vol,
"max_volume_db": max_vol,
})
except Exception as e:
levels.append({
"timestamp": ts,
"error": str(e),
})
return ToolResult(
success=True,
data={
"operation": "audio_levels",
"input": input_path,
"levels": levels,
},
)
def _get_duration(self, path: str) -> float:
cmd = [
"ffprobe", "-v", "error",
"-show_entries", "format=duration",
"-of", "csv=p=0",
path,
]
return float(self.run_command(cmd, capture=True).strip().split("\n")[0])
+149 -3
View File
@@ -40,7 +40,7 @@ class AudioMixer(BaseTool):
)
agent_skills = ["ffmpeg", "video_toolkit"]
capabilities = ["mix", "duck", "fade", "normalize", "extract_audio"]
capabilities = ["mix", "duck", "fade", "normalize", "extract_audio", "segmented_music"]
input_schema = {
"type": "object",
@@ -48,13 +48,16 @@ class AudioMixer(BaseTool):
"properties": {
"operation": {
"type": "string",
"enum": ["mix", "duck", "extract", "full_mix"],
"enum": ["mix", "duck", "extract", "full_mix", "segmented_music"],
"description": (
"mix: layer multiple tracks with volume/delay/fades. "
"duck: lower music volume when speech is present. "
"extract: extract audio from video file. "
"full_mix: combine narration tracks + music with ducking + normalize "
"in a single call (preferred for compose-director)."
"in a single call (preferred for compose-director). "
"segmented_music: mix music into a video only during specified "
"time segments (e.g. music during talking head, silence during "
"showcase clips)."
),
},
"tracks": {
@@ -128,6 +131,45 @@ class AudioMixer(BaseTool):
},
},
"normalize": {"type": "boolean", "default": True},
"video_path": {
"type": "string",
"description": (
"Path to the assembled video (segmented_music operation). "
"Music is mixed into this video's audio at specified segments."
),
},
"music_path": {
"type": "string",
"description": "Path to background music file (segmented_music operation).",
},
"music_volume": {
"type": "number",
"minimum": 0,
"maximum": 1.0,
"default": 0.20,
"description": "Volume level for music during active segments.",
},
"segments": {
"type": "array",
"description": (
"Time segments where music should play (segmented_music operation). "
"Each segment: {start: seconds, end: seconds}. Music fades in/out "
"at segment boundaries. Outside these segments, music is silent."
),
"items": {
"type": "object",
"required": ["start", "end"],
"properties": {
"start": {"type": "number", "minimum": 0},
"end": {"type": "number", "minimum": 0},
},
},
},
"fade_duration": {
"type": "number",
"default": 0.5,
"description": "Duration of fade in/out at segment boundaries (seconds).",
},
},
}
@@ -151,6 +193,8 @@ class AudioMixer(BaseTool):
result = self._extract(inputs)
elif operation == "full_mix":
result = self._full_mix(inputs)
elif operation == "segmented_music":
result = self._segmented_music(inputs)
else:
return ToolResult(success=False, error=f"Unknown operation: {operation}")
except Exception as e:
@@ -558,3 +602,105 @@ class AudioMixer(BaseTool):
},
artifacts=[str(output_path)],
)
def _segmented_music(self, inputs: dict[str, Any]) -> ToolResult:
"""Mix background music into a video only during specified time segments.
Uses FFmpeg volume expressions with smooth fades at segment boundaries.
Music is silent outside the specified segments.
Input format:
{
"operation": "segmented_music",
"video_path": "assembled.mp4",
"music_path": "bg_music.mp3",
"music_volume": 0.20,
"segments": [
{"start": 0, "end": 17.0},
{"start": 167.0, "end": 175.0}
],
"fade_duration": 0.5,
"output_path": "final_with_music.mp4"
}
"""
video_path = inputs.get("video_path")
music_path = inputs.get("music_path")
output_path = Path(inputs.get("output_path", "segmented_music_output.mp4"))
segments = inputs.get("segments", [])
music_volume = inputs.get("music_volume", 0.20)
fade_dur = inputs.get("fade_duration", 0.5)
if not video_path or not Path(video_path).exists():
return ToolResult(success=False, error=f"Video not found: {video_path}")
if not music_path or not Path(music_path).exists():
return ToolResult(success=False, error=f"Music not found: {music_path}")
if not segments:
return ToolResult(success=False, error="No segments specified")
output_path.parent.mkdir(parents=True, exist_ok=True)
# Get video duration
dur_cmd = [
"ffprobe", "-v", "error",
"-show_entries", "format=duration",
"-of", "csv=p=0",
video_path,
]
total_dur = float(self.run_command(dur_cmd, capture=True).strip().split("\n")[0])
# Build volume expression for each segment with smooth fades
parts = []
for seg in sorted(segments, key=lambda s: s["start"]):
s = seg["start"]
e = seg["end"]
fade_in_end = s + fade_dur
fade_out_start = e - fade_dur
parts.append(
f"if(lt(t,{s}),0,"
f"if(lt(t,{fade_in_end}),{music_volume}*(t-{s})/{fade_dur},"
f"if(lt(t,{fade_out_start}),{music_volume},"
f"if(lt(t,{e}),{music_volume}*({e}-t)/{fade_dur},"
f"0))))"
)
vol_expr = "+".join(f"({p})" for p in parts) if len(parts) > 1 else parts[0]
filter_complex = (
f"[1:a]atrim=0:{total_dur},asetpts=PTS-STARTPTS,"
f"volume='{vol_expr}':eval=frame[music_shaped];"
f"[0:a]aformat=sample_fmts=fltp:sample_rates=44100:channel_layouts=stereo[speech];"
f"[music_shaped]aformat=sample_fmts=fltp:sample_rates=44100:channel_layouts=stereo[music_fmt];"
f"[speech][music_fmt]amix=inputs=2:duration=first:dropout_transition=2[aout]"
)
cmd = [
"ffmpeg", "-y",
"-i", video_path,
"-stream_loop", "-1",
"-i", music_path,
"-filter_complex", filter_complex,
"-map", "0:v",
"-map", "[aout]",
"-c:v", "copy",
"-c:a", "aac", "-b:a", "192k",
str(output_path),
]
self.run_command(cmd)
if not output_path.exists():
return ToolResult(success=False, error="No output produced")
return ToolResult(
success=True,
data={
"operation": "segmented_music",
"video": video_path,
"music": music_path,
"segments": segments,
"music_volume": music_volume,
"fade_duration": fade_dur,
"output": str(output_path),
},
artifacts=[str(output_path)],
)
+578
View File
@@ -0,0 +1,578 @@
"""Eye enhancement tool using MediaPipe Face Mesh + OpenCV.
Targets the eye region for talking-head footage:
- Under-eye dark circle brightening
- Eye/iris sharpening and brightening
- Subtle under-eye smoothing
Uses MediaPipe Face Mesh (468 landmarks) to precisely locate eye regions,
then applies targeted OpenCV adjustments. Processes video frame-by-frame.
Falls back to FFmpeg region-based filters if MediaPipe is not installed.
"""
from __future__ import annotations
import json
import time
from pathlib import Path
from typing import Any
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
ToolResult,
ToolStability,
ToolStatus,
ToolTier,
)
# MediaPipe Face Mesh landmark indices for eye regions.
# These form polygons around each eye area.
# Lower eyelid landmarks (used to define the under-eye region):
LEFT_LOWER_EYELID = [33, 7, 163, 144, 145, 153, 154, 155, 133]
RIGHT_LOWER_EYELID = [263, 249, 390, 373, 374, 380, 381, 382, 362]
# Full eye contour (used for iris/eye brightening):
LEFT_EYE = [33, 7, 163, 144, 145, 153, 154, 155, 133, 173, 157, 158, 159, 160, 161, 246]
RIGHT_EYE = [263, 249, 390, 373, 374, 380, 381, 382, 362, 398, 384, 385, 386, 387, 388, 466]
# Iris landmarks (available when refine_landmarks=True, indices 468-477):
LEFT_IRIS = [468, 469, 470, 471, 472]
RIGHT_IRIS = [473, 474, 475, 476, 477]
class EyeEnhance(BaseTool):
name = "eye_enhance"
version = "0.1.0"
tier = ToolTier.ENHANCE
capability = "enhancement"
provider = "mediapipe"
stability = ToolStability.EXPERIMENTAL
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
dependencies = ["cmd:ffmpeg"]
install_instructions = (
"For best results install MediaPipe and OpenCV:\n"
"pip install mediapipe opencv-python numpy\n\n"
"Without MediaPipe, falls back to FFmpeg eye-region filter (less precise)."
)
agent_skills = ["ffmpeg"]
capabilities = [
"under_eye_brightening",
"dark_circle_removal",
"eye_sharpening",
"eye_brightening",
]
input_schema = {
"type": "object",
"required": ["input_path"],
"properties": {
"input_path": {"type": "string"},
"output_path": {"type": "string"},
"operations": {
"type": "array",
"items": {
"type": "string",
"enum": ["dark_circles", "brighten_eyes", "sharpen_eyes"],
},
"default": ["dark_circles", "brighten_eyes"],
"description": "Which enhancements to apply",
},
"dark_circle_intensity": {
"type": "number",
"default": 0.4,
"minimum": 0.0,
"maximum": 1.0,
"description": "Strength of dark circle removal (0=none, 1=max)",
},
"eye_brighten_intensity": {
"type": "number",
"default": 0.3,
"minimum": 0.0,
"maximum": 1.0,
"description": "Strength of eye brightening (0=none, 1=max)",
},
"sharpen_intensity": {
"type": "number",
"default": 0.3,
"minimum": 0.0,
"maximum": 1.0,
"description": "Strength of eye sharpening (0=none, 1=max)",
},
"codec": {"type": "string", "default": "libx264"},
"crf": {"type": "integer", "default": 18},
},
}
resource_profile = ResourceProfile(
cpu_cores=4, ram_mb=2048, vram_mb=0, disk_mb=4000, network_required=False
)
idempotency_key_fields = [
"input_path", "operations", "dark_circle_intensity",
"eye_brighten_intensity", "sharpen_intensity",
]
side_effects = ["writes enhanced video to output_path"]
user_visible_verification = [
"Compare eyes in before/after — enhancement should be subtle and natural",
"Check for artifacts around eye region (halos, color shifts)",
"Verify enhancement doesn't make eyes look unnatural",
]
def _has_mediapipe(self) -> bool:
try:
import mediapipe # noqa: F401
return True
except ImportError:
return False
def _has_opencv(self) -> bool:
try:
import cv2 # noqa: F401
import numpy # noqa: F401
return True
except ImportError:
return False
def get_status(self) -> ToolStatus:
if self._has_mediapipe() and self._has_opencv():
return ToolStatus.AVAILABLE
if self._has_opencv():
return ToolStatus.DEGRADED
return ToolStatus.UNAVAILABLE
def execute(self, inputs: dict[str, Any]) -> ToolResult:
input_path = Path(inputs["input_path"])
if not input_path.exists():
return ToolResult(success=False, error=f"Input not found: {input_path}")
operations = inputs.get("operations", ["dark_circles", "brighten_eyes"])
output_path = Path(
inputs.get("output_path", str(input_path.with_stem(f"{input_path.stem}_eye_enhanced")))
)
output_path.parent.mkdir(parents=True, exist_ok=True)
start = time.time()
if self._has_mediapipe() and self._has_opencv():
result = self._enhance_mediapipe(input_path, output_path, inputs)
elif self._has_opencv():
result = self._enhance_opencv_only(input_path, output_path, inputs)
else:
result = self._enhance_ffmpeg_fallback(input_path, output_path, inputs)
if not result.success:
return result
elapsed = time.time() - start
result.duration_seconds = round(elapsed, 2)
return result
def _enhance_mediapipe(
self, input_path: Path, output_path: Path, inputs: dict[str, Any]
) -> ToolResult:
"""Full MediaPipe Face Mesh + OpenCV pipeline for precise eye enhancement."""
import cv2
import numpy as np
import mediapipe as mp
operations = inputs.get("operations", ["dark_circles", "brighten_eyes"])
dark_intensity = inputs.get("dark_circle_intensity", 0.4)
brighten_intensity = inputs.get("eye_brighten_intensity", 0.3)
sharpen_intensity = inputs.get("sharpen_intensity", 0.3)
codec_fourcc = inputs.get("codec", "libx264")
crf = inputs.get("crf", 18)
mp_face_mesh = mp.solutions.face_mesh
cap = cv2.VideoCapture(str(input_path))
fps = cap.get(cv2.CAP_PROP_FPS) or 30.0
width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))
height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
# Write to temp file, then mux audio via FFmpeg
temp_video = output_path.parent / f".{output_path.stem}_temp.mp4"
fourcc = cv2.VideoWriter_fourcc(*"mp4v")
writer = cv2.VideoWriter(str(temp_video), fourcc, fps, (width, height))
frames_processed = 0
frames_enhanced = 0
with mp_face_mesh.FaceMesh(
static_image_mode=False,
max_num_faces=1,
refine_landmarks=True, # Enables iris landmarks (468-477)
min_detection_confidence=0.5,
min_tracking_confidence=0.5,
) as face_mesh:
while cap.isOpened():
ret, frame = cap.read()
if not ret:
break
frames_processed += 1
rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
results = face_mesh.process(rgb)
if results.multi_face_landmarks:
landmarks = results.multi_face_landmarks[0]
frame = self._apply_eye_enhancements(
frame, landmarks, width, height,
operations, dark_intensity, brighten_intensity, sharpen_intensity,
)
frames_enhanced += 1
writer.write(frame)
cap.release()
writer.release()
# Mux original audio back with FFmpeg
cmd = [
"ffmpeg", "-y",
"-i", str(temp_video),
"-i", str(input_path),
"-c:v", "libx264", "-crf", str(crf), "-preset", "fast",
"-c:a", "aac", "-b:a", "192k",
"-map", "0:v:0", "-map", "1:a:0?",
"-shortest",
str(output_path),
]
try:
self.run_command(cmd, timeout=600)
except Exception as e:
return ToolResult(success=False, error=f"Audio mux failed: {e}")
finally:
if temp_video.exists():
temp_video.unlink()
return ToolResult(
success=True,
data={
"input": str(input_path),
"output": str(output_path),
"method": "mediapipe_face_mesh",
"frames_processed": frames_processed,
"frames_enhanced": frames_enhanced,
"operations": operations,
},
artifacts=[str(output_path)],
)
def _apply_eye_enhancements(
self,
frame,
landmarks,
width: int,
height: int,
operations: list[str],
dark_intensity: float,
brighten_intensity: float,
sharpen_intensity: float,
):
"""Apply eye enhancements to a single frame using detected landmarks."""
import cv2
import numpy as np
result = frame.copy()
# Convert landmarks to pixel coordinates
def lm_to_px(indices):
points = []
for idx in indices:
if idx < len(landmarks.landmark):
lm = landmarks.landmark[idx]
points.append((int(lm.x * width), int(lm.y * height)))
return np.array(points, dtype=np.int32)
for side in ["left", "right"]:
lower_lid = lm_to_px(LEFT_LOWER_EYELID if side == "left" else RIGHT_LOWER_EYELID)
eye_contour = lm_to_px(LEFT_EYE if side == "left" else RIGHT_EYE)
if len(lower_lid) < 3 or len(eye_contour) < 3:
continue
# Under-eye region: expand lower eyelid downward
if "dark_circles" in operations:
result = self._remove_dark_circles(
result, lower_lid, width, height, dark_intensity
)
if "brighten_eyes" in operations:
iris = lm_to_px(LEFT_IRIS if side == "left" else RIGHT_IRIS)
result = self._brighten_eyes(
result, eye_contour, iris, brighten_intensity
)
if "sharpen_eyes" in operations:
result = self._sharpen_eyes(
result, eye_contour, sharpen_intensity
)
return result
def _remove_dark_circles(
self, frame, lower_lid_points, width: int, height: int, intensity: float
):
"""Brighten the under-eye area to reduce dark circles."""
import cv2
import numpy as np
# Create under-eye region by shifting lower eyelid points downward
under_eye = lower_lid_points.copy()
# Shift down by ~15% of face height (approximate eye-to-cheek distance)
shift = max(5, int(height * 0.025))
under_eye_shifted = under_eye.copy()
under_eye_shifted[:, 1] += shift
# Create polygon from lower lid + shifted points (forms a band)
polygon = np.vstack([lower_lid_points, under_eye_shifted[::-1]])
# Create soft mask
mask = np.zeros(frame.shape[:2], dtype=np.float32)
cv2.fillPoly(mask, [polygon], 1.0)
# Gaussian blur for soft edges
blur_size = max(15, int(width * 0.02)) | 1 # Ensure odd
mask = cv2.GaussianBlur(mask, (blur_size, blur_size), 0)
# Apply brightening in LAB color space (perceptually uniform)
lab = cv2.cvtColor(frame, cv2.COLOR_BGR2LAB).astype(np.float32)
# Boost L channel (lightness) in the masked region
boost = 25 * intensity # 0-25 range
lab[:, :, 0] += mask * boost
lab[:, :, 0] = np.clip(lab[:, :, 0], 0, 255)
# Also reduce saturation slightly (dark circles are often purplish)
hsv = cv2.cvtColor(frame, cv2.COLOR_BGR2HSV).astype(np.float32)
hsv[:, :, 1] -= mask * (20 * intensity)
hsv[:, :, 1] = np.clip(hsv[:, :, 1], 0, 255)
# Blend: use LAB for brightness, HSV for desaturation
brightened = cv2.cvtColor(lab.astype(np.uint8), cv2.COLOR_LAB2BGR)
desaturated = cv2.cvtColor(hsv.astype(np.uint8), cv2.COLOR_HSV2BGR)
# Combine: weighted blend of both adjustments
mask_3ch = np.stack([mask] * 3, axis=-1)
combined = frame.astype(np.float32)
combined = combined * (1 - mask_3ch * intensity) + \
(brightened.astype(np.float32) * 0.6 + desaturated.astype(np.float32) * 0.4) * (mask_3ch * intensity)
return np.clip(combined, 0, 255).astype(np.uint8)
def _brighten_eyes(self, frame, eye_contour, iris_points, intensity: float):
"""Subtly brighten the eye/sclera area."""
import cv2
import numpy as np
if len(eye_contour) < 3:
return frame
# Create mask from eye contour
mask = np.zeros(frame.shape[:2], dtype=np.float32)
cv2.fillPoly(mask, [eye_contour], 1.0)
blur_size = max(5, int(frame.shape[1] * 0.005)) | 1
mask = cv2.GaussianBlur(mask, (blur_size, blur_size), 0)
# Brighten in LAB space
lab = cv2.cvtColor(frame, cv2.COLOR_BGR2LAB).astype(np.float32)
boost = 15 * intensity
lab[:, :, 0] += mask * boost
lab[:, :, 0] = np.clip(lab[:, :, 0], 0, 255)
brightened = cv2.cvtColor(lab.astype(np.uint8), cv2.COLOR_LAB2BGR)
mask_3ch = np.stack([mask] * 3, axis=-1)
result = frame.astype(np.float32) * (1 - mask_3ch * intensity) + \
brightened.astype(np.float32) * (mask_3ch * intensity)
return np.clip(result, 0, 255).astype(np.uint8)
def _sharpen_eyes(self, frame, eye_contour, intensity: float):
"""Sharpen the eye region for more detail."""
import cv2
import numpy as np
if len(eye_contour) < 3:
return frame
# Create mask
mask = np.zeros(frame.shape[:2], dtype=np.float32)
cv2.fillPoly(mask, [eye_contour], 1.0)
# Expand slightly for natural blend
kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (5, 5))
mask = cv2.dilate(mask, kernel, iterations=1)
blur_size = max(5, int(frame.shape[1] * 0.005)) | 1
mask = cv2.GaussianBlur(mask, (blur_size, blur_size), 0)
# Unsharp mask for sharpening
blur = cv2.GaussianBlur(frame, (0, 0), 3)
sharpened = cv2.addWeighted(frame, 1.0 + intensity, blur, -intensity, 0)
mask_3ch = np.stack([mask] * 3, axis=-1)
result = frame.astype(np.float32) * (1 - mask_3ch) + \
sharpened.astype(np.float32) * mask_3ch
return np.clip(result, 0, 255).astype(np.uint8)
def _enhance_opencv_only(
self, input_path: Path, output_path: Path, inputs: dict[str, Any]
) -> ToolResult:
"""Fallback: OpenCV Haar cascade for face detection + generic eye region enhancement."""
import cv2
import numpy as np
operations = inputs.get("operations", ["dark_circles", "brighten_eyes"])
dark_intensity = inputs.get("dark_circle_intensity", 0.4)
crf = inputs.get("crf", 18)
face_cascade = cv2.CascadeClassifier(
cv2.data.haarcascades + "haarcascade_frontalface_default.xml"
)
eye_cascade = cv2.CascadeClassifier(
cv2.data.haarcascades + "haarcascade_eye.xml"
)
cap = cv2.VideoCapture(str(input_path))
fps = cap.get(cv2.CAP_PROP_FPS) or 30.0
width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))
height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
temp_video = output_path.parent / f".{output_path.stem}_temp.mp4"
fourcc = cv2.VideoWriter_fourcc(*"mp4v")
writer = cv2.VideoWriter(str(temp_video), fourcc, fps, (width, height))
frames_processed = 0
frames_enhanced = 0
while cap.isOpened():
ret, frame = cap.read()
if not ret:
break
frames_processed += 1
gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)
faces = face_cascade.detectMultiScale(gray, 1.1, 5, minSize=(60, 60))
if len(faces) > 0:
# Use largest face
areas = [w * h for (_, _, w, h) in faces]
fi = areas.index(max(areas))
fx, fy, fw, fh = faces[fi]
# Detect eyes within face region
face_roi_gray = gray[fy:fy + fh, fx:fx + fw]
eyes = eye_cascade.detectMultiScale(face_roi_gray, 1.1, 5, minSize=(20, 20))
for (ex, ey, ew, eh) in eyes:
# Under-eye region: below the detected eye box
under_y = fy + ey + eh
under_h = max(5, int(eh * 0.4))
under_x = fx + ex
under_w = ew
if under_y + under_h <= height and under_x + under_w <= width:
if "dark_circles" in operations:
# Create soft mask for under-eye
mask = np.zeros((height, width), dtype=np.float32)
cv2.ellipse(
mask,
(under_x + under_w // 2, under_y + under_h // 2),
(under_w // 2, under_h // 2),
0, 0, 360, 1.0, -1,
)
mask = cv2.GaussianBlur(mask, (15, 15), 0)
lab = cv2.cvtColor(frame, cv2.COLOR_BGR2LAB).astype(np.float32)
lab[:, :, 0] += mask * (25 * dark_intensity)
lab[:, :, 0] = np.clip(lab[:, :, 0], 0, 255)
frame = cv2.cvtColor(lab.astype(np.uint8), cv2.COLOR_LAB2BGR)
frames_enhanced += 1
writer.write(frame)
cap.release()
writer.release()
# Mux audio
cmd = [
"ffmpeg", "-y",
"-i", str(temp_video),
"-i", str(input_path),
"-c:v", "libx264", "-crf", str(crf), "-preset", "fast",
"-c:a", "aac", "-b:a", "192k",
"-map", "0:v:0", "-map", "1:a:0?",
"-shortest",
str(output_path),
]
try:
self.run_command(cmd, timeout=600)
except Exception as e:
return ToolResult(success=False, error=f"Audio mux failed: {e}")
finally:
if temp_video.exists():
temp_video.unlink()
return ToolResult(
success=True,
data={
"input": str(input_path),
"output": str(output_path),
"method": "opencv_haar_cascade",
"frames_processed": frames_processed,
"frames_enhanced": frames_enhanced,
"operations": operations,
},
artifacts=[str(output_path)],
)
def _enhance_ffmpeg_fallback(
self, input_path: Path, output_path: Path, inputs: dict[str, Any]
) -> ToolResult:
"""Last resort: FFmpeg-only. Applies general face-area enhancement (not eye-specific)."""
crf = inputs.get("crf", 18)
intensity = inputs.get("dark_circle_intensity", 0.4)
# General brightness + contrast lift for the whole frame
# Not eye-specific but better than nothing
brightness = 0.02 * intensity
contrast = 1.0 + (0.05 * intensity)
cmd = [
"ffmpeg", "-y",
"-i", str(input_path),
"-vf", f"eq=brightness={brightness}:contrast={contrast}",
"-c:v", "libx264", "-crf", str(crf), "-preset", "fast",
"-c:a", "copy",
str(output_path),
]
try:
self.run_command(cmd, timeout=600)
except Exception as e:
return ToolResult(success=False, error=f"FFmpeg fallback failed: {e}")
return ToolResult(
success=True,
data={
"input": str(input_path),
"output": str(output_path),
"method": "ffmpeg_global_brightness",
"operations": ["global_brightness_contrast"],
"note": "Install mediapipe + opencv-python for precise eye-region enhancement",
},
artifacts=[str(output_path)],
)
def estimate_runtime(self, inputs: dict[str, Any]) -> float:
"""Eye enhancement is roughly 0.5x-1x realtime depending on resolution."""
return 90.0
+51
View File
@@ -60,6 +60,15 @@ class SubtitleGen(BaseTool):
"enum": ["none", "word_by_word", "karaoke"],
"default": "none",
},
"corrections": {
"type": "object",
"description": (
"Dictionary of word corrections for common ASR misrecognitions. "
"Keys are the wrong word (case-insensitive), values are the "
"correct replacement. Applied before generating subtitles. "
"Example: {\"cloud\": \"Claude\", \"co-pilot\": \"Copilot\"}."
),
},
},
}
@@ -77,9 +86,14 @@ class SubtitleGen(BaseTool):
max_chars = inputs.get("max_chars_per_line", 42)
highlight_style = inputs.get("highlight_style", "none")
output_path = inputs.get("output_path")
corrections = inputs.get("corrections")
start = time.time()
# Apply word corrections if provided
if corrections:
segments = self._apply_corrections(segments, corrections)
# Build cues from word-level timestamps
cues = self._build_cues(segments, max_words, max_chars)
@@ -114,6 +128,43 @@ class SubtitleGen(BaseTool):
duration_seconds=round(elapsed, 2),
)
@staticmethod
def _apply_corrections(
segments: list[dict], corrections: dict[str, str]
) -> list[dict]:
"""Apply word-level corrections to transcript segments.
Handles case-insensitive matching and preserves punctuation.
"""
import copy
corr = {k.lower(): v for k, v in corrections.items()}
result = copy.deepcopy(segments)
for seg in result:
words = seg.get("words", [])
for w in words:
raw = w.get("word", "").strip()
# Strip punctuation for lookup, preserve it
stripped = raw.lower().rstrip(".,!?;:'\"")
if stripped in corr:
trailing = raw[len(stripped):]
w["word"] = corr[stripped] + trailing
# Also fix segment-level text
if "text" in seg and words:
seg["text"] = " ".join(w["word"] for w in words)
elif "text" in seg:
for wrong, right in corr.items():
import re as _re
seg["text"] = _re.sub(
r"\b" + _re.escape(wrong) + r"\b",
right,
seg["text"],
flags=_re.IGNORECASE,
)
return result
def _build_cues(
self, segments: list[dict], max_words: int, max_chars: int
) -> list[dict]:
+537
View File
@@ -0,0 +1,537 @@
"""Auto-reframe tool for aspect ratio conversion with face tracking.
Converts video between aspect ratios (e.g. 16:9 → 9:16 for Instagram Reels)
while keeping the speaker's face centered in frame. Uses face_tracker data
for smooth, content-aware cropping.
Primary use: converting talking-head footage shot in landscape to vertical
format for social media (TikTok, Reels, Shorts).
Approach: MediaPipe/OpenCV face detection → smoothed bounding box trajectory
→ FFmpeg crop filter. No GPU required.
"""
from __future__ import annotations
import json
import time
from pathlib import Path
from typing import Any
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
RetryPolicy,
ResumeSupport,
ToolResult,
ToolStability,
ToolTier,
)
# Common target aspect ratios
ASPECT_PRESETS = {
"portrait": (9, 16), # Instagram Reels, TikTok, YouTube Shorts
"square": (1, 1), # Instagram Feed
"landscape": (16, 9), # YouTube, LinkedIn
"cinematic": (21, 9), # Ultra-wide
"vertical_4_5": (4, 5), # Instagram portrait post
}
class AutoReframe(BaseTool):
name = "auto_reframe"
version = "0.1.0"
tier = ToolTier.CORE
capability = "video_post"
provider = "ffmpeg"
stability = ToolStability.EXPERIMENTAL
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
dependencies = ["cmd:ffmpeg"]
install_instructions = (
"FFmpeg is required. For face-tracked reframing, also install:\n"
"pip install mediapipe opencv-python\n\n"
"Without MediaPipe/OpenCV, falls back to center-crop."
)
agent_skills = ["ffmpeg"]
capabilities = [
"aspect_ratio_conversion",
"face_tracked_crop",
"smart_reframe",
"center_crop",
]
input_schema = {
"type": "object",
"required": ["input_path"],
"properties": {
"input_path": {"type": "string"},
"output_path": {"type": "string"},
"target_aspect": {
"type": "string",
"enum": list(ASPECT_PRESETS.keys()),
"default": "portrait",
"description": "Target aspect ratio preset",
},
"target_width": {
"type": "integer",
"description": "Explicit target width (overrides preset)",
},
"target_height": {
"type": "integer",
"description": "Explicit target height (overrides preset)",
},
"face_tracking_json": {
"type": "string",
"description": "Path to pre-computed face_tracker JSON. If omitted, runs face detection internally.",
},
"smoothing_window": {
"type": "integer",
"default": 15,
"minimum": 1,
"description": "Number of frames for position smoothing (higher = smoother pan, lower = more responsive)",
},
"face_padding": {
"type": "number",
"default": 0.4,
"minimum": 0.0,
"maximum": 1.0,
"description": "Extra space around face as fraction of face size (0.4 = 40% padding)",
},
"sample_fps": {
"type": "number",
"default": 5,
"description": "Face detection sample rate (only used if no face_tracking_json)",
},
"codec": {"type": "string", "default": "libx264"},
"crf": {"type": "integer", "default": 18},
},
}
resource_profile = ResourceProfile(
cpu_cores=4, ram_mb=2048, vram_mb=0, disk_mb=4000, network_required=False
)
retry_policy = RetryPolicy(max_retries=1, retryable_errors=["FFmpeg error"])
resume_support = ResumeSupport.FROM_START
idempotency_key_fields = [
"input_path", "target_aspect", "target_width", "target_height",
"smoothing_window", "face_padding",
]
side_effects = ["writes reframed video to output_path"]
user_visible_verification = [
"Play reframed output — verify face stays centered and framing is smooth",
"Check that no important content is cropped out",
]
def execute(self, inputs: dict[str, Any]) -> ToolResult:
input_path = Path(inputs["input_path"])
if not input_path.exists():
return ToolResult(success=False, error=f"Input not found: {input_path}")
start = time.time()
# Get source video dimensions
src_w, src_h, src_fps = self._get_video_info(input_path)
if src_w == 0 or src_h == 0:
return ToolResult(success=False, error="Could not read video dimensions")
# Determine target crop dimensions (in source pixel space)
target_w, target_h = self._compute_crop_size(inputs, src_w, src_h)
# If source already matches target aspect, no crop needed
if target_w == src_w and target_h == src_h:
return ToolResult(
success=True,
data={"message": "Source already matches target aspect ratio", "output": str(input_path)},
artifacts=[str(input_path)],
)
# Get face tracking data
face_data = self._get_face_data(inputs, input_path, src_fps)
# Compute per-frame crop positions
if face_data and len(face_data) > 0:
crop_x, crop_y = self._compute_face_tracked_crop(
face_data, src_w, src_h, target_w, target_h,
src_fps,
inputs.get("smoothing_window", 15),
inputs.get("face_padding", 0.4),
)
method = "face_tracked"
else:
# Fallback: center crop
crop_x = (src_w - target_w) // 2
crop_y = (src_h - target_h) // 2
method = "center_crop"
# Determine output resolution
out_w, out_h = self._compute_output_resolution(inputs, target_w, target_h, src_w, src_h)
# Build output path
aspect_name = inputs.get("target_aspect", "portrait")
output_path = Path(
inputs.get("output_path", str(input_path.with_stem(f"{input_path.stem}_{aspect_name}")))
)
output_path.parent.mkdir(parents=True, exist_ok=True)
# Render via FFmpeg
codec = inputs.get("codec", "libx264")
crf = inputs.get("crf", 18)
if method == "face_tracked" and isinstance(crop_x, list):
# Dynamic crop: write crop coordinates to a file and use sendcmd
result = self._render_dynamic_crop(
input_path, output_path, crop_x, crop_y,
target_w, target_h, out_w, out_h,
src_fps, codec, crf,
)
else:
# Static crop
result = self._render_static_crop(
input_path, output_path,
crop_x, crop_y, target_w, target_h,
out_w, out_h, codec, crf,
)
if not result.success:
return result
elapsed = time.time() - start
return ToolResult(
success=True,
data={
"input": str(input_path),
"output": str(output_path),
"source_resolution": f"{src_w}x{src_h}",
"crop_resolution": f"{target_w}x{target_h}",
"output_resolution": f"{out_w}x{out_h}",
"method": method,
"target_aspect": inputs.get("target_aspect", "portrait"),
},
artifacts=[str(output_path)],
duration_seconds=round(elapsed, 2),
)
def _get_video_info(self, path: Path) -> tuple[int, int, float]:
"""Get video width, height, fps via ffprobe."""
cmd = [
"ffprobe", "-v", "quiet",
"-select_streams", "v:0",
"-show_entries", "stream=width,height,r_frame_rate",
"-of", "json", str(path),
]
try:
result = self.run_command(cmd)
data = json.loads(result.stdout)
stream = data["streams"][0]
w = int(stream["width"])
h = int(stream["height"])
# Parse r_frame_rate (e.g. "30000/1001")
fps_parts = stream["r_frame_rate"].split("/")
fps = float(fps_parts[0]) / float(fps_parts[1]) if len(fps_parts) == 2 else float(fps_parts[0])
return w, h, fps
except Exception:
return 0, 0, 30.0
def _compute_crop_size(
self, inputs: dict[str, Any], src_w: int, src_h: int
) -> tuple[int, int]:
"""Compute crop dimensions in source pixel space that match the target aspect ratio."""
if "target_width" in inputs and "target_height" in inputs:
# Explicit dimensions — compute crop in source space matching this ratio
tw, th = inputs["target_width"], inputs["target_height"]
else:
aspect_name = inputs.get("target_aspect", "portrait")
tw, th = ASPECT_PRESETS.get(aspect_name, (9, 16))
target_ratio = tw / th
src_ratio = src_w / src_h
if target_ratio > src_ratio:
# Target is wider — crop height
crop_w = src_w
crop_h = int(src_w / target_ratio)
else:
# Target is taller/narrower — crop width
crop_h = src_h
crop_w = int(src_h * target_ratio)
# Ensure even dimensions (required by most codecs)
crop_w = crop_w - (crop_w % 2)
crop_h = crop_h - (crop_h % 2)
return crop_w, crop_h
def _compute_output_resolution(
self, inputs: dict[str, Any],
crop_w: int, crop_h: int,
src_w: int, src_h: int,
) -> tuple[int, int]:
"""Determine final output resolution. Scales to standard sizes."""
if "target_width" in inputs and "target_height" in inputs:
out_w = inputs["target_width"]
out_h = inputs["target_height"]
else:
aspect_name = inputs.get("target_aspect", "portrait")
if aspect_name == "portrait":
out_w, out_h = 1080, 1920
elif aspect_name == "square":
out_w, out_h = 1080, 1080
elif aspect_name == "landscape":
out_w, out_h = 1920, 1080
elif aspect_name == "cinematic":
out_w, out_h = 2560, 1080
elif aspect_name == "vertical_4_5":
out_w, out_h = 1080, 1350
else:
out_w, out_h = crop_w, crop_h
# Ensure even
out_w = out_w - (out_w % 2)
out_h = out_h - (out_h % 2)
return out_w, out_h
def _get_face_data(
self, inputs: dict[str, Any], input_path: Path, src_fps: float
) -> list[dict]:
"""Get face tracking data — from pre-computed JSON or by running detection."""
# Check for pre-computed tracking data
tracking_json = inputs.get("face_tracking_json")
if tracking_json:
p = Path(tracking_json)
if p.exists():
data = json.loads(p.read_text(encoding="utf-8"))
return data.get("faces", [])
# Try to run face_tracker internally
try:
from tools.analysis.face_tracker import FaceTracker
tracker = FaceTracker()
if tracker.get_status().name == "UNAVAILABLE":
return []
sample_fps = inputs.get("sample_fps", 5)
result = tracker.execute({
"input_path": str(input_path),
"sample_fps": sample_fps,
})
if result.success and result.data:
# Read the generated JSON
output_file = result.data.get("output")
if output_file:
data = json.loads(Path(output_file).read_text(encoding="utf-8"))
return data.get("faces", [])
except Exception:
pass
return []
def _compute_face_tracked_crop(
self,
faces: list[dict],
src_w: int, src_h: int,
crop_w: int, crop_h: int,
fps: float,
smoothing_window: int,
face_padding: float,
) -> tuple[list[int], list[int]]:
"""Compute smoothed crop positions from face tracking data.
Returns a single (x, y) if face positions are stable enough,
or lists of per-frame positions for dynamic cropping.
"""
if not faces:
cx = (src_w - crop_w) // 2
cy = (src_h - crop_h) // 2
return cx, cy
# Convert relative bbox centers to pixel positions
face_centers_x = []
face_centers_y = []
face_timestamps = []
for f in faces:
bbox = f["bbox"]
# Center of face in pixel space
center_x = (bbox["x"] + bbox["width"] / 2) * src_w
center_y = (bbox["y"] + bbox["height"] / 2) * src_h
face_centers_x.append(center_x)
face_centers_y.append(center_y)
face_timestamps.append(f["timestamp_seconds"])
# Check if face position is stable (talking head usually is)
x_range = max(face_centers_x) - min(face_centers_x)
y_range = max(face_centers_y) - min(face_centers_y)
# If face barely moves (<10% of frame), use a single static crop
if x_range < src_w * 0.10 and y_range < src_h * 0.10:
avg_x = sum(face_centers_x) / len(face_centers_x)
avg_y = sum(face_centers_y) / len(face_centers_y)
# Position crop window centered on face, with bias toward upper third
crop_x = int(avg_x - crop_w / 2)
crop_y = int(avg_y - crop_h * 0.35) # Face in upper 35% of frame
# Clamp to frame bounds
crop_x = max(0, min(crop_x, src_w - crop_w))
crop_y = max(0, min(crop_y, src_h - crop_h))
return crop_x, crop_y
# Dynamic crop: smooth the trajectory
smoothed_x = self._smooth_positions(face_centers_x, smoothing_window)
smoothed_y = self._smooth_positions(face_centers_y, smoothing_window)
# Convert to crop positions (top-left corner), clamped
crop_xs = []
crop_ys = []
for sx, sy in zip(smoothed_x, smoothed_y):
cx = int(sx - crop_w / 2)
cy = int(sy - crop_h * 0.35)
cx = max(0, min(cx, src_w - crop_w))
cy = max(0, min(cy, src_h - crop_h))
crop_xs.append(cx)
crop_ys.append(cy)
return crop_xs, crop_ys
def _smooth_positions(self, values: list[float], window: int) -> list[float]:
"""Simple moving average smoothing."""
smoothed = []
for i in range(len(values)):
start = max(0, i - window // 2)
end = min(len(values), i + window // 2 + 1)
smoothed.append(sum(values[start:end]) / (end - start))
return smoothed
def _render_static_crop(
self,
input_path: Path, output_path: Path,
crop_x: int, crop_y: int,
crop_w: int, crop_h: int,
out_w: int, out_h: int,
codec: str, crf: int,
) -> ToolResult:
"""Render with a static crop position."""
vf = f"crop={crop_w}:{crop_h}:{crop_x}:{crop_y},scale={out_w}:{out_h}"
cmd = [
"ffmpeg", "-y",
"-i", str(input_path),
"-vf", vf,
"-c:v", codec, "-crf", str(crf), "-preset", "fast",
"-c:a", "aac", "-b:a", "192k",
str(output_path),
]
try:
self.run_command(cmd, timeout=600)
except Exception as e:
return ToolResult(success=False, error=f"FFmpeg render failed: {e}")
return ToolResult(success=True)
def _render_dynamic_crop(
self,
input_path: Path, output_path: Path,
crop_xs: list[int], crop_ys: list[int],
crop_w: int, crop_h: int,
out_w: int, out_h: int,
fps: float,
codec: str, crf: int,
) -> ToolResult:
"""Render with dynamic crop positions that follow the face.
Uses FFmpeg's sendcmd filter to update crop position over time.
For simplicity and reliability, we interpolate between key positions
using FFmpeg expression-based crop.
"""
# Build a piecewise-linear x(t) and y(t) using FFmpeg expressions
# We'll sample at the face tracking rate and interpolate between points
if not crop_xs:
return ToolResult(success=False, error="No crop positions computed")
# If very few data points, fall back to static using average
if len(crop_xs) < 3:
avg_x = int(sum(crop_xs) / len(crop_xs))
avg_y = int(sum(crop_ys) / len(crop_ys))
return self._render_static_crop(
input_path, output_path, avg_x, avg_y,
crop_w, crop_h, out_w, out_h, codec, crf,
)
# Build sendcmd script for crop filter position updates
# Each command sets the crop x,y at the corresponding timestamp
temp_dir = output_path.parent / ".reframe_tmp"
temp_dir.mkdir(parents=True, exist_ok=True)
sendcmd_path = temp_dir / "crop_commands.txt"
# Approximate timestamps from the face tracking sample rate
# The face data was sampled at sample_fps intervals
sample_interval = 1.0 / (fps / max(1, int(fps / 5))) # Approximate
lines = []
for i, (cx, cy) in enumerate(zip(crop_xs, crop_ys)):
ts = i * sample_interval
lines.append(f"{ts:.3f} [enter] crop x {cx};")
lines.append(f"{ts:.3f} [enter] crop y {cy};")
sendcmd_path.write_text("\n".join(lines), encoding="utf-8")
# Use crop with sendcmd for dynamic positioning
vf = (
f"sendcmd=f='{str(sendcmd_path).replace(chr(92), '/')}':flags=enter,"
f"crop={crop_w}:{crop_h}:{crop_xs[0]}:{crop_ys[0]},"
f"scale={out_w}:{out_h}"
)
cmd = [
"ffmpeg", "-y",
"-i", str(input_path),
"-vf", vf,
"-c:v", codec, "-crf", str(crf), "-preset", "fast",
"-c:a", "aac", "-b:a", "192k",
str(output_path),
]
try:
self.run_command(cmd, timeout=600)
except Exception:
# sendcmd can be finicky — fall back to static crop with average position
avg_x = int(sum(crop_xs) / len(crop_xs))
avg_y = int(sum(crop_ys) / len(crop_ys))
result = self._render_static_crop(
input_path, output_path, avg_x, avg_y,
crop_w, crop_h, out_w, out_h, codec, crf,
)
if result.success:
result.data = result.data or {}
result.data["fallback"] = "sendcmd failed, used static average crop"
return result
finally:
# Clean up
if sendcmd_path.exists():
sendcmd_path.unlink()
try:
temp_dir.rmdir()
except OSError:
pass
return ToolResult(success=True)
def estimate_runtime(self, inputs: dict[str, Any]) -> float:
"""Estimate runtime in seconds. Roughly 1x realtime for face tracking + render."""
return 60.0 # Conservative default
@staticmethod
def list_presets() -> dict[str, str]:
"""Return available aspect ratio presets."""
return {
name: f"{w}:{h}"
for name, (w, h) in ASPECT_PRESETS.items()
}
+439
View File
@@ -0,0 +1,439 @@
"""Remotion caption burn tool.
Renders animated word-by-word captions onto a talking-head video using
the Remotion CaptionOverlay component. Falls back to FFmpeg subtitle
burning if Remotion is not available.
The tool:
1. Converts word-level transcript segments to Remotion WordCaption JSON
2. Writes a props file for the TalkingHead composition
3. Renders via ``npx remotion render``
4. Returns the captioned video path
Fallback: if Remotion is unavailable, burns subtitles at the bottom of
the frame using FFmpeg's ``subtitles`` filter with bold styling.
"""
from __future__ import annotations
import json
import math
import re
import shutil
import time
from pathlib import Path
from typing import Any
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
ToolResult,
ToolStability,
ToolTier,
)
class RemotionCaptionBurn(BaseTool):
name = "remotion_caption_burn"
version = "0.1.0"
tier = ToolTier.CORE
capability = "subtitle"
provider = "remotion"
stability = ToolStability.EXPERIMENTAL
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
dependencies = ["cmd:ffmpeg", "cmd:ffprobe"]
install_instructions = (
"Remotion (optional, preferred): npm install in remotion-composer/\n"
"FFmpeg (required for fallback): https://ffmpeg.org/download.html"
)
agent_skills = ["remotion-best-practices", "ffmpeg"]
capabilities = [
"burn_remotion_captions",
"burn_ffmpeg_captions_fallback",
]
input_schema = {
"type": "object",
"required": ["input_path", "output_path"],
"properties": {
"input_path": {
"type": "string",
"description": "Path to the input video (enhanced talking-head footage).",
},
"output_path": {
"type": "string",
"description": "Path for the output video with captions burned in.",
},
"segments": {
"type": "array",
"description": (
"Word-level transcript segments from transcriber tool. "
"Each segment has 'words' array with {word, start, end}."
),
},
"srt_path": {
"type": "string",
"description": (
"Path to an SRT file. Used as an alternative to segments. "
"If both provided, segments take priority."
),
},
"words_per_page": {
"type": "integer",
"default": 4,
"description": "Words shown at once in the caption overlay.",
},
"font_size": {
"type": "integer",
"default": 52,
"description": "Font size for captions.",
},
"highlight_color": {
"type": "string",
"default": "#22D3EE",
"description": "Highlight color for the active word (hex).",
},
"corrections": {
"type": "object",
"description": (
"Dictionary of word corrections for common misrecognitions. "
"Keys are the wrong word (case-insensitive), values are the "
"correct replacement. Example: {\"cloud\": \"Claude\"}."
),
},
"force_ffmpeg": {
"type": "boolean",
"default": False,
"description": "Force FFmpeg fallback even if Remotion is available.",
},
},
}
resource_profile = ResourceProfile(cpu_cores=4, ram_mb=2048, vram_mb=0, disk_mb=500)
idempotency_key_fields = ["input_path", "segments", "srt_path"]
side_effects = ["writes captioned video to output_path"]
user_visible_verification = [
"Play the output video and verify captions appear at the bottom of the frame",
"Check that the active word is highlighted in the specified color",
"Verify face is not occluded by caption text",
]
# ------------------------------------------------------------------ #
# Remotion detection
# ------------------------------------------------------------------ #
def _find_remotion_root(self) -> Path | None:
"""Find the remotion-composer directory relative to the repo."""
candidates = [
Path.cwd() / "remotion-composer",
Path(__file__).resolve().parent.parent.parent / "remotion-composer",
]
for p in candidates:
if (
p.is_dir()
and (p / "package.json").exists()
and (p / "node_modules").is_dir()
):
return p
return None
def _remotion_available(self) -> bool:
return (
shutil.which("npx") is not None
and self._find_remotion_root() is not None
)
# ------------------------------------------------------------------ #
# Word caption conversion
# ------------------------------------------------------------------ #
def _segments_to_word_captions(
self, segments: list[dict], corrections: dict[str, str] | None = None
) -> list[dict]:
"""Convert transcriber segments to [{word, startMs, endMs}, ...]."""
captions: list[dict] = []
corr = {k.lower(): v for k, v in (corrections or {}).items()}
for seg in segments:
words = seg.get("words", [])
if words:
for w in words:
raw = w["word"].strip()
fixed = corr.get(raw.lower().strip(".,!?;:"), raw)
# Preserve trailing punctuation from original
trailing = ""
if raw and raw[-1] in ".,!?;:":
trailing = raw[-1]
if fixed != raw and not fixed.endswith(trailing):
fixed = fixed + trailing
captions.append({
"word": fixed,
"startMs": int(w["start"] * 1000),
"endMs": int(w["end"] * 1000),
})
elif "text" in seg:
text_words = seg["text"].strip().split()
dur = seg["end"] - seg["start"]
per_word = dur / max(len(text_words), 1)
for i, tw in enumerate(text_words):
fixed = corr.get(tw.lower().strip(".,!?;:"), tw)
captions.append({
"word": fixed,
"startMs": int((seg["start"] + i * per_word) * 1000),
"endMs": int((seg["start"] + (i + 1) * per_word) * 1000),
})
return captions
def _srt_to_word_captions(
self, srt_path: str, corrections: dict[str, str] | None = None
) -> list[dict]:
"""Parse SRT file into word captions."""
content = Path(srt_path).read_text(encoding="utf-8")
blocks = re.split(r"\n\n+", content.strip())
corr = {k.lower(): v for k, v in (corrections or {}).items()}
captions: list[dict] = []
for block in blocks:
lines = block.strip().split("\n")
if len(lines) < 3:
continue
m = re.match(
r"(\d{2}):(\d{2}):(\d{2}),(\d{3})\s*-->\s*"
r"(\d{2}):(\d{2}):(\d{2}),(\d{3})",
lines[1],
)
if not m:
continue
start_ms = (
int(m.group(1)) * 3600000
+ int(m.group(2)) * 60000
+ int(m.group(3)) * 1000
+ int(m.group(4))
)
end_ms = (
int(m.group(5)) * 3600000
+ int(m.group(6)) * 60000
+ int(m.group(7)) * 1000
+ int(m.group(8))
)
text = " ".join(lines[2:]).strip()
words = text.split()
per_word = (end_ms - start_ms) / max(len(words), 1)
for i, w in enumerate(words):
fixed = corr.get(w.lower().strip(".,!?;:"), w)
captions.append({
"word": fixed,
"startMs": int(start_ms + i * per_word),
"endMs": int(start_ms + (i + 1) * per_word),
})
return captions
# ------------------------------------------------------------------ #
# Remotion render
# ------------------------------------------------------------------ #
def _render_remotion(
self,
input_path: str,
output_path: str,
captions: list[dict],
words_per_page: int,
font_size: int,
highlight_color: str,
) -> ToolResult:
root = self._find_remotion_root()
if root is None:
return ToolResult(success=False, error="Remotion root not found")
# Get video duration in frames
dur_cmd = [
"ffprobe", "-v", "error",
"-show_entries", "format=duration",
"-of", "csv=p=0",
input_path,
]
dur_out = self.run_command(dur_cmd, capture=True)
duration_s = float(dur_out.strip().split("\n")[0])
total_frames = math.ceil(duration_s * 30)
# Copy video to Remotion public folder
pub_dir = root / "public" / "talking-head"
pub_dir.mkdir(parents=True, exist_ok=True)
video_filename = Path(input_path).name
dest_video = pub_dir / video_filename
shutil.copy2(input_path, dest_video)
# Build props JSON
props = {
"videoSrc": f"public/talking-head/{video_filename}",
"captions": captions,
"wordsPerPage": words_per_page,
"fontSize": font_size,
"highlightColor": highlight_color,
}
props_dir = root / "public" / "demo-props"
props_dir.mkdir(parents=True, exist_ok=True)
props_file = props_dir / f"caption-burn-{Path(input_path).stem}.json"
props_file.write_text(json.dumps(props, indent=2), encoding="utf-8")
# Render
render_cmd = [
"npx", "remotion", "render",
"src/index.tsx", "TalkingHead",
f"--props={props_file.relative_to(root)}",
"--width=1080", "--height=1920", "--fps=30",
f"--frames=0-{total_frames - 1}",
"--codec=h264", "--crf=18",
str(Path(output_path).resolve()),
]
self.run_command(render_cmd, cwd=str(root))
if not Path(output_path).exists():
return ToolResult(success=False, error="Remotion render produced no output")
return ToolResult(
success=True,
data={
"method": "remotion",
"output": output_path,
"duration_seconds": round(duration_s, 2),
"total_frames": total_frames,
"caption_count": len(captions),
"words_per_page": words_per_page,
},
artifacts=[output_path],
)
# ------------------------------------------------------------------ #
# FFmpeg fallback
# ------------------------------------------------------------------ #
def _render_ffmpeg(
self,
input_path: str,
output_path: str,
captions: list[dict],
) -> ToolResult:
"""Fall back to FFmpeg subtitle burning at bottom of frame."""
# Generate temporary SRT from word captions
tmp_srt = Path(output_path).parent / f"_tmp_captions_{int(time.time())}.srt"
tmp_srt.parent.mkdir(parents=True, exist_ok=True)
srt_lines = []
idx = 1
# Group into pages of ~4 words
page_size = 4
for i in range(0, len(captions), page_size):
page = captions[i : i + page_size]
text = " ".join(c["word"] for c in page)
start = page[0]["startMs"]
end = page[-1]["endMs"]
srt_lines.append(str(idx))
srt_lines.append(
f"{self._ms_to_srt(start)} --> {self._ms_to_srt(end)}"
)
srt_lines.append(text)
srt_lines.append("")
idx += 1
tmp_srt.write_text("\n".join(srt_lines), encoding="utf-8")
# Escape path for FFmpeg subtitles filter (Windows colon issue)
srt_escaped = str(tmp_srt).replace("\\", "/").replace(":", "\\:")
cmd = [
"ffmpeg", "-y",
"-i", input_path,
"-vf", (
f"subtitles='{srt_escaped}'"
":force_style='FontName=Segoe UI,FontSize=24,Bold=1,"
"PrimaryColour=&H00FFFFFF,OutlineColour=&H00000000,"
"Outline=3,Shadow=2,Alignment=2,MarginV=100'"
),
"-c:v", "libx264", "-preset", "fast", "-crf", "18",
"-pix_fmt", "yuv420p",
"-c:a", "copy",
output_path,
]
self.run_command(cmd)
# Clean up temp SRT
try:
tmp_srt.unlink()
except OSError:
pass
if not Path(output_path).exists():
return ToolResult(success=False, error="FFmpeg subtitle burn produced no output")
return ToolResult(
success=True,
data={
"method": "ffmpeg_fallback",
"output": output_path,
"caption_count": len(captions),
"note": "Used FFmpeg fallback. Install Remotion for animated captions.",
},
artifacts=[output_path],
)
@staticmethod
def _ms_to_srt(ms: int) -> str:
h = ms // 3600000
m = (ms % 3600000) // 60000
s = (ms % 60000) // 1000
rem = ms % 1000
return f"{h:02d}:{m:02d}:{s:02d},{rem:03d}"
# ------------------------------------------------------------------ #
# Main execute
# ------------------------------------------------------------------ #
def execute(self, inputs: dict[str, Any]) -> ToolResult:
input_path = inputs["input_path"]
output_path = inputs["output_path"]
corrections = inputs.get("corrections")
force_ffmpeg = inputs.get("force_ffmpeg", False)
words_per_page = inputs.get("words_per_page", 4)
font_size = inputs.get("font_size", 52)
highlight_color = inputs.get("highlight_color", "#22D3EE")
if not Path(input_path).exists():
return ToolResult(success=False, error=f"Input video not found: {input_path}")
Path(output_path).parent.mkdir(parents=True, exist_ok=True)
start = time.time()
# Build word captions from segments or SRT
segments = inputs.get("segments")
srt_path = inputs.get("srt_path")
if segments:
captions = self._segments_to_word_captions(segments, corrections)
elif srt_path:
captions = self._srt_to_word_captions(srt_path, corrections)
else:
return ToolResult(
success=False,
error="Provide either 'segments' (from transcriber) or 'srt_path'.",
)
if not captions:
return ToolResult(success=False, error="No caption words extracted.")
# Choose render method
if not force_ffmpeg and self._remotion_available():
result = self._render_remotion(
input_path, output_path, captions,
words_per_page, font_size, highlight_color,
)
else:
result = self._render_ffmpeg(input_path, output_path, captions)
result.duration_seconds = round(time.time() - start, 2)
return result
+226
View File
@@ -0,0 +1,226 @@
"""Showcase card tool wrapping FFmpeg.
Creates a presentation-ready 9:16 card from a source video: letterboxes
the content, adds a bold title at the top, a subtitle description at the
bottom, and a dark background. Designed for Instagram Reels / TikTok
showcase segments.
"""
from __future__ import annotations
import time
from pathlib import Path
from typing import Any
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
ToolResult,
ToolStability,
ToolTier,
)
class ShowcaseCard(BaseTool):
name = "showcase_card"
version = "0.1.0"
tier = ToolTier.CORE
capability = "video_post"
provider = "ffmpeg"
stability = ToolStability.EXPERIMENTAL
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
dependencies = ["cmd:ffmpeg", "cmd:ffprobe"]
install_instructions = "Install FFmpeg: https://ffmpeg.org/download.html"
agent_skills = ["ffmpeg", "video_toolkit"]
capabilities = ["create_showcase_card"]
input_schema = {
"type": "object",
"required": ["input_path", "output_path", "title"],
"properties": {
"input_path": {
"type": "string",
"description": "Path to the source video.",
},
"output_path": {
"type": "string",
"description": "Path for the output showcase card video.",
},
"title": {
"type": "string",
"description": "Bold title text displayed at the top of the card.",
},
"subtitle": {
"type": "string",
"default": "",
"description": "Subtitle text displayed at the bottom of the card.",
},
"output_width": {
"type": "integer",
"default": 1080,
"description": "Output width in pixels.",
},
"output_height": {
"type": "integer",
"default": 1920,
"description": "Output height in pixels.",
},
"background_color": {
"type": "string",
"default": "0x0A0F1A",
"description": "Background color in hex (FFmpeg format, e.g. 0x0A0F1A).",
},
"title_font": {
"type": "string",
"default": "segoeuib.ttf",
"description": "Font file for the title. Uses system font lookup.",
},
"title_font_size": {
"type": "integer",
"default": 52,
"description": "Font size for the title.",
},
"subtitle_font_size": {
"type": "integer",
"default": 28,
"description": "Font size for the subtitle.",
},
"title_color": {
"type": "string",
"default": "white",
"description": "Title text color.",
},
"watermark": {
"type": "string",
"default": "",
"description": "Optional watermark text overlaid on the video (e.g. brand name).",
},
},
}
resource_profile = ResourceProfile(cpu_cores=2, ram_mb=1024, vram_mb=0, disk_mb=500)
idempotency_key_fields = ["input_path", "title", "subtitle"]
side_effects = ["writes showcase card video to output_path"]
user_visible_verification = [
"Play output and verify title, subtitle, and video are positioned correctly",
"Verify the video content is fully visible (not cropped)",
]
def execute(self, inputs: dict[str, Any]) -> ToolResult:
input_path = inputs["input_path"]
output_path = inputs["output_path"]
title = inputs["title"]
subtitle = inputs.get("subtitle", "")
out_w = inputs.get("output_width", 1080)
out_h = inputs.get("output_height", 1920)
bg_color = inputs.get("background_color", "0x0A0F1A")
title_font = inputs.get("title_font", "segoeuib.ttf")
title_font_size = inputs.get("title_font_size", 52)
subtitle_font_size = inputs.get("subtitle_font_size", 28)
title_color = inputs.get("title_color", "white")
watermark = inputs.get("watermark", "")
if not Path(input_path).exists():
return ToolResult(success=False, error=f"Input not found: {input_path}")
Path(output_path).parent.mkdir(parents=True, exist_ok=True)
start = time.time()
# Get source dimensions
probe_cmd = [
"ffprobe", "-v", "error",
"-select_streams", "v:0",
"-show_entries", "stream=width,height",
"-of", "csv=p=0",
input_path,
]
probe_out = self.run_command(probe_cmd, capture=True).strip()
src_w, src_h = [int(x.strip()) for x in probe_out.split(",")[:2]]
# Calculate letterbox dimensions — fit source into output width,
# center vertically in the frame.
scale_factor = out_w / src_w
scaled_h = int(src_h * scale_factor)
# Ensure even dimensions
scaled_h = scaled_h if scaled_h % 2 == 0 else scaled_h + 1
pad_y = (out_h - scaled_h) // 2
# Build filter chain
filters = [
f"scale={out_w}:{scaled_h}",
f"pad={out_w}:{out_h}:0:{pad_y}:color={bg_color}",
]
# Title text at top
title_escaped = title.replace("'", "\\'").replace(":", "\\:")
filters.append(
f"drawtext=text='{title_escaped}'"
f":fontfile='{title_font}'"
f":fontsize={title_font_size}"
f":fontcolor={title_color}"
f":borderw=3:bordercolor=black"
f":x=(w-text_w)/2:y=60"
)
# Subtitle text at bottom
if subtitle:
sub_escaped = subtitle.replace("'", "\\'").replace(":", "\\:")
filters.append(
f"drawtext=text='{sub_escaped}'"
f":fontfile='segoeui.ttf'"
f":fontsize={subtitle_font_size}"
f":fontcolor=white@0.85"
f":x=(w-text_w)/2:y=h-100"
)
# Watermark centered on video
if watermark:
wm_escaped = watermark.replace("'", "\\'").replace(":", "\\:")
filters.append(
f"drawtext=text='{wm_escaped}'"
f":fontfile='segoeui.ttf'"
f":fontsize=36"
f":fontcolor=white@0.3"
f":x=(w-text_w)/2:y=(h-text_h)/2"
)
vf = ",".join(filters)
cmd = [
"ffmpeg", "-y",
"-i", input_path,
"-vf", vf,
"-c:v", "libx264", "-preset", "fast", "-crf", "18",
"-pix_fmt", "yuv420p",
"-c:a", "aac", "-b:a", "192k",
output_path,
]
try:
self.run_command(cmd)
except Exception as e:
return ToolResult(success=False, error=f"FFmpeg failed: {e}")
if not Path(output_path).exists():
return ToolResult(success=False, error="No output produced")
elapsed = round(time.time() - start, 2)
return ToolResult(
success=True,
data={
"output": output_path,
"source_resolution": f"{src_w}x{src_h}",
"output_resolution": f"{out_w}x{out_h}",
"title": title,
"subtitle": subtitle,
"letterbox_y_offset": pad_y,
},
artifacts=[output_path],
duration_seconds=elapsed,
)
+481
View File
@@ -0,0 +1,481 @@
"""Silence cutter tool for automatic jump cuts.
Detects silent segments in talking-head footage and removes them,
creating tight jump cuts. Uses FFmpeg's silencedetect filter — no
external dependencies beyond FFmpeg.
Modes:
- remove: Cut out silent segments entirely (jump cut)
- speed_up: Speed up silent segments instead of cutting (less jarring)
- mark: Don't cut — just output silence timestamps for manual review
"""
from __future__ import annotations
import json
import re
import time
from pathlib import Path
from typing import Any
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
RetryPolicy,
ResumeSupport,
ToolResult,
ToolStability,
ToolTier,
)
class SilenceCutter(BaseTool):
name = "silence_cutter"
version = "0.1.0"
tier = ToolTier.CORE
capability = "video_post"
provider = "ffmpeg"
stability = ToolStability.EXPERIMENTAL
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
dependencies = ["cmd:ffmpeg"]
install_instructions = "Install FFmpeg: https://ffmpeg.org/download.html"
agent_skills = ["ffmpeg"]
capabilities = [
"silence_detection",
"jump_cut",
"silence_removal",
"silence_speedup",
]
input_schema = {
"type": "object",
"required": ["input_path"],
"properties": {
"input_path": {"type": "string"},
"output_path": {"type": "string"},
"mode": {
"type": "string",
"enum": ["remove", "speed_up", "mark"],
"default": "remove",
"description": "remove=jump cut, speed_up=fast-forward silence, mark=detect only",
},
"silence_threshold_db": {
"type": "number",
"default": -35,
"description": "Audio level below this (in dB) is considered silence. Lower = more sensitive.",
},
"min_silence_duration": {
"type": "number",
"default": 0.5,
"minimum": 0.1,
"description": "Minimum silence duration in seconds to trigger a cut",
},
"padding_seconds": {
"type": "number",
"default": 0.08,
"minimum": 0.0,
"description": "Seconds of silence to keep on each side of speech (prevents clipped words)",
},
"silence_speed_factor": {
"type": "number",
"default": 6.0,
"minimum": 1.5,
"maximum": 100.0,
"description": "Speed multiplier for silent segments (only used in speed_up mode)",
},
"codec": {"type": "string", "default": "libx264"},
"crf": {"type": "integer", "default": 18},
},
}
resource_profile = ResourceProfile(
cpu_cores=4, ram_mb=2048, vram_mb=0, disk_mb=4000, network_required=False
)
retry_policy = RetryPolicy(max_retries=1, retryable_errors=["FFmpeg error"])
resume_support = ResumeSupport.FROM_START
idempotency_key_fields = [
"input_path", "mode", "silence_threshold_db",
"min_silence_duration", "padding_seconds",
]
side_effects = ["writes cut video to output_path"]
user_visible_verification = [
"Watch output for unnaturally clipped words at cut points",
"Compare duration: output should be noticeably shorter than input",
]
def execute(self, inputs: dict[str, Any]) -> ToolResult:
input_path = Path(inputs["input_path"])
if not input_path.exists():
return ToolResult(success=False, error=f"Input not found: {input_path}")
mode = inputs.get("mode", "remove")
start = time.time()
# Step 1: Detect silence segments
threshold_db = inputs.get("silence_threshold_db", -35)
min_dur = inputs.get("min_silence_duration", 0.5)
padding = inputs.get("padding_seconds", 0.08)
silences = self._detect_silence(input_path, threshold_db, min_dur)
if not silences:
elapsed = time.time() - start
return ToolResult(
success=True,
data={
"message": "No silence detected — video unchanged",
"silence_segments": 0,
"input": str(input_path),
"output": str(input_path),
},
artifacts=[str(input_path)],
duration_seconds=round(elapsed, 2),
)
# Get total duration
total_duration = self._get_duration(input_path)
# Step 2: Compute speech segments (inverse of silence)
speech_segments = self._compute_speech_segments(
silences, total_duration, padding
)
# Step 3: Handle based on mode
if mode == "mark":
elapsed = time.time() - start
output_json = Path(
inputs.get("output_path", str(input_path.with_suffix(".silence.json")))
)
result_data = {
"silences": silences,
"speech_segments": speech_segments,
"total_duration": total_duration,
"silence_duration": sum(s["duration"] for s in silences),
"speech_duration": sum(s["end"] - s["start"] for s in speech_segments),
}
output_json.parent.mkdir(parents=True, exist_ok=True)
output_json.write_text(json.dumps(result_data, indent=2), encoding="utf-8")
return ToolResult(
success=True,
data={
"mode": "mark",
"silence_segments": len(silences),
"speech_segments": len(speech_segments),
"silence_duration_seconds": round(result_data["silence_duration"], 2),
"speech_duration_seconds": round(result_data["speech_duration"], 2),
"output": str(output_json),
},
artifacts=[str(output_json)],
duration_seconds=round(elapsed, 2),
)
output_path = Path(
inputs.get("output_path", str(input_path.with_stem(f"{input_path.stem}_cut")))
)
output_path.parent.mkdir(parents=True, exist_ok=True)
codec = inputs.get("codec", "libx264")
crf = inputs.get("crf", 18)
if mode == "speed_up":
speed_factor = inputs.get("silence_speed_factor", 6.0)
result = self._render_speed_up(
input_path, output_path, silences, speech_segments,
total_duration, speed_factor, codec, crf,
)
else:
result = self._render_jump_cut(
input_path, output_path, speech_segments, codec, crf,
)
if not result.success:
return result
elapsed = time.time() - start
silence_dur = sum(s["duration"] for s in silences)
speech_dur = sum(s["end"] - s["start"] for s in speech_segments)
return ToolResult(
success=True,
data={
"mode": mode,
"input": str(input_path),
"output": str(output_path),
"input_duration": round(total_duration, 2),
"output_duration": round(speech_dur, 2) if mode == "remove" else None,
"silence_removed_seconds": round(silence_dur, 2),
"silence_segments": len(silences),
"speech_segments": len(speech_segments),
"time_saved_percent": round(silence_dur / total_duration * 100, 1) if total_duration > 0 else 0,
},
artifacts=[str(output_path)],
duration_seconds=round(elapsed, 2),
)
def _detect_silence(
self, input_path: Path, threshold_db: float, min_duration: float
) -> list[dict]:
"""Detect silent segments using FFmpeg silencedetect filter."""
cmd = [
"ffmpeg",
"-i", str(input_path),
"-af", f"silencedetect=noise={threshold_db}dB:d={min_duration}",
"-f", "null", "-",
]
try:
result = self.run_command(cmd, timeout=300)
output = result.stderr
except Exception as e:
# FFmpeg writes to stderr even on success for filters
output = str(e)
# Parse silencedetect output
# Format: [silencedetect @ ...] silence_start: 1.234
# [silencedetect @ ...] silence_end: 2.567 | silence_duration: 1.333
starts = re.findall(r"silence_start:\s*([\d.]+)", output)
ends = re.findall(r"silence_end:\s*([\d.]+)", output)
durations = re.findall(r"silence_duration:\s*([\d.]+)", output)
silences = []
for i in range(min(len(starts), len(ends))):
silences.append({
"start": float(starts[i]),
"end": float(ends[i]),
"duration": float(durations[i]) if i < len(durations) else float(ends[i]) - float(starts[i]),
})
return silences
def _get_duration(self, input_path: Path) -> float:
"""Get video duration via ffprobe."""
cmd = [
"ffprobe", "-v", "quiet",
"-show_entries", "format=duration",
"-of", "json", str(input_path),
]
try:
result = self.run_command(cmd)
data = json.loads(result.stdout)
return float(data["format"]["duration"])
except Exception:
return 0.0
def _compute_speech_segments(
self, silences: list[dict], total_duration: float, padding: float
) -> list[dict]:
"""Compute speech segments as the inverse of silence segments, with padding."""
segments = []
cursor = 0.0
for silence in silences:
speech_end = silence["start"] + padding
if speech_end > cursor:
segments.append({"start": cursor, "end": min(speech_end, total_duration)})
cursor = max(cursor, silence["end"] - padding)
# Final segment after last silence
if cursor < total_duration:
segments.append({"start": cursor, "end": total_duration})
# Merge very short gaps (segments < 0.05s apart)
merged = []
for seg in segments:
if seg["end"] - seg["start"] < 0.01:
continue # Skip tiny segments
if merged and seg["start"] - merged[-1]["end"] < 0.05:
merged[-1]["end"] = seg["end"]
else:
merged.append(seg)
return merged
def _render_jump_cut(
self,
input_path: Path, output_path: Path,
speech_segments: list[dict],
codec: str, crf: int,
) -> ToolResult:
"""Remove silence by concatenating speech segments."""
if not speech_segments:
return ToolResult(success=False, error="No speech segments found")
temp_dir = output_path.parent / ".silence_cut_tmp"
temp_dir.mkdir(parents=True, exist_ok=True)
try:
# Cut each speech segment
seg_files = []
for i, seg in enumerate(speech_segments):
seg_path = temp_dir / f"seg_{i:04d}.mp4"
cmd = [
"ffmpeg", "-y",
"-i", str(input_path),
"-ss", f"{seg['start']:.3f}",
"-to", f"{seg['end']:.3f}",
"-c:v", codec, "-crf", str(crf), "-preset", "fast",
"-c:a", "aac", "-b:a", "192k",
# Force keyframe at start for clean cuts
"-force_key_frames", f"{seg['start']:.3f}",
str(seg_path),
]
self.run_command(cmd, timeout=120)
if seg_path.exists() and seg_path.stat().st_size > 0:
seg_files.append(seg_path)
if not seg_files:
return ToolResult(success=False, error="No segments were successfully cut")
# Concat all segments
list_path = temp_dir / "concat_list.txt"
with open(list_path, "w", encoding="utf-8") as f:
for sf in seg_files:
safe_path = str(sf.resolve()).replace("\\", "/")
f.write(f"file '{safe_path}'\n")
cmd = [
"ffmpeg", "-y",
"-f", "concat", "-safe", "0",
"-i", str(list_path),
"-c", "copy",
str(output_path),
]
self.run_command(cmd, timeout=120)
return ToolResult(success=True)
except Exception as e:
return ToolResult(success=False, error=f"Jump cut render failed: {e}")
finally:
# Clean up temp files
for f in temp_dir.glob("*"):
try:
f.unlink()
except OSError:
pass
try:
temp_dir.rmdir()
except OSError:
pass
def _render_speed_up(
self,
input_path: Path, output_path: Path,
silences: list[dict], speech_segments: list[dict],
total_duration: float,
speed_factor: float,
codec: str, crf: int,
) -> ToolResult:
"""Speed up silent segments instead of removing them.
This is less jarring than jump cuts — the viewer sees a brief
fast-forward during pauses.
"""
temp_dir = output_path.parent / ".silence_speed_tmp"
temp_dir.mkdir(parents=True, exist_ok=True)
try:
# Build a timeline of segments: speech at 1x, silence at Nx
all_segments = []
for seg in speech_segments:
all_segments.append({"start": seg["start"], "end": seg["end"], "speed": 1.0})
for sil in silences:
all_segments.append({"start": sil["start"], "end": sil["end"], "speed": speed_factor})
# Sort by start time and merge overlaps
all_segments.sort(key=lambda s: s["start"])
# Process each segment
seg_files = []
for i, seg in enumerate(all_segments):
seg_path = temp_dir / f"seg_{i:04d}.mp4"
duration = seg["end"] - seg["start"]
if duration < 0.05:
continue
if seg["speed"] == 1.0:
# Normal speed
cmd = [
"ffmpeg", "-y",
"-i", str(input_path),
"-ss", f"{seg['start']:.3f}",
"-to", f"{seg['end']:.3f}",
"-c:v", codec, "-crf", str(crf), "-preset", "fast",
"-c:a", "aac", "-b:a", "192k",
str(seg_path),
]
else:
# Speed up
pts = 1.0 / seg["speed"]
atempo_chain = self._build_atempo_chain(seg["speed"])
cmd = [
"ffmpeg", "-y",
"-i", str(input_path),
"-ss", f"{seg['start']:.3f}",
"-to", f"{seg['end']:.3f}",
"-filter:v", f"setpts={pts:.4f}*PTS",
"-filter:a", atempo_chain,
"-c:v", codec, "-crf", str(crf), "-preset", "fast",
"-c:a", "aac", "-b:a", "192k",
str(seg_path),
]
self.run_command(cmd, timeout=120)
if seg_path.exists() and seg_path.stat().st_size > 0:
seg_files.append(seg_path)
if not seg_files:
return ToolResult(success=False, error="No segments rendered")
# Concat
list_path = temp_dir / "concat_list.txt"
with open(list_path, "w", encoding="utf-8") as f:
for sf in seg_files:
safe_path = str(sf.resolve()).replace("\\", "/")
f.write(f"file '{safe_path}'\n")
cmd = [
"ffmpeg", "-y",
"-f", "concat", "-safe", "0",
"-i", str(list_path),
"-c", "copy",
str(output_path),
]
self.run_command(cmd, timeout=120)
return ToolResult(success=True)
except Exception as e:
return ToolResult(success=False, error=f"Speed-up render failed: {e}")
finally:
for f in temp_dir.glob("*"):
try:
f.unlink()
except OSError:
pass
try:
temp_dir.rmdir()
except OSError:
pass
@staticmethod
def _build_atempo_chain(factor: float) -> str:
"""Build atempo filter chain. atempo accepts [0.5, 100.0]."""
filters = []
remaining = factor
while remaining > 100.0:
filters.append("atempo=100.0")
remaining /= 100.0
while remaining < 0.5:
filters.append("atempo=0.5")
remaining /= 0.5
filters.append(f"atempo={remaining:.4f}")
return ",".join(filters)
def estimate_runtime(self, inputs: dict[str, Any]) -> float:
return 45.0