One-key formula: AI images, TTS narration, auto music, subtitles, and self-review pipeline

Prove that adding one API key (OPENAI_API_KEY) to the zero-key foundation
produces dramatically better videos for ~$0.69 each. Two compositions built:
The Abyss (deep ocean visual essay) and VOID (neural interface product ad).

New tools:
- audio_probe: ffprobe wrapper with probe_duration() helper
- composition_validator: pre-render checks (asset existence, audio-video sync)
- pixabay_music: royalty-free music scraper (no API key needed)
- freesound_music: Freesound API search + download

Remotion upgrades:
- BackgroundImageLayer: AI images behind data scenes with ken-burns + dark overlay
- Gradient support: all 9 components changed from backgroundColor to background CSS
- CaptionOverlay: word spacing fix, WhisperX word-level subtitles
- HeroTitle: reduced overlay opacity so background images show through

Process codified in agent skills:
- compose-director: audio acquisition flow (present user with voice/music/subtitle
  options), mandatory pre-render validation, post-render self-review (extract
  frames + transcribe + inspect + present findings to user)
- scene-director: narration duration budgeting (word budget from video duration)
- remotion skill: pre-render validation section
- TTS tool: now returns audio_duration_seconds in result

README updated with VOID product ad video embed.
This commit is contained in:
calesthio
2026-03-30 16:54:13 -07:00
parent 4ef335ac0b
commit 8cac647193
21 changed files with 1329 additions and 112 deletions
+6
View File
@@ -29,6 +29,12 @@ Turn your AI coding assistant into a full video production studio. Describe what
> **"SIGNAL FROM TOMORROW"** — a cinematic sci-fi trailer fully produced through OpenMontage: concept, script, scene plan, Veo-generated motion clips, soundtrack, and Remotion composition.
<div align="center">
<video src="https://github.com/user-attachments/assets/8a6d2cc3-7ad2-46f5-922f-a8e3e5848d9f" width="100%" controls></video>
</div>
> **"VOID — Neural Interface"** — a product ad produced with just one API key (OpenAI). 4 AI-generated images (gpt-image-1), TTS narration, auto-sourced royalty-free music, word-level subtitles via WhisperX, and Remotion data visualizations. Total cost: **$0.69**. Zero manual asset work.
Works with **Claude Code, Cursor, Copilot, Windsurf, Codex** — any AI coding assistant that can read files and run code.
---
+104 -94
View File
@@ -88,6 +88,8 @@ interface Cut {
heroSubtitle?: string;
// Styling overrides
backgroundColor?: string;
backgroundImage?: string; // AI-generated or stock image rendered behind the component
backgroundOverlay?: number; // Opacity of dark overlay on backgroundImage (0-1, default 0.55)
color?: string;
accentColor?: string;
fontSize?: number;
@@ -219,7 +221,7 @@ const ImageScene: React.FC<{ src: string; animation?: string }> = ({
// "static" or "none" → just display
return (
<AbsoluteFill style={{ overflow: "hidden", backgroundColor: "#0F172A" }}>
<AbsoluteFill style={{ overflow: "hidden", background: "#0F172A" }}>
<Img
src={resolveAsset(src)}
style={{
@@ -255,7 +257,7 @@ const VideoScene: React.FC<{ src: string; startFrom?: number }> = ({
});
return (
<AbsoluteFill style={{ backgroundColor: "#0F172A" }}>
<AbsoluteFill style={{ background: "#0F172A" }}>
<OffthreadVideo
src={resolveAsset(src)}
startFrom={Math.round(startFrom * fps)}
@@ -276,154 +278,162 @@ const VideoScene: React.FC<{ src: string; startFrom?: number }> = ({
// Scene renderer — maps cut type / source to the right component
// ---------------------------------------------------------------------------
// Background image layer — renders an AI-generated/stock image behind data components
const BackgroundImageLayer: React.FC<{
src: string;
overlayOpacity?: number;
children: React.ReactNode;
}> = ({ src, overlayOpacity = 0.55, children }) => {
const frame = useCurrentFrame();
const { fps, durationInFrames } = useVideoConfig();
// Subtle ken-burns on the background
const progress = interpolate(frame, [0, durationInFrames], [0, 1], {
extrapolateLeft: "clamp",
extrapolateRight: "clamp",
});
const bgScale = 1 + progress * 0.08;
return (
<AbsoluteFill style={{ overflow: "hidden" }}>
{/* Background image with subtle zoom */}
<Img
src={resolveAsset(src)}
style={{
width: "100%",
height: "100%",
objectFit: "cover",
transform: `scale(${bgScale})`,
willChange: "transform",
}}
/>
{/* Dark overlay for readability */}
<AbsoluteFill
style={{
background: `rgba(15, 23, 42, ${overlayOpacity})`,
}}
/>
{/* Component content on top */}
{children}
</AbsoluteFill>
);
};
const SceneRenderer: React.FC<{ cut: Cut }> = ({ cut }) => {
// Wrap component with background image if specified
const maybeWrapWithBgImage = (element: React.ReactElement) => {
if (cut.backgroundImage) {
return (
<BackgroundImageLayer
src={cut.backgroundImage}
overlayOpacity={cut.backgroundOverlay ?? 0.55}
>
{element}
</BackgroundImageLayer>
);
}
return element;
};
// Resolve the scene element based on cut type, then wrap with backgroundImage if set
const bgColor = cut.backgroundImage ? "transparent" : cut.backgroundColor;
// Explicit component types
if (cut.type === "text_card" && cut.text) {
return (
<TextCard
text={cut.text}
fontSize={cut.fontSize}
color={cut.color}
backgroundColor={cut.backgroundColor}
/>
return maybeWrapWithBgImage(
<TextCard text={cut.text} fontSize={cut.fontSize} color={cut.color} backgroundColor={bgColor} />
);
}
if (cut.type === "stat_card" && cut.stat) {
return (
<StatCard
stat={cut.stat}
subtitle={cut.subtitle}
accentColor={cut.accentColor}
backgroundColor={cut.backgroundColor}
/>
return maybeWrapWithBgImage(
<StatCard stat={cut.stat} subtitle={cut.subtitle} accentColor={cut.accentColor} backgroundColor={bgColor} />
);
}
if (cut.type === "callout" && cut.text) {
return (
return maybeWrapWithBgImage(
<CalloutBox
text={cut.text}
type={cut.callout_type}
title={cut.title}
borderColor={cut.accentColor}
backgroundColor={cut.backgroundColor}
textColor={cut.color}
containerBackgroundColor={cut.backgroundColor}
text={cut.text} type={cut.callout_type} title={cut.title}
borderColor={cut.accentColor} backgroundColor={cut.backgroundColor}
textColor={cut.color} containerBackgroundColor={bgColor}
/>
);
}
if (
cut.type === "comparison" &&
cut.leftLabel &&
cut.rightLabel &&
cut.leftValue &&
cut.rightValue
) {
return (
if (cut.type === "comparison" && cut.leftLabel && cut.rightLabel && cut.leftValue && cut.rightValue) {
return maybeWrapWithBgImage(
<ComparisonCard
leftLabel={cut.leftLabel}
rightLabel={cut.rightLabel}
leftValue={cut.leftValue}
rightValue={cut.rightValue}
title={cut.title}
backgroundColor={cut.backgroundColor}
textColor={cut.color}
leftLabel={cut.leftLabel} rightLabel={cut.rightLabel}
leftValue={cut.leftValue} rightValue={cut.rightValue}
title={cut.title} backgroundColor={bgColor} textColor={cut.color}
/>
);
}
if (cut.type === "hero_title" && cut.text) {
return <HeroTitle title={cut.text} subtitle={cut.heroSubtitle || cut.subtitle} />;
return maybeWrapWithBgImage(
<HeroTitle title={cut.text} subtitle={cut.heroSubtitle || cut.subtitle} />
);
}
// --- Chart types ---
if (cut.type === "bar_chart" && cut.chartData) {
return (
return maybeWrapWithBgImage(
<BarChart
data={cut.chartData}
title={cut.title}
colors={cut.chartColors}
data={cut.chartData} title={cut.title} colors={cut.chartColors}
animationStyle={(cut.chartAnimation as any) || "grow-up"}
showGrid={cut.showGrid}
showValues={cut.showValues}
backgroundColor={cut.backgroundColor}
showGrid={cut.showGrid} showValues={cut.showValues} backgroundColor={bgColor}
/>
);
}
if (cut.type === "line_chart" && cut.chartSeries) {
return (
return maybeWrapWithBgImage(
<LineChart
series={cut.chartSeries}
title={cut.title}
colors={cut.chartColors}
series={cut.chartSeries} title={cut.title} colors={cut.chartColors}
animationStyle={(cut.chartAnimation as any) || "draw"}
showGrid={cut.showGrid}
showMarkers={cut.showMarkers}
showLegend={cut.showLegend}
xLabel={cut.xLabel}
yLabel={cut.yLabel}
backgroundColor={cut.backgroundColor}
showGrid={cut.showGrid} showMarkers={cut.showMarkers} showLegend={cut.showLegend}
xLabel={cut.xLabel} yLabel={cut.yLabel} backgroundColor={bgColor}
/>
);
}
if (cut.type === "pie_chart" && cut.chartData) {
return (
return maybeWrapWithBgImage(
<PieChart
data={cut.chartData}
title={cut.title}
colors={cut.chartColors}
data={cut.chartData} title={cut.title} colors={cut.chartColors}
animationStyle={(cut.chartAnimation as any) || "expand"}
donut={cut.donut}
centerLabel={cut.centerLabel}
centerValue={cut.centerValue}
showLegend={cut.showLegend}
backgroundColor={cut.backgroundColor}
donut={cut.donut} centerLabel={cut.centerLabel} centerValue={cut.centerValue}
showLegend={cut.showLegend} backgroundColor={bgColor}
/>
);
}
if (cut.type === "kpi_grid" && cut.chartData) {
return (
return maybeWrapWithBgImage(
<KPIGrid
metrics={cut.chartData}
title={cut.title}
columns={cut.columns}
colors={cut.chartColors}
animationStyle={(cut.chartAnimation as any) || "count-up"}
backgroundColor={cut.backgroundColor}
metrics={cut.chartData} title={cut.title} columns={cut.columns}
colors={cut.chartColors} animationStyle={(cut.chartAnimation as any) || "count-up"}
backgroundColor={bgColor}
/>
);
}
if (cut.type === "progress_bar" && cut.progress !== undefined) {
return (
return maybeWrapWithBgImage(
<AbsoluteFill
style={{
backgroundColor: cut.backgroundColor || "#FFFFFF",
display: "flex",
alignItems: "center",
justifyContent: "center",
background: bgColor || "#FFFFFF",
display: "flex", alignItems: "center", justifyContent: "center",
padding: "80px 120px",
}}
>
{cut.title && (
<div
style={{
position: "absolute",
top: 120,
fontSize: 48,
fontWeight: 700,
color: "#1F2937",
textAlign: "center",
width: "100%",
}}
>
<div style={{
position: "absolute", top: 120, fontSize: 48, fontWeight: 700,
color: "#1F2937", textAlign: "center", width: "100%",
}}>
{cut.title}
</div>
)}
<ProgressBar
progress={cut.progress}
label={cut.progressLabel}
progress={cut.progress} label={cut.progressLabel}
color={cut.progressColor || cut.accentColor}
animationStyle={(cut.progressAnimation as any) || "fill"}
segments={cut.progressSegments}
backgroundColor={cut.backgroundColor}
segments={cut.progressSegments} backgroundColor={cut.backgroundColor}
/>
</AbsoluteFill>
);
@@ -493,7 +503,7 @@ export const Explainer: React.FC<ExplainerProps> = ({
const { fps, durationInFrames } = useVideoConfig();
return (
<AbsoluteFill style={{ backgroundColor: "#0F172A", fontFamily }}>
<AbsoluteFill style={{ background: "#0F172A", fontFamily }}>
{/* Layer 1: Visual scenes */}
{cuts.map((cut) => {
const from = Math.round(cut.in_seconds * fps);
@@ -111,7 +111,7 @@ export const CalloutBox: React.FC<CalloutBoxProps> = ({
return (
<AbsoluteFill
style={{
backgroundColor: containerBackgroundColor,
background: containerBackgroundColor,
justifyContent: "center",
alignItems: "center",
}}
@@ -107,7 +107,7 @@ const PageRenderer: React.FC<{
: "0 2px 4px rgba(0,0,0,0.5)",
}}
>
{w.word}
{w.word}{i < page.words.length - 1 ? " " : ""}
</span>
);
})}
@@ -134,7 +134,7 @@ export const ComparisonCard: React.FC<ComparisonCardProps> = ({
return (
<AbsoluteFill
style={{
backgroundColor,
background: backgroundColor,
justifyContent: "center",
alignItems: "center",
}}
@@ -24,7 +24,7 @@ export const HeroTitle: React.FC<HeroTitleProps> = ({ title, subtitle }) => {
justifyContent: "center",
alignItems: "center",
background:
"radial-gradient(ellipse at center, rgba(15,23,42,0.85) 0%, rgba(15,23,42,0.95) 100%)",
"radial-gradient(ellipse at center, rgba(15,23,42,0.35) 0%, rgba(15,23,42,0.55) 100%)",
}}
>
<div style={{ textAlign: "center", maxWidth: "85%" }}>
@@ -129,7 +129,7 @@ export const ProgressBar: React.FC<ProgressBarProps> = ({
return (
<AbsoluteFill
style={{
backgroundColor,
background: backgroundColor,
justifyContent: "center",
alignItems: "center",
}}
@@ -41,7 +41,7 @@ export const StatCard: React.FC<StatCardProps> = ({
style={{
justifyContent: "center",
alignItems: "center",
backgroundColor,
background: backgroundColor,
}}
>
<div style={{ textAlign: "center" }}>
@@ -30,7 +30,7 @@ export const TextCard: React.FC<TextCardProps> = ({
style={{
justifyContent: "center",
alignItems: "center",
backgroundColor,
background: backgroundColor,
}}
>
<div
@@ -73,7 +73,7 @@ export const BarChart: React.FC<BarChartProps> = ({
return (
<AbsoluteFill
style={{
backgroundColor,
background: backgroundColor,
justifyContent: "flex-start",
alignItems: "center",
padding: 40,
@@ -78,7 +78,7 @@ export const KPIGrid: React.FC<KPIGridProps> = ({
return (
<AbsoluteFill
style={{
backgroundColor,
background: backgroundColor,
justifyContent: "flex-start",
alignItems: "center",
fontFamily,
@@ -101,7 +101,7 @@ export const LineChart: React.FC<LineChartProps> = ({
return (
<AbsoluteFill
style={{
backgroundColor,
background: backgroundColor,
justifyContent: "flex-start",
alignItems: "center",
padding: 40,
@@ -91,7 +91,7 @@ export const PieChart: React.FC<PieChartProps> = ({
return (
<AbsoluteFill
style={{
backgroundColor,
background: backgroundColor,
justifyContent: "flex-start",
alignItems: "center",
padding: 40,
+23
View File
@@ -88,6 +88,29 @@ across bar/pie/line scenes for visual unity.
**Reference compositions:** See `remotion-composer/public/demo-props/climate-dashboard.json`
as the gold standard, and other demo files for additional patterns.
### Pre-Render Validation (mandatory)
**Always run `composition_validator` before rendering.** It catches:
- Missing asset files (images, audio) that would cause render failures
- Narration audio longer than video duration (audio gets cut off)
- Music shorter than video (silence at end)
- Invalid cut timings (out ≤ in)
```python
from tools.analysis.composition_validator import CompositionValidator
result = CompositionValidator().execute({
"composition_path": "path/to/composition.json",
"assets_root": "remotion-composer/public",
})
# result.data["valid"] must be True before rendering
```
**Audio duration alignment:**
- After generating TTS narration, the tool returns `audio_duration_seconds`.
- If narration exceeds video duration: shorten script and regenerate, OR extend the last scene.
- Use `tools.analysis.audio_probe.probe_duration(path)` to check any audio file's duration.
- Music should be ≥ video duration; the player handles fade-out via `fadeOutSeconds`.
## Architecture
```
+162 -7
View File
@@ -36,7 +36,84 @@ Based on the edit decisions, pick the rendering approach:
You can combine both: Remotion for animated segments, FFmpeg for final assembly.
### Step 2: Prepare Render Inputs
### Step 2: Audio Acquisition (Narration, Music, Subtitles)
Before rendering, present the user with audio options and get their preferences.
**Present to the user:**
> **Audio setup for this video:**
>
> **Narration:** I can generate TTS narration using OpenAI TTS (`gpt-4o-mini-tts` — $0.015/min, 6 voices, voice direction). Which voice and tone would you like? I'll propose a voice based on the video topic, or you can choose:
> - `onyx` — deep, authoritative (documentaries, tech)
> - `echo` — resonant, futuristic (product ads, sci-fi)
> - `nova` — bright, energetic (upbeat, explainers)
> - `fable` — warm, storytelling (narratives, education)
> - `shimmer` — expressive, warm (organic, lifestyle)
> - `alloy` — neutral, balanced (general purpose)
>
> **Music:** I can automatically find royalty-free background music from Pixabay (no key needed). If you have a `FREESOUND_API_KEY`, I can also search Freesound as a backup.
>
> **Subtitles:** I'll generate word-level subtitles using WhisperX transcription of the final narration, burned into the video via Remotion captions.
>
> Want me to proceed with my recommendations, or adjust anything?
**After user confirms:**
1. **Write narration script with duration budget** (see scene-director Step 4b):
- Calculate video duration from cuts
- Budget at 85-90% of video duration
- Use 2.0-2.5 words/sec for documentary, 2.5-3.0 for energetic
- Verify word count before generating TTS
2. **Generate TTS narration:**
```python
from tools.audio.openai_tts import OpenAITTS
result = OpenAITTS().execute({
'text': narration_script,
'voice': '<user-chosen or agent-recommended>',
'instructions': '<voice direction matching video tone>',
'output_path': 'path/to/narration.mp3',
})
# CRITICAL: Check result.data['audio_duration_seconds'] vs video duration
# If narration exceeds video by >1s: shorten script and regenerate
```
3. **Download background music:**
```python
from tools.audio.pixabay_music import PixabayMusic
result = PixabayMusic().execute({
'query': '<mood/genre matching video topic>',
'min_duration': video_duration_seconds,
'max_duration': 300,
'output_path': 'path/to/music.mp3',
})
```
4. **Generate subtitles via WhisperX:**
```python
from tools.analysis.transcriber import Transcriber
result = Transcriber().execute({
'input_path': 'path/to/narration.mp3',
'model_size': 'base',
'language': 'en',
})
# Convert word_timestamps to Remotion caption format:
# [{ "word": "Hello", "startMs": 0, "endMs": 340 }, ...]
```
5. **Assemble composition JSON** with audio config:
```json
{
"audio": {
"narration": { "src": "path/to/narration.mp3", "volume": 1 },
"music": { "src": "path/to/music.mp3", "volume": 0.1, "fadeInSeconds": 2, "fadeOutSeconds": 3 }
},
"captions": [ ... word-level captions from WhisperX ... ]
}
```
### Step 3: Prepare Render Inputs
For each cut in the edit decisions:
1. Verify the source asset exists at its declared path
@@ -44,8 +121,8 @@ For each cut in the edit decisions:
3. Prepare transform parameters (scale, position, crop)
For audio:
1. Verify all narration segments exist
2. Verify music track exists
1. Verify narration duration fits within video duration (use `audio_probe`)
2. Verify music duration covers video duration
3. Prepare ducking parameters from edit decisions
### Step 3: Determine Output Profile
@@ -152,7 +229,85 @@ Subtitles are mandatory for all explainer content. Generate them from the narrat
**The final deliverable is the subtitled version**, not the pre-subtitle render.
### Step 6: Verify Output
### Step 5c: Pre-Render Validation (Mandatory)
**Always run the composition validator before rendering.** This catches problems that waste render time.
```python
from tools.analysis.composition_validator import CompositionValidator
result = CompositionValidator().execute({
'composition_path': 'path/to/composition.json',
'assets_root': 'remotion-composer/public',
})
# result.data['valid'] MUST be True before proceeding to render
# If False: fix the reported errors first (missing assets, audio-video mismatch, etc.)
```
Common catches:
- Narration audio longer than video (would be cut off)
- Missing image/audio files (render would fail)
- Music shorter than video (silence at end)
**Do not skip this step.** If validation fails, fix the issue and re-validate before rendering.
### Step 6: Post-Render Self-Review (Mandatory)
After rendering, the agent **must review its own output** before presenting to the user. This catches issues the validator can't see (visual quality, audio sync, subtitle readability).
**6a. Extract review frames:**
```python
from tools.analysis.frame_sampler import FrameSampler
# Extract one frame per scene at the midpoint
midpoints = [(cut['in_seconds'] + cut['out_seconds']) / 2 for cut in cuts]
FrameSampler().execute({
'input_path': 'path/to/rendered_video.mp4',
'strategy': 'timestamps',
'timestamps': midpoints,
'output_dir': 'path/to/review-frames',
'format': 'png',
})
```
**6b. Transcribe rendered audio:**
```python
from tools.analysis.transcriber import Transcriber
Transcriber().execute({
'input_path': 'path/to/rendered_video.mp4',
'model_size': 'base',
'language': 'en',
'output_dir': 'path/to/review-frames',
})
# Verify all narration words are present and not cut off
```
**6c. Visual inspection — review each frame:**
- Does the background color/gradient match intent? (watch for white backgrounds on dark-themed videos)
- Are images rendering correctly? (not blank, not stretched)
- Are subtitles visible and properly spaced?
- Are overlays (section titles, stat reveals) positioned correctly?
- Is the opening scene visually strong? (important for social media thumbnails)
**6d. Audio inspection — check transcript:**
- Is the full narration captured? (compare last transcribed word to last scripted word)
- Any words cut off at the end? (narration exceeding video duration)
- Timing alignment — do narration segments roughly match their intended scenes?
**6e. Compile and present review to user:**
> **Post-render review for "[Video Title]":**
>
> **Audio:** [Complete/Cut off at Xs] — all N words captured / last sentence missing
> **Visuals:** [N scenes inspected] — [issues or "all scenes rendering correctly"]
> **Subtitles:** [Present/Missing] — [spacing ok / words running together]
> **Issues found:** [list any issues with severity]
>
> **Recommendations:** [what to fix, if anything]
>
> Want me to fix these issues and re-render, or is this good to go?
**Only after user approves (or agent finds zero issues) should the video be considered final.**
### Step 6-old: File and Content Verification
**File verification:**
- [ ] Output file exists at declared path
@@ -165,9 +320,9 @@ Subtitles are mandatory for all explainer content. Generate them from the narrat
- [ ] Audio channels present (stereo)
- [ ] No audio clipping or silence gaps > 1s
**Quality check:**
- [ ] Visual: scrub through at 25%, 50%, 75% marks — images display correctly
- [ ] Audio: narration is audible and clear throughout
**Quality check (covered by self-review above):**
- [ ] Visual: all scene frames inspected
- [ ] Audio: full transcription verified
- [ ] Subtitles: visible and correctly timed
### Step 7: Build Render Report
@@ -131,6 +131,31 @@ Show code with syntax highlighting. Highlight specific lines as the narrator exp
- Tools: `code_snippet` tool + Remotion
- Example: "Python code: `results = collection.query(embedding, n_results=5)`. Highlight `embedding` parameter when narrator says 'vector'."
### Step 4b: Write Narration with Duration Budget
If the video includes narration, the script **must** be written to fit the video duration.
**Duration budgeting formula:**
1. Calculate total video duration from scene timings (last cut's `out_seconds`).
2. Target narration at **85-90%** of video duration to leave breathing room at intro/outro.
3. Budget words: **2.0-2.5 words/second** for documentary style with natural pauses; **2.5-3.0 words/second** for energetic/fast-paced delivery.
4. Example: 53s video → target 45-48s of narration → 90-120 words max (documentary) or 112-144 words (energetic).
**Per-scene word budgets:**
- Allocate words proportionally to each scene's duration.
- A 5s scene gets ~10-12 words. A 6s scene gets ~12-15 words.
- Leave 0.5-1s of silence between scene transitions for visual breathing room.
**Validation (mandatory before TTS generation):**
- [ ] Total word count is within budget for the target duration
- [ ] No single scene's narration exceeds its time slot
- [ ] Opening and closing scenes have brief narration (let visuals breathe)
**After TTS generation:**
- The TTS tool returns `audio_duration_seconds` — compare it against video duration.
- If narration exceeds video by >1s, either trim the script and regenerate, or extend the video's closing scene.
- Always run `composition_validator` before rendering to catch mismatches automatically.
### Step 5: Validate Against Playbook
The style playbook constrains your visual choices:
+178
View File
@@ -0,0 +1,178 @@
"""Lightweight audio/video file probe using ffprobe.
Returns duration, format, sample rate, channels, and codec info
for any media file ffprobe can read. No heavy dependencies — just
requires ffmpeg/ffprobe on PATH.
"""
from __future__ import annotations
import json
import shutil
import subprocess
import time
from pathlib import Path
from typing import Any
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
RetryPolicy,
ToolResult,
ToolRuntime,
ToolStability,
ToolStatus,
ToolTier,
)
def probe_duration(file_path: str | Path) -> float | None:
"""Quick helper: return duration in seconds, or None on failure.
Use this from other tools that just need the duration without
going through the full tool execute() flow.
"""
ffprobe = shutil.which("ffprobe")
if not ffprobe:
return None
try:
result = subprocess.run(
[
ffprobe,
"-v", "quiet",
"-print_format", "json",
"-show_format",
str(file_path),
],
capture_output=True,
text=True,
timeout=10,
)
data = json.loads(result.stdout)
return float(data["format"]["duration"])
except Exception:
return None
class AudioProbe(BaseTool):
name = "audio_probe"
version = "0.1.0"
tier = ToolTier.CORE
capability = "analysis"
provider = "ffprobe"
stability = ToolStability.PRODUCTION
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
runtime = ToolRuntime.LOCAL
dependencies = ["binary:ffprobe"]
install_instructions = (
"Install ffmpeg (includes ffprobe):\n"
" Windows: winget install ffmpeg\n"
" macOS: brew install ffmpeg\n"
" Linux: sudo apt install ffmpeg"
)
capabilities = ["probe_duration", "probe_format", "probe_streams"]
best_for = [
"getting audio/video duration before composition",
"validating media file format and codec",
"pre-render checks on asset files",
]
input_schema = {
"type": "object",
"required": ["input_path"],
"properties": {
"input_path": {
"type": "string",
"description": "Path to audio or video file",
},
},
}
resource_profile = ResourceProfile(
cpu_cores=1, ram_mb=64, vram_mb=0, disk_mb=0, network_required=False
)
retry_policy = RetryPolicy(max_retries=0, retryable_errors=[])
idempotency_key_fields = ["input_path"]
side_effects = []
def get_status(self) -> ToolStatus:
if shutil.which("ffprobe"):
return ToolStatus.AVAILABLE
return ToolStatus.UNAVAILABLE
def estimate_cost(self, inputs: dict[str, Any]) -> float:
return 0.0
def execute(self, inputs: dict[str, Any]) -> ToolResult:
input_path = Path(inputs["input_path"])
if not input_path.exists():
return ToolResult(success=False, error=f"File not found: {input_path}")
ffprobe = shutil.which("ffprobe")
if not ffprobe:
return ToolResult(success=False, error="ffprobe not found on PATH")
start = time.time()
try:
result = subprocess.run(
[
ffprobe,
"-v", "quiet",
"-print_format", "json",
"-show_format",
"-show_streams",
str(input_path),
],
capture_output=True,
text=True,
timeout=15,
)
if result.returncode != 0:
return ToolResult(
success=False,
error=f"ffprobe failed: {result.stderr.strip()}",
)
data = json.loads(result.stdout)
except subprocess.TimeoutExpired:
return ToolResult(success=False, error="ffprobe timed out (15s)")
except json.JSONDecodeError:
return ToolResult(success=False, error="ffprobe returned invalid JSON")
fmt = data.get("format", {})
streams = data.get("streams", [])
# Find audio stream
audio_stream = next((s for s in streams if s.get("codec_type") == "audio"), None)
probe_data: dict[str, Any] = {
"file": str(input_path),
"duration_seconds": round(float(fmt.get("duration", 0)), 3),
"format_name": fmt.get("format_name"),
"format_long_name": fmt.get("format_long_name"),
"size_bytes": int(fmt.get("size", 0)),
"bit_rate": int(fmt.get("bit_rate", 0)),
"stream_count": len(streams),
}
if audio_stream:
probe_data["audio"] = {
"codec": audio_stream.get("codec_name"),
"sample_rate": int(audio_stream.get("sample_rate", 0)),
"channels": audio_stream.get("channels"),
"channel_layout": audio_stream.get("channel_layout"),
"bit_rate": int(audio_stream.get("bit_rate", 0)) if audio_stream.get("bit_rate") else None,
}
return ToolResult(
success=True,
data=probe_data,
duration_seconds=round(time.time() - start, 2),
)
+231
View File
@@ -0,0 +1,231 @@
"""Pre-render composition validator.
Checks an ExplainerProps JSON for common issues before rendering:
- Missing asset files (images, audio)
- Narration duration exceeding video duration
- Music duration shorter than video (warning)
- Overlapping or out-of-order cuts
- Required fields present
Run this before every render to catch problems that would otherwise
produce broken or truncated output.
"""
from __future__ import annotations
import json
import time
from pathlib import Path
from typing import Any
from tools.analysis.audio_probe import probe_duration
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
ToolResult,
ToolRuntime,
ToolStability,
ToolStatus,
ToolTier,
)
class CompositionValidator(BaseTool):
name = "composition_validator"
version = "0.1.0"
tier = ToolTier.CORE
capability = "analysis"
provider = "local"
stability = ToolStability.PRODUCTION
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
runtime = ToolRuntime.LOCAL
dependencies = ["binary:ffprobe"]
install_instructions = "Requires ffprobe on PATH (part of ffmpeg)."
capabilities = ["validate_composition", "pre_render_check"]
best_for = [
"catching audio-video duration mismatches before render",
"verifying all referenced assets exist",
"pre-flight check before expensive render operations",
]
input_schema = {
"type": "object",
"required": ["composition_path"],
"properties": {
"composition_path": {
"type": "string",
"description": "Path to the ExplainerProps JSON file",
},
"assets_root": {
"type": "string",
"description": "Root directory for resolving relative asset paths (defaults to composition's parent dir)",
},
},
}
resource_profile = ResourceProfile(
cpu_cores=1, ram_mb=64, vram_mb=0, disk_mb=0, network_required=False
)
side_effects = []
def get_status(self) -> ToolStatus:
return ToolStatus.AVAILABLE
def estimate_cost(self, inputs: dict[str, Any]) -> float:
return 0.0
def execute(self, inputs: dict[str, Any]) -> ToolResult:
comp_path = Path(inputs["composition_path"])
if not comp_path.exists():
return ToolResult(success=False, error=f"Composition not found: {comp_path}")
start = time.time()
try:
comp = json.loads(comp_path.read_text(encoding="utf-8"))
except (json.JSONDecodeError, UnicodeDecodeError) as e:
return ToolResult(success=False, error=f"Invalid JSON: {e}")
# Determine assets root (Remotion public dir)
assets_root = Path(inputs.get("assets_root", ""))
if not assets_root.is_dir():
# Default: look for remotion-composer/public relative to composition
candidate = comp_path
for _ in range(5):
candidate = candidate.parent
public = candidate / "remotion-composer" / "public"
if public.is_dir():
assets_root = public
break
else:
# Fall back to composition's parent
assets_root = comp_path.parent
errors: list[str] = []
warnings: list[str] = []
info: list[str] = []
cuts = comp.get("cuts", [])
audio = comp.get("audio", {})
# --- Check 1: Cuts exist ---
if not cuts:
errors.append("No cuts defined in composition")
return self._result(errors, warnings, info, start)
# --- Check 2: Video duration ---
video_duration = 0.0
for cut in cuts:
out_s = cut.get("out_seconds", 0)
if out_s > video_duration:
video_duration = out_s
info.append(f"Video duration: {video_duration}s ({len(cuts)} cuts)")
# --- Check 3: Cut ordering and gaps ---
sorted_cuts = sorted(cuts, key=lambda c: c.get("in_seconds", 0))
for i, cut in enumerate(sorted_cuts):
in_s = cut.get("in_seconds", 0)
out_s = cut.get("out_seconds", 0)
if out_s <= in_s:
errors.append(
f"Cut '{cut.get('id', i)}': out_seconds ({out_s}) <= in_seconds ({in_s})"
)
# --- Check 4: Asset files exist ---
for cut in cuts:
source = cut.get("source", "")
if source:
asset_path = assets_root / source
if not asset_path.exists():
errors.append(f"Missing asset: {source} (looked in {assets_root})")
bg_img = cut.get("backgroundImage", "")
if bg_img:
bg_path = assets_root / bg_img
if not bg_path.exists():
errors.append(f"Missing background image: {bg_img}")
# --- Check 5: Narration duration vs video duration ---
narration = audio.get("narration", {})
narration_src = narration.get("src", "")
if narration_src:
narration_path = assets_root / narration_src
if not narration_path.exists():
errors.append(f"Missing narration audio: {narration_src}")
else:
narration_dur = probe_duration(narration_path)
if narration_dur is not None:
info.append(f"Narration duration: {narration_dur:.1f}s")
overshoot = narration_dur - video_duration
if overshoot > 1.0:
errors.append(
f"Narration ({narration_dur:.1f}s) exceeds video ({video_duration}s) "
f"by {overshoot:.1f}s — audio will be cut off"
)
elif overshoot > 0:
warnings.append(
f"Narration ({narration_dur:.1f}s) slightly exceeds video ({video_duration}s) "
f"by {overshoot:.1f}s"
)
else:
warnings.append(f"Could not probe narration duration: {narration_src}")
# --- Check 6: Music duration ---
music = audio.get("music", {})
music_src = music.get("src", "")
if music_src:
music_path = assets_root / music_src
if not music_path.exists():
errors.append(f"Missing music audio: {music_src}")
else:
music_dur = probe_duration(music_path)
if music_dur is not None:
info.append(f"Music duration: {music_dur:.1f}s")
if music_dur < video_duration:
warnings.append(
f"Music ({music_dur:.1f}s) is shorter than video ({video_duration}s) "
f"— will end early"
)
# --- Check 7: No audio at all ---
if not narration_src and not music_src:
warnings.append("No audio configured (no narration or music)")
return self._result(errors, warnings, info, start)
def _result(
self,
errors: list[str],
warnings: list[str],
info: list[str],
start: float,
) -> ToolResult:
passed = len(errors) == 0
data = {
"valid": passed,
"errors": errors,
"warnings": warnings,
"info": info,
"error_count": len(errors),
"warning_count": len(warnings),
}
if not passed:
summary = "; ".join(errors[:3])
return ToolResult(
success=False,
error=f"Composition has {len(errors)} error(s): {summary}",
data=data,
duration_seconds=round(time.time() - start, 2),
)
return ToolResult(
success=True,
data=data,
duration_seconds=round(time.time() - start, 2),
)
+229
View File
@@ -0,0 +1,229 @@
"""Music search and download from Freesound.org (free with API key).
Searches Freesound's extensive library of Creative Commons audio and
downloads high-quality MP3 previews for use as background music.
"""
from __future__ import annotations
import json
import os
import time
import urllib.parse
import urllib.request
from pathlib import Path
from typing import Any
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
RetryPolicy,
ToolResult,
ToolRuntime,
ToolStability,
ToolStatus,
ToolTier,
)
class FreesoundMusic(BaseTool):
name = "freesound_music"
version = "0.1.0"
tier = ToolTier.SOURCE
capability = "music_search"
provider = "freesound"
stability = ToolStability.BETA
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
runtime = ToolRuntime.API
dependencies = [] # checked dynamically via env var
install_instructions = (
"Set the FREESOUND_API_KEY environment variable:\n"
" export FREESOUND_API_KEY=your_key_here\n"
"Get a free key at https://freesound.org/apiv2/apply/"
)
agent_skills = ["music"]
capabilities = ["search_music", "download_music", "stock_music"]
supports = {
"duration_filter": True,
"rating_sort": True,
"tag_metadata": True,
"free_creative_commons": True,
}
best_for = [
"ambient and atmospheric background music",
"free Creative Commons licensed audio",
"searching by mood, genre, or instrument tags",
"finding loops, drones, and textural audio",
]
not_good_for = [
"full produced songs with vocals",
"commercially licensed music (check individual CC licenses)",
"offline use",
]
fallback_tools = ["pixabay_music", "music_gen"]
input_schema = {
"type": "object",
"required": ["query"],
"properties": {
"query": {
"type": "string",
"description": "Search query describing desired music mood/genre (e.g., 'dark ambient cinematic underwater')",
},
"min_duration": {
"type": "number",
"default": 30,
"minimum": 1,
"description": "Minimum duration in seconds",
},
"max_duration": {
"type": "number",
"default": 120,
"maximum": 600,
"description": "Maximum duration in seconds",
},
"output_path": {
"type": "string",
"description": "File path to save the downloaded MP3",
},
},
}
resource_profile = ResourceProfile(
cpu_cores=1, ram_mb=256, vram_mb=0, disk_mb=50, network_required=True
)
retry_policy = RetryPolicy(max_retries=2, retryable_errors=["rate_limit", "timeout"])
idempotency_key_fields = ["query", "min_duration", "max_duration"]
side_effects = ["writes audio file to output_path", "calls Freesound API"]
user_visible_verification = [
"Listen to downloaded track for mood and quality",
"Check Creative Commons license terms for your use case",
]
_BASE_URL = "https://freesound.org/apiv2"
def get_status(self) -> ToolStatus:
if os.environ.get("FREESOUND_API_KEY"):
return ToolStatus.AVAILABLE
return ToolStatus.UNAVAILABLE
def estimate_cost(self, inputs: dict[str, Any]) -> float:
return 0.0 # Freesound is free
def execute(self, inputs: dict[str, Any]) -> ToolResult:
api_key = os.environ.get("FREESOUND_API_KEY")
if not api_key:
return ToolResult(
success=False,
error="FREESOUND_API_KEY not set. " + self.install_instructions,
)
start = time.time()
try:
# Step 1: Search for matching sounds
search_result = self._search(inputs, api_key)
if not search_result:
return ToolResult(
success=False,
error=f"No music found on Freesound for query: {inputs['query']}",
data={"query": inputs["query"]},
duration_seconds=round(time.time() - start, 2),
)
# Step 2: Pick the top result (sorted by rating)
sound = search_result[0]
# Step 3: Download the HQ MP3 preview
output_path = self._download(sound, inputs, api_key)
except Exception as e:
return ToolResult(
success=False,
error=f"Freesound music search failed: {e}",
duration_seconds=round(time.time() - start, 2),
)
return ToolResult(
success=True,
data={
"provider": "freesound",
"sound_id": sound.get("id"),
"name": sound.get("name", "Unknown"),
"duration_seconds": sound.get("duration"),
"avg_rating": sound.get("avg_rating"),
"tags": sound.get("tags", []),
"query": inputs["query"],
"output": str(output_path),
"format": "mp3",
"license": "Creative Commons (check individual sound license)",
"freesound_url": f"https://freesound.org/people/{sound.get('username', '')}/sounds/{sound.get('id', '')}/",
"results_found": len(search_result),
},
artifacts=[str(output_path)],
cost_usd=0.0,
duration_seconds=round(time.time() - start, 2),
)
def _search(self, inputs: dict[str, Any], api_key: str) -> list[dict]:
"""Search Freesound for sounds matching the query and duration filter."""
query = inputs["query"]
min_dur = inputs.get("min_duration", 30)
max_dur = inputs.get("max_duration", 120)
params = urllib.parse.urlencode({
"query": query,
"filter": f"duration:[{min_dur} TO {max_dur}]",
"sort": "rating_desc",
"fields": "id,name,duration,previews,tags,avg_rating,username",
"token": api_key,
"page_size": 15,
})
url = f"{self._BASE_URL}/search/text/?{params}"
request = urllib.request.Request(
url,
headers={"User-Agent": "OpenMontage/0.1 (music acquisition tool)"},
)
with urllib.request.urlopen(request, timeout=30) as response:
data = json.loads(response.read().decode("utf-8"))
results = data.get("results", [])
return results
def _download(self, sound: dict, inputs: dict[str, Any], api_key: str) -> Path:
"""Download the HQ MP3 preview of a Freesound sound."""
previews = sound.get("previews", {})
# Prefer the HQ MP3 preview; fall back to LQ MP3
audio_url = previews.get("preview-hq-mp3") or previews.get("preview-lq-mp3")
if not audio_url:
raise RuntimeError(
f"No preview URL available for sound {sound.get('id')} ({sound.get('name')})"
)
# Build output path
sound_name = sound.get("name", f"freesound_{sound.get('id', 'unknown')}")
safe_name = "".join(c if c.isalnum() or c in "._- " else "_" for c in sound_name)
default_filename = f"freesound_{sound.get('id')}_{safe_name}.mp3"
output_path = Path(inputs.get("output_path", default_filename))
output_path.parent.mkdir(parents=True, exist_ok=True)
request = urllib.request.Request(
audio_url,
headers={"User-Agent": "OpenMontage/0.1 (music acquisition tool)"},
)
with urllib.request.urlopen(request, timeout=60) as response:
output_path.write_bytes(response.read())
return output_path
+5
View File
@@ -122,6 +122,8 @@ class OpenAITTS(BaseTool):
def _generate(self, inputs: dict[str, Any]) -> ToolResult:
from openai import OpenAI
from tools.analysis.audio_probe import probe_duration
client = OpenAI()
text = inputs["text"]
model = inputs.get("model", "gpt-4o-mini-tts")
@@ -139,6 +141,8 @@ class OpenAITTS(BaseTool):
) as response:
response.stream_to_file(output_path)
audio_duration = probe_duration(output_path)
return ToolResult(
success=True,
data={
@@ -147,6 +151,7 @@ class OpenAITTS(BaseTool):
"voice": voice,
"format": fmt,
"text_length": len(text),
"audio_duration_seconds": round(audio_duration, 2) if audio_duration else None,
"output": str(output_path),
},
artifacts=[str(output_path)],
+355
View File
@@ -0,0 +1,355 @@
"""Music search and download from Pixabay Music (free, no API key).
Scrapes Pixabay's music section to find and download royalty-free
background music tracks. No API key required — uses web scraping.
Stability: EXPERIMENTAL — Pixabay's HTML structure may change without
notice, which could break the scraper. Use freesound_music or music_gen
as more stable alternatives.
"""
from __future__ import annotations
import json
import re
import time
import urllib.parse
import urllib.request
from pathlib import Path
from typing import Any
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
RetryPolicy,
ToolResult,
ToolRuntime,
ToolStability,
ToolStatus,
ToolTier,
)
class PixabayMusic(BaseTool):
name = "pixabay_music"
version = "0.1.0"
tier = ToolTier.SOURCE
capability = "music_search"
provider = "pixabay_music"
stability = ToolStability.EXPERIMENTAL
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
runtime = ToolRuntime.API
dependencies = [] # no API key needed — web scraping
install_instructions = (
"No setup required. Pixabay Music is free and needs no API key.\n"
"Note: This tool scrapes the Pixabay website. If it breaks, the\n"
"site's HTML structure may have changed. Use freesound_music as fallback."
)
agent_skills = ["music"]
capabilities = ["search_music", "download_music", "stock_music"]
supports = {
"duration_filter": True,
"free_commercial_use": True,
"no_api_key": True,
}
best_for = [
"quick background music with zero setup (no API key)",
"royalty-free music for any commercial project",
"high-quality produced tracks (not raw samples)",
]
not_good_for = [
"reliable long-term automation (scraping may break)",
"precise metadata filtering",
"offline use",
]
fallback_tools = ["freesound_music", "music_gen"]
input_schema = {
"type": "object",
"required": ["query"],
"properties": {
"query": {
"type": "string",
"description": "Search query for music (e.g., 'upbeat corporate background')",
},
"min_duration": {
"type": "number",
"default": 30,
"minimum": 1,
"description": "Minimum duration in seconds",
},
"max_duration": {
"type": "number",
"default": 120,
"maximum": 600,
"description": "Maximum duration in seconds",
},
"output_path": {
"type": "string",
"description": "File path to save the downloaded MP3",
},
},
}
resource_profile = ResourceProfile(
cpu_cores=1, ram_mb=256, vram_mb=0, disk_mb=50, network_required=True
)
retry_policy = RetryPolicy(max_retries=2, retryable_errors=["timeout"])
idempotency_key_fields = ["query", "min_duration", "max_duration"]
side_effects = ["writes audio file to output_path", "scrapes Pixabay website"]
user_visible_verification = [
"Listen to downloaded track for mood and quality",
]
_USER_AGENT = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/131.0.0.0 Safari/537.36"
)
_BROWSER_HEADERS = {
"Accept": (
"text/html,application/xhtml+xml,application/xml;"
"q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8"
),
"Accept-Language": "en-US,en;q=0.9",
"Sec-Ch-Ua": '"Chromium";v="131", "Not_A Brand";v="24"',
"Sec-Ch-Ua-Mobile": "?0",
"Sec-Ch-Ua-Platform": '"Windows"',
"Sec-Fetch-Dest": "document",
"Sec-Fetch-Mode": "navigate",
"Sec-Fetch-Site": "none",
"Sec-Fetch-User": "?1",
"Upgrade-Insecure-Requests": "1",
}
def get_status(self) -> ToolStatus:
# Always available — no API key required
return ToolStatus.AVAILABLE
def estimate_cost(self, inputs: dict[str, Any]) -> float:
return 0.0 # Pixabay Music is free
def execute(self, inputs: dict[str, Any]) -> ToolResult:
start = time.time()
try:
# Step 1: Search Pixabay Music
tracks = self._search(inputs)
if not tracks:
return ToolResult(
success=False,
error=f"No music found on Pixabay for query: {inputs['query']}",
data={"query": inputs["query"]},
duration_seconds=round(time.time() - start, 2),
)
# Step 2: Filter by duration
min_dur = inputs.get("min_duration", 30)
max_dur = inputs.get("max_duration", 120)
filtered = [
t for t in tracks
if t.get("duration") is not None
and min_dur <= t["duration"] <= max_dur
]
# Fall back to unfiltered if no matches within duration range
if not filtered:
filtered = tracks
# Step 3: Pick the first matching track
track = filtered[0]
# Step 4: Download the audio
output_path = self._download(track, inputs)
except Exception as e:
return ToolResult(
success=False,
error=f"Pixabay music search failed: {e}",
duration_seconds=round(time.time() - start, 2),
)
return ToolResult(
success=True,
data={
"provider": "pixabay_music",
"track_title": track.get("title", "Unknown"),
"artist": track.get("artist", "Unknown"),
"duration_seconds": track.get("duration"),
"query": inputs["query"],
"output": str(output_path),
"format": "mp3",
"license": "Pixabay Content License (free, no attribution required)",
"results_found": len(tracks),
"results_after_filter": len(filtered),
},
artifacts=[str(output_path)],
cost_usd=0.0,
duration_seconds=round(time.time() - start, 2),
)
def _build_opener(self) -> urllib.request.OpenerDirector:
"""Build a URL opener with cookie support for session persistence."""
import http.cookiejar
cj = http.cookiejar.CookieJar()
return urllib.request.build_opener(
urllib.request.HTTPCookieProcessor(cj)
)
def _search(self, inputs: dict[str, Any]) -> list[dict]:
"""Search Pixabay Music via the bootstrap JSON API.
Pixabay's music page loads track data from a bootstrap JSON endpoint
whose URL is embedded in the HTML. We:
1. Fetch the search page HTML (which sets session cookies).
2. Extract the __BOOTSTRAP_URL__ from an inline script tag.
3. Fetch the bootstrap JSON (same session) to get structured track data
including direct CDN MP3 URLs, durations, and metadata.
4. Fall back to HTML-scraping if bootstrap extraction fails.
"""
query = inputs["query"]
slug = re.sub(r"\s+", "-", query.strip().lower())
slug = urllib.parse.quote(slug, safe="-")
search_url = f"https://pixabay.com/music/search/{slug}/"
opener = self._build_opener()
# Step 1: Fetch search page HTML (sets cookies)
request = urllib.request.Request(search_url)
request.add_header("User-Agent", self._USER_AGENT)
for key, val in self._BROWSER_HEADERS.items():
request.add_header(key, val)
with opener.open(request, timeout=30) as response:
html = response.read().decode("utf-8", errors="replace")
# Step 2: Extract bootstrap URL and fetch track data
tracks = self._parse_bootstrap(html, search_url, opener)
if tracks:
return tracks
# Step 3: Fallback — scrape HTML directly (legacy strategies)
return self._parse_tracks_html(html)
def _parse_bootstrap(
self,
html: str,
referer: str,
opener: urllib.request.OpenerDirector,
) -> list[dict]:
"""Extract tracks from Pixabay's bootstrap JSON endpoint."""
match = re.search(
r'window\.__BOOTSTRAP_URL__\s*=\s*["\']([^"\']+)["\']',
html,
)
if not match:
return []
bootstrap_path = match.group(1)
if not bootstrap_path or bootstrap_path == "":
return []
bootstrap_url = f"https://pixabay.com{bootstrap_path}"
req = urllib.request.Request(bootstrap_url)
req.add_header("User-Agent", self._USER_AGENT)
req.add_header("Accept", "application/json, text/plain, */*")
req.add_header("Referer", referer)
req.add_header("Sec-Fetch-Dest", "empty")
req.add_header("Sec-Fetch-Mode", "cors")
req.add_header("Sec-Fetch-Site", "same-origin")
try:
with opener.open(req, timeout=15) as response:
data = json.loads(response.read().decode("utf-8"))
except Exception:
return []
results = data.get("page", {}).get("results", [])
tracks: list[dict] = []
for item in results:
sources = item.get("sources", {})
audio_url = sources.get("src")
if not audio_url:
continue
user = item.get("user", {}) or {}
tracks.append({
"title": item.get("name") or sources.get("filename", "Unknown"),
"audio_url": audio_url,
"duration": item.get("duration"),
"artist": user.get("username", "Unknown"),
"rating": item.get("rating"),
"download_count": item.get("downloadCount"),
"pixabay_id": item.get("id"),
})
return tracks
def _parse_tracks_html(self, html: str) -> list[dict]:
"""Fallback: extract track info from HTML when bootstrap fails.
Tries brute-force scan for CDN MP3 URLs in the page source.
"""
tracks: list[dict] = []
mp3_urls = re.findall(
r'(https?://cdn\.pixabay\.com/audio/[^\s"\'<>]+\.mp3[^\s"\'<>]*)',
html,
)
seen: set[str] = set()
for url in mp3_urls:
if url not in seen:
seen.add(url)
tracks.append({
"title": "Unknown",
"audio_url": url,
"duration": None,
"artist": "Unknown",
})
return tracks
def _download(self, track: dict, inputs: dict[str, Any]) -> Path:
"""Download an MP3 track to the output path."""
audio_url = track.get("audio_url")
if not audio_url:
raise RuntimeError("No audio URL found for the selected track.")
# Ensure URL is absolute
if audio_url.startswith("//"):
audio_url = "https:" + audio_url
elif audio_url.startswith("/"):
audio_url = "https://pixabay.com" + audio_url
# Build output path
track_title = track.get("title", "pixabay_music")
safe_title = "".join(
c if c.isalnum() or c in "._- " else "_" for c in track_title
)
default_filename = f"pixabay_music_{safe_title[:60]}.mp3"
output_path = Path(inputs.get("output_path", default_filename))
output_path.parent.mkdir(parents=True, exist_ok=True)
request = urllib.request.Request(
audio_url,
headers={
"User-Agent": self._USER_AGENT,
"Referer": "https://pixabay.com/music/",
},
)
with urllib.request.urlopen(request, timeout=60) as response:
output_path.write_bytes(response.read())
return output_path