One-key formula: AI images, TTS narration, auto music, subtitles, and self-review pipeline

Prove that adding one API key (OPENAI_API_KEY) to the zero-key foundation
produces dramatically better videos for ~$0.69 each. Two compositions built:
The Abyss (deep ocean visual essay) and VOID (neural interface product ad).

New tools:
- audio_probe: ffprobe wrapper with probe_duration() helper
- composition_validator: pre-render checks (asset existence, audio-video sync)
- pixabay_music: royalty-free music scraper (no API key needed)
- freesound_music: Freesound API search + download

Remotion upgrades:
- BackgroundImageLayer: AI images behind data scenes with ken-burns + dark overlay
- Gradient support: all 9 components changed from backgroundColor to background CSS
- CaptionOverlay: word spacing fix, WhisperX word-level subtitles
- HeroTitle: reduced overlay opacity so background images show through

Process codified in agent skills:
- compose-director: audio acquisition flow (present user with voice/music/subtitle
  options), mandatory pre-render validation, post-render self-review (extract
  frames + transcribe + inspect + present findings to user)
- scene-director: narration duration budgeting (word budget from video duration)
- remotion skill: pre-render validation section
- TTS tool: now returns audio_duration_seconds in result

README updated with VOID product ad video embed.
This commit is contained in:
calesthio
2026-03-30 16:54:13 -07:00
parent 4ef335ac0b
commit 8cac647193
21 changed files with 1329 additions and 112 deletions
+178
View File
@@ -0,0 +1,178 @@
"""Lightweight audio/video file probe using ffprobe.
Returns duration, format, sample rate, channels, and codec info
for any media file ffprobe can read. No heavy dependencies — just
requires ffmpeg/ffprobe on PATH.
"""
from __future__ import annotations
import json
import shutil
import subprocess
import time
from pathlib import Path
from typing import Any
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
RetryPolicy,
ToolResult,
ToolRuntime,
ToolStability,
ToolStatus,
ToolTier,
)
def probe_duration(file_path: str | Path) -> float | None:
"""Quick helper: return duration in seconds, or None on failure.
Use this from other tools that just need the duration without
going through the full tool execute() flow.
"""
ffprobe = shutil.which("ffprobe")
if not ffprobe:
return None
try:
result = subprocess.run(
[
ffprobe,
"-v", "quiet",
"-print_format", "json",
"-show_format",
str(file_path),
],
capture_output=True,
text=True,
timeout=10,
)
data = json.loads(result.stdout)
return float(data["format"]["duration"])
except Exception:
return None
class AudioProbe(BaseTool):
name = "audio_probe"
version = "0.1.0"
tier = ToolTier.CORE
capability = "analysis"
provider = "ffprobe"
stability = ToolStability.PRODUCTION
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
runtime = ToolRuntime.LOCAL
dependencies = ["binary:ffprobe"]
install_instructions = (
"Install ffmpeg (includes ffprobe):\n"
" Windows: winget install ffmpeg\n"
" macOS: brew install ffmpeg\n"
" Linux: sudo apt install ffmpeg"
)
capabilities = ["probe_duration", "probe_format", "probe_streams"]
best_for = [
"getting audio/video duration before composition",
"validating media file format and codec",
"pre-render checks on asset files",
]
input_schema = {
"type": "object",
"required": ["input_path"],
"properties": {
"input_path": {
"type": "string",
"description": "Path to audio or video file",
},
},
}
resource_profile = ResourceProfile(
cpu_cores=1, ram_mb=64, vram_mb=0, disk_mb=0, network_required=False
)
retry_policy = RetryPolicy(max_retries=0, retryable_errors=[])
idempotency_key_fields = ["input_path"]
side_effects = []
def get_status(self) -> ToolStatus:
if shutil.which("ffprobe"):
return ToolStatus.AVAILABLE
return ToolStatus.UNAVAILABLE
def estimate_cost(self, inputs: dict[str, Any]) -> float:
return 0.0
def execute(self, inputs: dict[str, Any]) -> ToolResult:
input_path = Path(inputs["input_path"])
if not input_path.exists():
return ToolResult(success=False, error=f"File not found: {input_path}")
ffprobe = shutil.which("ffprobe")
if not ffprobe:
return ToolResult(success=False, error="ffprobe not found on PATH")
start = time.time()
try:
result = subprocess.run(
[
ffprobe,
"-v", "quiet",
"-print_format", "json",
"-show_format",
"-show_streams",
str(input_path),
],
capture_output=True,
text=True,
timeout=15,
)
if result.returncode != 0:
return ToolResult(
success=False,
error=f"ffprobe failed: {result.stderr.strip()}",
)
data = json.loads(result.stdout)
except subprocess.TimeoutExpired:
return ToolResult(success=False, error="ffprobe timed out (15s)")
except json.JSONDecodeError:
return ToolResult(success=False, error="ffprobe returned invalid JSON")
fmt = data.get("format", {})
streams = data.get("streams", [])
# Find audio stream
audio_stream = next((s for s in streams if s.get("codec_type") == "audio"), None)
probe_data: dict[str, Any] = {
"file": str(input_path),
"duration_seconds": round(float(fmt.get("duration", 0)), 3),
"format_name": fmt.get("format_name"),
"format_long_name": fmt.get("format_long_name"),
"size_bytes": int(fmt.get("size", 0)),
"bit_rate": int(fmt.get("bit_rate", 0)),
"stream_count": len(streams),
}
if audio_stream:
probe_data["audio"] = {
"codec": audio_stream.get("codec_name"),
"sample_rate": int(audio_stream.get("sample_rate", 0)),
"channels": audio_stream.get("channels"),
"channel_layout": audio_stream.get("channel_layout"),
"bit_rate": int(audio_stream.get("bit_rate", 0)) if audio_stream.get("bit_rate") else None,
}
return ToolResult(
success=True,
data=probe_data,
duration_seconds=round(time.time() - start, 2),
)
+231
View File
@@ -0,0 +1,231 @@
"""Pre-render composition validator.
Checks an ExplainerProps JSON for common issues before rendering:
- Missing asset files (images, audio)
- Narration duration exceeding video duration
- Music duration shorter than video (warning)
- Overlapping or out-of-order cuts
- Required fields present
Run this before every render to catch problems that would otherwise
produce broken or truncated output.
"""
from __future__ import annotations
import json
import time
from pathlib import Path
from typing import Any
from tools.analysis.audio_probe import probe_duration
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
ToolResult,
ToolRuntime,
ToolStability,
ToolStatus,
ToolTier,
)
class CompositionValidator(BaseTool):
name = "composition_validator"
version = "0.1.0"
tier = ToolTier.CORE
capability = "analysis"
provider = "local"
stability = ToolStability.PRODUCTION
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
runtime = ToolRuntime.LOCAL
dependencies = ["binary:ffprobe"]
install_instructions = "Requires ffprobe on PATH (part of ffmpeg)."
capabilities = ["validate_composition", "pre_render_check"]
best_for = [
"catching audio-video duration mismatches before render",
"verifying all referenced assets exist",
"pre-flight check before expensive render operations",
]
input_schema = {
"type": "object",
"required": ["composition_path"],
"properties": {
"composition_path": {
"type": "string",
"description": "Path to the ExplainerProps JSON file",
},
"assets_root": {
"type": "string",
"description": "Root directory for resolving relative asset paths (defaults to composition's parent dir)",
},
},
}
resource_profile = ResourceProfile(
cpu_cores=1, ram_mb=64, vram_mb=0, disk_mb=0, network_required=False
)
side_effects = []
def get_status(self) -> ToolStatus:
return ToolStatus.AVAILABLE
def estimate_cost(self, inputs: dict[str, Any]) -> float:
return 0.0
def execute(self, inputs: dict[str, Any]) -> ToolResult:
comp_path = Path(inputs["composition_path"])
if not comp_path.exists():
return ToolResult(success=False, error=f"Composition not found: {comp_path}")
start = time.time()
try:
comp = json.loads(comp_path.read_text(encoding="utf-8"))
except (json.JSONDecodeError, UnicodeDecodeError) as e:
return ToolResult(success=False, error=f"Invalid JSON: {e}")
# Determine assets root (Remotion public dir)
assets_root = Path(inputs.get("assets_root", ""))
if not assets_root.is_dir():
# Default: look for remotion-composer/public relative to composition
candidate = comp_path
for _ in range(5):
candidate = candidate.parent
public = candidate / "remotion-composer" / "public"
if public.is_dir():
assets_root = public
break
else:
# Fall back to composition's parent
assets_root = comp_path.parent
errors: list[str] = []
warnings: list[str] = []
info: list[str] = []
cuts = comp.get("cuts", [])
audio = comp.get("audio", {})
# --- Check 1: Cuts exist ---
if not cuts:
errors.append("No cuts defined in composition")
return self._result(errors, warnings, info, start)
# --- Check 2: Video duration ---
video_duration = 0.0
for cut in cuts:
out_s = cut.get("out_seconds", 0)
if out_s > video_duration:
video_duration = out_s
info.append(f"Video duration: {video_duration}s ({len(cuts)} cuts)")
# --- Check 3: Cut ordering and gaps ---
sorted_cuts = sorted(cuts, key=lambda c: c.get("in_seconds", 0))
for i, cut in enumerate(sorted_cuts):
in_s = cut.get("in_seconds", 0)
out_s = cut.get("out_seconds", 0)
if out_s <= in_s:
errors.append(
f"Cut '{cut.get('id', i)}': out_seconds ({out_s}) <= in_seconds ({in_s})"
)
# --- Check 4: Asset files exist ---
for cut in cuts:
source = cut.get("source", "")
if source:
asset_path = assets_root / source
if not asset_path.exists():
errors.append(f"Missing asset: {source} (looked in {assets_root})")
bg_img = cut.get("backgroundImage", "")
if bg_img:
bg_path = assets_root / bg_img
if not bg_path.exists():
errors.append(f"Missing background image: {bg_img}")
# --- Check 5: Narration duration vs video duration ---
narration = audio.get("narration", {})
narration_src = narration.get("src", "")
if narration_src:
narration_path = assets_root / narration_src
if not narration_path.exists():
errors.append(f"Missing narration audio: {narration_src}")
else:
narration_dur = probe_duration(narration_path)
if narration_dur is not None:
info.append(f"Narration duration: {narration_dur:.1f}s")
overshoot = narration_dur - video_duration
if overshoot > 1.0:
errors.append(
f"Narration ({narration_dur:.1f}s) exceeds video ({video_duration}s) "
f"by {overshoot:.1f}s — audio will be cut off"
)
elif overshoot > 0:
warnings.append(
f"Narration ({narration_dur:.1f}s) slightly exceeds video ({video_duration}s) "
f"by {overshoot:.1f}s"
)
else:
warnings.append(f"Could not probe narration duration: {narration_src}")
# --- Check 6: Music duration ---
music = audio.get("music", {})
music_src = music.get("src", "")
if music_src:
music_path = assets_root / music_src
if not music_path.exists():
errors.append(f"Missing music audio: {music_src}")
else:
music_dur = probe_duration(music_path)
if music_dur is not None:
info.append(f"Music duration: {music_dur:.1f}s")
if music_dur < video_duration:
warnings.append(
f"Music ({music_dur:.1f}s) is shorter than video ({video_duration}s) "
f"— will end early"
)
# --- Check 7: No audio at all ---
if not narration_src and not music_src:
warnings.append("No audio configured (no narration or music)")
return self._result(errors, warnings, info, start)
def _result(
self,
errors: list[str],
warnings: list[str],
info: list[str],
start: float,
) -> ToolResult:
passed = len(errors) == 0
data = {
"valid": passed,
"errors": errors,
"warnings": warnings,
"info": info,
"error_count": len(errors),
"warning_count": len(warnings),
}
if not passed:
summary = "; ".join(errors[:3])
return ToolResult(
success=False,
error=f"Composition has {len(errors)} error(s): {summary}",
data=data,
duration_seconds=round(time.time() - start, 2),
)
return ToolResult(
success=True,
data=data,
duration_seconds=round(time.time() - start, 2),
)
+229
View File
@@ -0,0 +1,229 @@
"""Music search and download from Freesound.org (free with API key).
Searches Freesound's extensive library of Creative Commons audio and
downloads high-quality MP3 previews for use as background music.
"""
from __future__ import annotations
import json
import os
import time
import urllib.parse
import urllib.request
from pathlib import Path
from typing import Any
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
RetryPolicy,
ToolResult,
ToolRuntime,
ToolStability,
ToolStatus,
ToolTier,
)
class FreesoundMusic(BaseTool):
name = "freesound_music"
version = "0.1.0"
tier = ToolTier.SOURCE
capability = "music_search"
provider = "freesound"
stability = ToolStability.BETA
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
runtime = ToolRuntime.API
dependencies = [] # checked dynamically via env var
install_instructions = (
"Set the FREESOUND_API_KEY environment variable:\n"
" export FREESOUND_API_KEY=your_key_here\n"
"Get a free key at https://freesound.org/apiv2/apply/"
)
agent_skills = ["music"]
capabilities = ["search_music", "download_music", "stock_music"]
supports = {
"duration_filter": True,
"rating_sort": True,
"tag_metadata": True,
"free_creative_commons": True,
}
best_for = [
"ambient and atmospheric background music",
"free Creative Commons licensed audio",
"searching by mood, genre, or instrument tags",
"finding loops, drones, and textural audio",
]
not_good_for = [
"full produced songs with vocals",
"commercially licensed music (check individual CC licenses)",
"offline use",
]
fallback_tools = ["pixabay_music", "music_gen"]
input_schema = {
"type": "object",
"required": ["query"],
"properties": {
"query": {
"type": "string",
"description": "Search query describing desired music mood/genre (e.g., 'dark ambient cinematic underwater')",
},
"min_duration": {
"type": "number",
"default": 30,
"minimum": 1,
"description": "Minimum duration in seconds",
},
"max_duration": {
"type": "number",
"default": 120,
"maximum": 600,
"description": "Maximum duration in seconds",
},
"output_path": {
"type": "string",
"description": "File path to save the downloaded MP3",
},
},
}
resource_profile = ResourceProfile(
cpu_cores=1, ram_mb=256, vram_mb=0, disk_mb=50, network_required=True
)
retry_policy = RetryPolicy(max_retries=2, retryable_errors=["rate_limit", "timeout"])
idempotency_key_fields = ["query", "min_duration", "max_duration"]
side_effects = ["writes audio file to output_path", "calls Freesound API"]
user_visible_verification = [
"Listen to downloaded track for mood and quality",
"Check Creative Commons license terms for your use case",
]
_BASE_URL = "https://freesound.org/apiv2"
def get_status(self) -> ToolStatus:
if os.environ.get("FREESOUND_API_KEY"):
return ToolStatus.AVAILABLE
return ToolStatus.UNAVAILABLE
def estimate_cost(self, inputs: dict[str, Any]) -> float:
return 0.0 # Freesound is free
def execute(self, inputs: dict[str, Any]) -> ToolResult:
api_key = os.environ.get("FREESOUND_API_KEY")
if not api_key:
return ToolResult(
success=False,
error="FREESOUND_API_KEY not set. " + self.install_instructions,
)
start = time.time()
try:
# Step 1: Search for matching sounds
search_result = self._search(inputs, api_key)
if not search_result:
return ToolResult(
success=False,
error=f"No music found on Freesound for query: {inputs['query']}",
data={"query": inputs["query"]},
duration_seconds=round(time.time() - start, 2),
)
# Step 2: Pick the top result (sorted by rating)
sound = search_result[0]
# Step 3: Download the HQ MP3 preview
output_path = self._download(sound, inputs, api_key)
except Exception as e:
return ToolResult(
success=False,
error=f"Freesound music search failed: {e}",
duration_seconds=round(time.time() - start, 2),
)
return ToolResult(
success=True,
data={
"provider": "freesound",
"sound_id": sound.get("id"),
"name": sound.get("name", "Unknown"),
"duration_seconds": sound.get("duration"),
"avg_rating": sound.get("avg_rating"),
"tags": sound.get("tags", []),
"query": inputs["query"],
"output": str(output_path),
"format": "mp3",
"license": "Creative Commons (check individual sound license)",
"freesound_url": f"https://freesound.org/people/{sound.get('username', '')}/sounds/{sound.get('id', '')}/",
"results_found": len(search_result),
},
artifacts=[str(output_path)],
cost_usd=0.0,
duration_seconds=round(time.time() - start, 2),
)
def _search(self, inputs: dict[str, Any], api_key: str) -> list[dict]:
"""Search Freesound for sounds matching the query and duration filter."""
query = inputs["query"]
min_dur = inputs.get("min_duration", 30)
max_dur = inputs.get("max_duration", 120)
params = urllib.parse.urlencode({
"query": query,
"filter": f"duration:[{min_dur} TO {max_dur}]",
"sort": "rating_desc",
"fields": "id,name,duration,previews,tags,avg_rating,username",
"token": api_key,
"page_size": 15,
})
url = f"{self._BASE_URL}/search/text/?{params}"
request = urllib.request.Request(
url,
headers={"User-Agent": "OpenMontage/0.1 (music acquisition tool)"},
)
with urllib.request.urlopen(request, timeout=30) as response:
data = json.loads(response.read().decode("utf-8"))
results = data.get("results", [])
return results
def _download(self, sound: dict, inputs: dict[str, Any], api_key: str) -> Path:
"""Download the HQ MP3 preview of a Freesound sound."""
previews = sound.get("previews", {})
# Prefer the HQ MP3 preview; fall back to LQ MP3
audio_url = previews.get("preview-hq-mp3") or previews.get("preview-lq-mp3")
if not audio_url:
raise RuntimeError(
f"No preview URL available for sound {sound.get('id')} ({sound.get('name')})"
)
# Build output path
sound_name = sound.get("name", f"freesound_{sound.get('id', 'unknown')}")
safe_name = "".join(c if c.isalnum() or c in "._- " else "_" for c in sound_name)
default_filename = f"freesound_{sound.get('id')}_{safe_name}.mp3"
output_path = Path(inputs.get("output_path", default_filename))
output_path.parent.mkdir(parents=True, exist_ok=True)
request = urllib.request.Request(
audio_url,
headers={"User-Agent": "OpenMontage/0.1 (music acquisition tool)"},
)
with urllib.request.urlopen(request, timeout=60) as response:
output_path.write_bytes(response.read())
return output_path
+5
View File
@@ -122,6 +122,8 @@ class OpenAITTS(BaseTool):
def _generate(self, inputs: dict[str, Any]) -> ToolResult:
from openai import OpenAI
from tools.analysis.audio_probe import probe_duration
client = OpenAI()
text = inputs["text"]
model = inputs.get("model", "gpt-4o-mini-tts")
@@ -139,6 +141,8 @@ class OpenAITTS(BaseTool):
) as response:
response.stream_to_file(output_path)
audio_duration = probe_duration(output_path)
return ToolResult(
success=True,
data={
@@ -147,6 +151,7 @@ class OpenAITTS(BaseTool):
"voice": voice,
"format": fmt,
"text_length": len(text),
"audio_duration_seconds": round(audio_duration, 2) if audio_duration else None,
"output": str(output_path),
},
artifacts=[str(output_path)],
+355
View File
@@ -0,0 +1,355 @@
"""Music search and download from Pixabay Music (free, no API key).
Scrapes Pixabay's music section to find and download royalty-free
background music tracks. No API key required — uses web scraping.
Stability: EXPERIMENTAL — Pixabay's HTML structure may change without
notice, which could break the scraper. Use freesound_music or music_gen
as more stable alternatives.
"""
from __future__ import annotations
import json
import re
import time
import urllib.parse
import urllib.request
from pathlib import Path
from typing import Any
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
RetryPolicy,
ToolResult,
ToolRuntime,
ToolStability,
ToolStatus,
ToolTier,
)
class PixabayMusic(BaseTool):
name = "pixabay_music"
version = "0.1.0"
tier = ToolTier.SOURCE
capability = "music_search"
provider = "pixabay_music"
stability = ToolStability.EXPERIMENTAL
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
runtime = ToolRuntime.API
dependencies = [] # no API key needed — web scraping
install_instructions = (
"No setup required. Pixabay Music is free and needs no API key.\n"
"Note: This tool scrapes the Pixabay website. If it breaks, the\n"
"site's HTML structure may have changed. Use freesound_music as fallback."
)
agent_skills = ["music"]
capabilities = ["search_music", "download_music", "stock_music"]
supports = {
"duration_filter": True,
"free_commercial_use": True,
"no_api_key": True,
}
best_for = [
"quick background music with zero setup (no API key)",
"royalty-free music for any commercial project",
"high-quality produced tracks (not raw samples)",
]
not_good_for = [
"reliable long-term automation (scraping may break)",
"precise metadata filtering",
"offline use",
]
fallback_tools = ["freesound_music", "music_gen"]
input_schema = {
"type": "object",
"required": ["query"],
"properties": {
"query": {
"type": "string",
"description": "Search query for music (e.g., 'upbeat corporate background')",
},
"min_duration": {
"type": "number",
"default": 30,
"minimum": 1,
"description": "Minimum duration in seconds",
},
"max_duration": {
"type": "number",
"default": 120,
"maximum": 600,
"description": "Maximum duration in seconds",
},
"output_path": {
"type": "string",
"description": "File path to save the downloaded MP3",
},
},
}
resource_profile = ResourceProfile(
cpu_cores=1, ram_mb=256, vram_mb=0, disk_mb=50, network_required=True
)
retry_policy = RetryPolicy(max_retries=2, retryable_errors=["timeout"])
idempotency_key_fields = ["query", "min_duration", "max_duration"]
side_effects = ["writes audio file to output_path", "scrapes Pixabay website"]
user_visible_verification = [
"Listen to downloaded track for mood and quality",
]
_USER_AGENT = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/131.0.0.0 Safari/537.36"
)
_BROWSER_HEADERS = {
"Accept": (
"text/html,application/xhtml+xml,application/xml;"
"q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8"
),
"Accept-Language": "en-US,en;q=0.9",
"Sec-Ch-Ua": '"Chromium";v="131", "Not_A Brand";v="24"',
"Sec-Ch-Ua-Mobile": "?0",
"Sec-Ch-Ua-Platform": '"Windows"',
"Sec-Fetch-Dest": "document",
"Sec-Fetch-Mode": "navigate",
"Sec-Fetch-Site": "none",
"Sec-Fetch-User": "?1",
"Upgrade-Insecure-Requests": "1",
}
def get_status(self) -> ToolStatus:
# Always available — no API key required
return ToolStatus.AVAILABLE
def estimate_cost(self, inputs: dict[str, Any]) -> float:
return 0.0 # Pixabay Music is free
def execute(self, inputs: dict[str, Any]) -> ToolResult:
start = time.time()
try:
# Step 1: Search Pixabay Music
tracks = self._search(inputs)
if not tracks:
return ToolResult(
success=False,
error=f"No music found on Pixabay for query: {inputs['query']}",
data={"query": inputs["query"]},
duration_seconds=round(time.time() - start, 2),
)
# Step 2: Filter by duration
min_dur = inputs.get("min_duration", 30)
max_dur = inputs.get("max_duration", 120)
filtered = [
t for t in tracks
if t.get("duration") is not None
and min_dur <= t["duration"] <= max_dur
]
# Fall back to unfiltered if no matches within duration range
if not filtered:
filtered = tracks
# Step 3: Pick the first matching track
track = filtered[0]
# Step 4: Download the audio
output_path = self._download(track, inputs)
except Exception as e:
return ToolResult(
success=False,
error=f"Pixabay music search failed: {e}",
duration_seconds=round(time.time() - start, 2),
)
return ToolResult(
success=True,
data={
"provider": "pixabay_music",
"track_title": track.get("title", "Unknown"),
"artist": track.get("artist", "Unknown"),
"duration_seconds": track.get("duration"),
"query": inputs["query"],
"output": str(output_path),
"format": "mp3",
"license": "Pixabay Content License (free, no attribution required)",
"results_found": len(tracks),
"results_after_filter": len(filtered),
},
artifacts=[str(output_path)],
cost_usd=0.0,
duration_seconds=round(time.time() - start, 2),
)
def _build_opener(self) -> urllib.request.OpenerDirector:
"""Build a URL opener with cookie support for session persistence."""
import http.cookiejar
cj = http.cookiejar.CookieJar()
return urllib.request.build_opener(
urllib.request.HTTPCookieProcessor(cj)
)
def _search(self, inputs: dict[str, Any]) -> list[dict]:
"""Search Pixabay Music via the bootstrap JSON API.
Pixabay's music page loads track data from a bootstrap JSON endpoint
whose URL is embedded in the HTML. We:
1. Fetch the search page HTML (which sets session cookies).
2. Extract the __BOOTSTRAP_URL__ from an inline script tag.
3. Fetch the bootstrap JSON (same session) to get structured track data
including direct CDN MP3 URLs, durations, and metadata.
4. Fall back to HTML-scraping if bootstrap extraction fails.
"""
query = inputs["query"]
slug = re.sub(r"\s+", "-", query.strip().lower())
slug = urllib.parse.quote(slug, safe="-")
search_url = f"https://pixabay.com/music/search/{slug}/"
opener = self._build_opener()
# Step 1: Fetch search page HTML (sets cookies)
request = urllib.request.Request(search_url)
request.add_header("User-Agent", self._USER_AGENT)
for key, val in self._BROWSER_HEADERS.items():
request.add_header(key, val)
with opener.open(request, timeout=30) as response:
html = response.read().decode("utf-8", errors="replace")
# Step 2: Extract bootstrap URL and fetch track data
tracks = self._parse_bootstrap(html, search_url, opener)
if tracks:
return tracks
# Step 3: Fallback — scrape HTML directly (legacy strategies)
return self._parse_tracks_html(html)
def _parse_bootstrap(
self,
html: str,
referer: str,
opener: urllib.request.OpenerDirector,
) -> list[dict]:
"""Extract tracks from Pixabay's bootstrap JSON endpoint."""
match = re.search(
r'window\.__BOOTSTRAP_URL__\s*=\s*["\']([^"\']+)["\']',
html,
)
if not match:
return []
bootstrap_path = match.group(1)
if not bootstrap_path or bootstrap_path == "":
return []
bootstrap_url = f"https://pixabay.com{bootstrap_path}"
req = urllib.request.Request(bootstrap_url)
req.add_header("User-Agent", self._USER_AGENT)
req.add_header("Accept", "application/json, text/plain, */*")
req.add_header("Referer", referer)
req.add_header("Sec-Fetch-Dest", "empty")
req.add_header("Sec-Fetch-Mode", "cors")
req.add_header("Sec-Fetch-Site", "same-origin")
try:
with opener.open(req, timeout=15) as response:
data = json.loads(response.read().decode("utf-8"))
except Exception:
return []
results = data.get("page", {}).get("results", [])
tracks: list[dict] = []
for item in results:
sources = item.get("sources", {})
audio_url = sources.get("src")
if not audio_url:
continue
user = item.get("user", {}) or {}
tracks.append({
"title": item.get("name") or sources.get("filename", "Unknown"),
"audio_url": audio_url,
"duration": item.get("duration"),
"artist": user.get("username", "Unknown"),
"rating": item.get("rating"),
"download_count": item.get("downloadCount"),
"pixabay_id": item.get("id"),
})
return tracks
def _parse_tracks_html(self, html: str) -> list[dict]:
"""Fallback: extract track info from HTML when bootstrap fails.
Tries brute-force scan for CDN MP3 URLs in the page source.
"""
tracks: list[dict] = []
mp3_urls = re.findall(
r'(https?://cdn\.pixabay\.com/audio/[^\s"\'<>]+\.mp3[^\s"\'<>]*)',
html,
)
seen: set[str] = set()
for url in mp3_urls:
if url not in seen:
seen.add(url)
tracks.append({
"title": "Unknown",
"audio_url": url,
"duration": None,
"artist": "Unknown",
})
return tracks
def _download(self, track: dict, inputs: dict[str, Any]) -> Path:
"""Download an MP3 track to the output path."""
audio_url = track.get("audio_url")
if not audio_url:
raise RuntimeError("No audio URL found for the selected track.")
# Ensure URL is absolute
if audio_url.startswith("//"):
audio_url = "https:" + audio_url
elif audio_url.startswith("/"):
audio_url = "https://pixabay.com" + audio_url
# Build output path
track_title = track.get("title", "pixabay_music")
safe_title = "".join(
c if c.isalnum() or c in "._- " else "_" for c in track_title
)
default_filename = f"pixabay_music_{safe_title[:60]}.mp3"
output_path = Path(inputs.get("output_path", default_filename))
output_path.parent.mkdir(parents=True, exist_ok=True)
request = urllib.request.Request(
audio_url,
headers={
"User-Agent": self._USER_AGENT,
"Referer": "https://pixabay.com/music/",
},
)
with urllib.request.urlopen(request, timeout=60) as response:
output_path.write_bytes(response.read())
return output_path