Initial release — OpenMontage: the first open-source agentic video production system

11 production pipelines, 47 tools, 124 agent skills.
Supports cloud APIs (fal.ai, OpenAI, ElevenLabs, Suno, HeyGen, Runway) and
free local providers (diffusers, Piper TTS, WAN 2.1, Hunyuan, CogVideo).

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
calesthio
2026-03-29 08:25:17 -07:00
commit a3e735cc7a
1147 changed files with 240221 additions and 0 deletions
+1
View File
@@ -0,0 +1 @@
"""Subtitle generation and formatting tools."""
+276
View File
@@ -0,0 +1,276 @@
"""Subtitle generation tool.
Converts word-level timestamps from the transcriber into SRT, VTT,
or caption JSON formats. Pure Python — no external dependencies beyond
the standard library.
"""
from __future__ import annotations
import json
import time
from pathlib import Path
from typing import Any
from tools.base_tool import (
BaseTool,
Determinism,
ExecutionMode,
ResourceProfile,
ToolResult,
ToolStability,
ToolTier,
)
class SubtitleGen(BaseTool):
name = "subtitle_gen"
version = "0.1.0"
tier = ToolTier.CORE
capability = "subtitle"
provider = "openmontage"
stability = ToolStability.EXPERIMENTAL
execution_mode = ExecutionMode.SYNC
determinism = Determinism.DETERMINISTIC
dependencies = [] # pure Python
install_instructions = "No external dependencies required."
agent_skills = ["remotion-best-practices"]
capabilities = ["generate_srt", "generate_vtt", "generate_caption_json"]
input_schema = {
"type": "object",
"required": ["segments"],
"properties": {
"segments": {
"type": "array",
"description": "Transcript segments from transcriber (with words and timestamps)",
},
"format": {
"type": "string",
"enum": ["srt", "vtt", "json"],
"default": "srt",
},
"output_path": {"type": "string"},
"max_chars_per_line": {"type": "integer", "default": 42},
"max_words_per_cue": {"type": "integer", "default": 8},
"highlight_style": {
"type": "string",
"enum": ["none", "word_by_word", "karaoke"],
"default": "none",
},
},
}
resource_profile = ResourceProfile(cpu_cores=1, ram_mb=128, vram_mb=0, disk_mb=10)
idempotency_key_fields = ["segments", "format", "max_words_per_cue"]
side_effects = ["writes subtitle file to output_path"]
user_visible_verification = [
"Play video with generated subtitles and verify timing",
]
def execute(self, inputs: dict[str, Any]) -> ToolResult:
segments = inputs["segments"]
fmt = inputs.get("format", "srt")
max_words = inputs.get("max_words_per_cue", 8)
max_chars = inputs.get("max_chars_per_line", 42)
highlight_style = inputs.get("highlight_style", "none")
output_path = inputs.get("output_path")
start = time.time()
# Build cues from word-level timestamps
cues = self._build_cues(segments, max_words, max_chars)
if fmt == "srt":
content = self._render_srt(cues, highlight_style)
ext = ".srt"
elif fmt == "vtt":
content = self._render_vtt(cues, highlight_style)
ext = ".vtt"
elif fmt == "json":
content = json.dumps({"cues": cues, "highlight_style": highlight_style}, indent=2)
ext = ".caption.json"
else:
return ToolResult(success=False, error=f"Unknown format: {fmt}")
if output_path is None:
output_path = f"subtitles{ext}"
out = Path(output_path)
out.parent.mkdir(parents=True, exist_ok=True)
out.write_text(content, encoding="utf-8")
elapsed = time.time() - start
return ToolResult(
success=True,
data={
"format": fmt,
"cue_count": len(cues),
"output": str(out),
},
artifacts=[str(out)],
duration_seconds=round(elapsed, 2),
)
def _build_cues(
self, segments: list[dict], max_words: int, max_chars: int
) -> list[dict]:
"""Group words into display cues respecting max_words and max_chars."""
# Collect all words with timestamps
all_words = []
for seg in segments:
words = seg.get("words", [])
if words:
all_words.extend(words)
elif "text" in seg:
# Fallback: segment-level only (no word timestamps)
all_words.append({
"word": seg["text"],
"start": seg["start"],
"end": seg["end"],
})
if not all_words:
return []
cues = []
buf: list[dict] = []
buf_text = ""
for w in all_words:
word_text = w["word"].strip()
candidate = f"{buf_text} {word_text}".strip() if buf_text else word_text
if buf and (len(buf) >= max_words or len(candidate) > max_chars):
cues.append({
"index": len(cues) + 1,
"start": buf[0]["start"],
"end": buf[-1]["end"],
"text": buf_text,
"words": [
{"word": b["word"].strip(), "start": b["start"], "end": b["end"]}
for b in buf
],
})
buf = []
buf_text = ""
buf.append(w)
buf_text = f"{buf_text} {word_text}".strip() if buf_text else word_text
# Flush remaining
if buf:
cues.append({
"index": len(cues) + 1,
"start": buf[0]["start"],
"end": buf[-1]["end"],
"text": buf_text,
"words": [
{"word": b["word"].strip(), "start": b["start"], "end": b["end"]}
for b in buf
],
})
return cues
def _render_srt(self, cues: list[dict], highlight_style: str = "none") -> str:
lines = []
if highlight_style == "word_by_word":
# Emit one cue per word for word-by-word reveal
idx = 1
for cue in cues:
for word_info in cue.get("words", []):
lines.append(str(idx))
lines.append(
f"{self._ts_srt(word_info['start'])} --> {self._ts_srt(word_info['end'])}"
)
lines.append(word_info["word"])
lines.append("")
idx += 1
elif highlight_style == "karaoke":
# Show full cue text but bold the active word using SRT HTML tags
for cue in cues:
words = cue.get("words", [])
if not words:
lines.append(str(cue["index"]))
lines.append(f"{self._ts_srt(cue['start'])} --> {self._ts_srt(cue['end'])}")
lines.append(cue["text"])
lines.append("")
continue
for wi, word_info in enumerate(words):
lines.append(str(cue["index"] * 100 + wi))
lines.append(
f"{self._ts_srt(word_info['start'])} --> {self._ts_srt(word_info['end'])}"
)
parts = []
for wj, w in enumerate(words):
if wj == wi:
parts.append(f"<b>{w['word']}</b>")
else:
parts.append(w["word"])
lines.append(" ".join(parts))
lines.append("")
else:
for cue in cues:
lines.append(str(cue["index"]))
lines.append(f"{self._ts_srt(cue['start'])} --> {self._ts_srt(cue['end'])}")
lines.append(cue["text"])
lines.append("")
return "\n".join(lines)
def _render_vtt(self, cues: list[dict], highlight_style: str = "none") -> str:
lines = ["WEBVTT", ""]
if highlight_style == "word_by_word":
for cue in cues:
for word_info in cue.get("words", []):
lines.append(
f"{self._ts_vtt(word_info['start'])} --> {self._ts_vtt(word_info['end'])}"
)
lines.append(word_info["word"])
lines.append("")
elif highlight_style == "karaoke":
for cue in cues:
words = cue.get("words", [])
if not words:
lines.append(f"{self._ts_vtt(cue['start'])} --> {self._ts_vtt(cue['end'])}")
lines.append(cue["text"])
lines.append("")
continue
for wi, word_info in enumerate(words):
lines.append(
f"{self._ts_vtt(word_info['start'])} --> {self._ts_vtt(word_info['end'])}"
)
parts = []
for wj, w in enumerate(words):
if wj == wi:
parts.append(f"<b>{w['word']}</b>")
else:
parts.append(w["word"])
lines.append(" ".join(parts))
lines.append("")
else:
for cue in cues:
lines.append(f"{self._ts_vtt(cue['start'])} --> {self._ts_vtt(cue['end'])}")
lines.append(cue["text"])
lines.append("")
return "\n".join(lines)
@staticmethod
def _ts_srt(seconds: float) -> str:
"""Format seconds as SRT timestamp: HH:MM:SS,mmm"""
h = int(seconds // 3600)
m = int((seconds % 3600) // 60)
s = int(seconds % 60)
ms = int(round((seconds % 1) * 1000))
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
@staticmethod
def _ts_vtt(seconds: float) -> str:
"""Format seconds as VTT timestamp: HH:MM:SS.mmm"""
h = int(seconds // 3600)
m = int((seconds % 3600) // 60)
s = int(seconds % 60)
ms = int(round((seconds % 1) * 1000))
return f"{h:02d}:{m:02d}:{s:02d}.{ms:03d}"