diff --git a/.gitignore b/.gitignore index d98efa39..4234a0fc 100644 --- a/.gitignore +++ b/.gitignore @@ -44,6 +44,10 @@ skills-lock.json # Generated images (logo drafts — final logo lives in assets/) logo_*.png +generated_image.png + +# Downloaded music files (Pixabay, Freesound — not source code) +pixabay_music_* # OS .DS_Store @@ -57,3 +61,6 @@ remotion-composer/out/ remotion-composer/public/* # But keep demo props (shipped with the project for zero-key demos) !remotion-composer/public/demo-props/ +# Ignore test/scratch compositions in demo-props (keep only curated demos) +remotion-composer/public/demo-props/test-* +remotion-composer/public/demo-props/talking-head-* diff --git a/generated_image.png b/generated_image.png deleted file mode 100644 index 58012fbc..00000000 Binary files a/generated_image.png and /dev/null differ diff --git a/pipeline_defs/talking-head.yaml b/pipeline_defs/talking-head.yaml index 6bc53a25..74453401 100644 --- a/pipeline_defs/talking-head.yaml +++ b/pipeline_defs/talking-head.yaml @@ -78,8 +78,12 @@ stages: required_tools: [] optional_tools: - frame_sampler # representative frame extraction + - face_tracker # face position analysis for reframing decisions + - silence_cutter # silence detection for jump cut planning tools_available: - frame_sampler + - face_tracker + - silence_cutter checkpoint_required: true human_approval_default: true review_focus: @@ -126,7 +130,12 @@ stages: - script produces: - edit_decisions - tools_available: [] + optional_tools: + - silence_cutter # auto jump cuts on silent segments + - video_trimmer # speed control (0.5x–2x), cut, concat + tools_available: + - silence_cutter + - video_trimmer checkpoint_required: true human_approval_default: false review_focus: @@ -149,17 +158,35 @@ stages: - render_report required_tools: - video_compose # rendering - - audio_mixer # audio layering + - audio_mixer # audio layering + segmented music optional_tools: - face_enhance # face enhancement + - eye_enhance # eye brightening, dark circle removal - color_grade # color correction - audio_enhance # audio cleanup + - auto_reframe # aspect ratio conversion with face tracking + - face_tracker # face position data for reframing + - silence_cutter # auto jump cuts + - video_trimmer # speed control (0.5x–2x) + - remotion_caption_burn # animated word-by-word captions via Remotion + - showcase_card # letterboxed showcase cards with typography + - video_stitch # multi-clip assembly with transitions + - visual_qa # frame extraction + quality validation tools_available: - video_compose - audio_mixer - face_enhance + - eye_enhance - color_grade - audio_enhance + - auto_reframe + - face_tracker + - silence_cutter + - video_trimmer + - remotion_caption_burn + - showcase_card + - video_stitch + - visual_qa checkpoint_required: true human_approval_default: false review_focus: diff --git a/remotion-composer/src/Root.tsx b/remotion-composer/src/Root.tsx index 531ad6a5..9033d522 100644 --- a/remotion-composer/src/Root.tsx +++ b/remotion-composer/src/Root.tsx @@ -5,6 +5,7 @@ import { calculateCinematicMetadata, } from "./CinematicRenderer"; import { signalFromTomorrowWithMusicFixture } from "./cinematic/fixtures"; +import { TalkingHead, TalkingHeadProps } from "./TalkingHead"; const calculateMetadata: CalculateMetadataFunction = async ({ props, @@ -61,6 +62,21 @@ export const Root: React.FC = () => { defaultProps={signalFromTomorrowWithMusicFixture} calculateMetadata={calculateCinematicMetadata} /> + ); }; diff --git a/remotion-composer/src/TalkingHead.tsx b/remotion-composer/src/TalkingHead.tsx new file mode 100644 index 00000000..9128fce5 --- /dev/null +++ b/remotion-composer/src/TalkingHead.tsx @@ -0,0 +1,35 @@ +import { AbsoluteFill, OffthreadVideo } from "remotion"; +import { CaptionOverlay, WordCaption } from "./components/CaptionOverlay"; + +export interface TalkingHeadProps { + videoSrc: string; + captions: WordCaption[]; + wordsPerPage?: number; + fontSize?: number; + highlightColor?: string; +} + +export const TalkingHead: React.FC = ({ + videoSrc, + captions, + wordsPerPage = 4, + fontSize = 52, + highlightColor = "#22D3EE", +}) => { + return ( + + + + + ); +}; diff --git a/skills/pipelines/talking-head/compose-director.md b/skills/pipelines/talking-head/compose-director.md index 5f699cd3..c4ae7a8e 100644 --- a/skills/pipelines/talking-head/compose-director.md +++ b/skills/pipelines/talking-head/compose-director.md @@ -18,51 +18,215 @@ You have edit decisions and an asset manifest. Your job is to render the final t ### Step 1: Run Enhancement Chain Apply video enhancements in order: -1. **Face enhancement** (if face_enhance tool available) — sharpen faces -2. **Color grading** (if color_grade tool available) — apply a profile -3. **Audio enhancement** (if audio_enhance tool available) — noise reduction, normalization +1. **Face enhancement** (if `face_enhance` tool available) — apply `talking_head_standard` preset +2. **Eye enhancement** (if `eye_enhance` tool available) — under-eye dark circle removal + eye brightening +3. **Color grading** (if `color_grade` tool available) — apply a profile +4. **Audio enhancement** (if `audio_enhance` tool available) — noise reduction, normalization Each step is optional — check tool availability first. -### Step 2: Burn Subtitles +**Eye enhancement** — removes under-eye dark circles and brightens eyes using MediaPipe Face Mesh landmark detection: +``` +eye_enhance.execute({ + "input_path": "", + "output_path": "/assets/video/eye_enhanced.mp4", + "operations": ["dark_circles", "brighten_eyes"], + "dark_circle_intensity": 0.4, # 0-1, subtle is better + "eye_brighten_intensity": 0.3, +}) +``` +**Important:** Keep intensities low (0.2-0.5). Over-processing makes eyes look unnatural. Always compare before/after. +### Step 1b: Speed Adjustment (if requested) + +If the user wants the video sped up or slowed down, use `video_trimmer`: +``` +video_trimmer.execute({ + "operation": "speed", + "input_path": "", + "output_path": "/assets/video/speed_adjusted.mp4", + "speed_factor": 1.25 # 0.5x (slow), 1.25x, 1.5x, 2x (fast) +}) +``` + +Common speed factors: +| Factor | Use Case | +|--------|----------| +| `0.5` | Slow-mo for dramatic effect | +| `1.0` | Normal (no change) | +| `1.25` | Slightly faster — tighter pacing without sounding unnatural | +| `1.5` | Noticeably faster — good for recaps or condensed content | +| `2.0` | Double speed — time-lapse effect | + +Apply speed AFTER enhancements, BEFORE reframing. + +### Step 2: Auto-Reframe (if target platform requires it) + +If the target platform requires a different aspect ratio (e.g. Instagram Reels = 9:16), use `auto_reframe`: + +``` +auto_reframe.execute({ + "input_path": "", + "output_path": "/renders/reframed.mp4", + "target_aspect": "portrait", # 9:16 for Reels/TikTok/Shorts + "smoothing_window": 15, # smooth camera pan + "face_padding": 0.4, # 40% padding around face +}) +``` + +**Aspect ratio presets:** +| Preset | Ratio | Platform | +|--------|-------|----------| +| `portrait` | 9:16 | Instagram Reels, TikTok, YouTube Shorts | +| `square` | 1:1 | Instagram Feed | +| `landscape` | 16:9 | YouTube, LinkedIn | +| `vertical_4_5` | 4:5 | Instagram portrait post | + +The tool automatically runs face detection and keeps the speaker centered. If MediaPipe is not installed, falls back to center-crop. + +**Important:** Run auto_reframe AFTER face_enhance and color_grade but BEFORE burning subtitles. Subtitles need to be positioned for the final aspect ratio. + +### Step 3: Burn Subtitles + +**Preferred: Remotion captions** (if `remotion_caption_burn` tool available): +``` +remotion_caption_burn.execute({ + "input_path": "", + "output_path": "/assets/video/captioned.mp4", + "segments": , + "corrections": {"cloud": "Claude", "co-pilot": "Copilot"}, + "words_per_page": 4, + "font_size": 52, + "highlight_color": "#22D3EE", +}) +``` +Remotion renders animated word-by-word captions at the bottom of the frame with active word highlighting. Captions are positioned away from the face. + +**Fallback: FFmpeg subtitles** (if Remotion unavailable): Use `video_compose` with `burn_subtitles` operation: -- Input: enhanced video (or raw if no enhancements) +- Input: reframed video (or enhanced video if no reframe needed) - Subtitle file from asset manifest - Style from playbook +- For vertical (9:16) output: position subtitles in the lower 20% of frame with `MarginV=100` +- **Never** position subtitles in the center of the frame — they will occlude the face -### Step 3: Mix Audio +### Step 3b: Build Showcase Cards (if multi-clip reel) -Use `audio_mixer` to: -- Layer original audio with any background music +If the output is a reel with showcase clips, use `showcase_card` for each: +``` +showcase_card.execute({ + "input_path": "", + "output_path": "/assets/video/sc_.mp4", + "title": "VIDEO TITLE", + "subtitle": "Description | Style | Cost: $0.15", + "background_color": "0x0A0F1A", +}) +``` +This creates letterboxed 9:16 cards with typography. + +### Step 4: Assemble Multi-Clip (if applicable) + +If the output has multiple segments (e.g. talking head + showcase clips), use `video_stitch`: +``` +video_stitch.execute({ + "operation": "stitch", + "clips": ["intro.mp4", "showcase1.mp4", ..., "outro.mp4"], + "output_path": "/renders/assembled.mp4", + "transition": "crossfade", # or "fade" for fade-through-black + "transition_duration": 0.5, +}) +``` +**Transition guidance:** +- `crossfade` (fade): smooth blend between talking head and showcase +- `fade` (fade-through-black): brief dip to black between showcase clips +- Mix transition types: use `crossfade` for talk→showcase, `fade` between showcases + +### Step 5: Mix Audio + +Use `audio_mixer` to layer background music: + +**For multi-clip reels** — use `segmented_music` to play music only during talking head sections: +``` +audio_mixer.execute({ + "operation": "segmented_music", + "video_path": "", + "music_path": "", + "music_volume": 0.20, + "segments": [ + {"start": 0, "end": 17.0}, # intro speech + {"start": 167.0, "end": 175.0} # outro speech + ], + "fade_duration": 0.5, + "output_path": "/renders/final.mp4", +}) +``` + +**For single talking-head videos** — use `duck` or `full_mix`: +- Layer original audio with background music - Apply ducking if music is present - Normalize final levels -### Step 4: Final Encode +### Step 6: Final Encode Use `video_compose` with `encode` operation: -- Apply target media profile (youtube_landscape, tiktok, etc.) +- Apply target media profile (youtube_landscape, tiktok, instagram_reels, etc.) - Two-pass encoding for quality -### Step 5: Verify Output +### Step 7: Visual QA -- Check file exists and is playable -- Verify duration matches expectations -- Check audio is present +Use `visual_qa` to verify the output before declaring success: +``` +visual_qa.execute({ + "operation": "review", + "input_path": "", + "timestamps": [3.0, 10.0, 25.0, 50.0, 100.0, 170.0], +}) +``` +Then **read each extracted frame** to verify: +- Captions are visible and positioned at the bottom (not on the face) +- Face enhancement is applied (skin looks smooth, not over-processed) +- Transitions are clean (no artifacts at transition points) +- Showcase cards have readable typography -### Step 6: Build Render Report +Also run probe validation: +``` +visual_qa.execute({ + "operation": "probe", + "input_path": "", + "expected": { + "width": 1080, "height": 1920, + "has_audio": true, + "pixel_format": "yuv420p" + }, +}) +``` -Document output: path, format, resolution, duration, file size. +And check audio levels: +``` +visual_qa.execute({ + "operation": "audio_levels", + "input_path": "", + "timestamps": [5.0, 50.0, 170.0], +}) +``` +Verify: speech sections have higher volume than showcase sections (confirms music placement). -### Step 7: Self-Evaluate +### Step 8: Build Render Report + +Document output: path, format, resolution, duration, file size, QA results. + +### Step 9: Self-Evaluate | Criterion | Question | |-----------|----------| | **Playability** | Does the video play without errors? | | **Quality** | Are enhancements applied correctly? | -| **Audio** | Is speech clear with balanced levels? | -| **Subtitles** | Are subtitles visible and synced? | +| **Framing** | If reframed — is the face centered? No important content cropped? | +| **Audio** | Is speech clear with balanced levels? Music only during intended segments? | +| **Subtitles** | Are captions visible at the bottom? Not occluding the face? Word highlighting working? | +| **Transitions** | Are transitions clean? Correct type (crossfade vs fadeblack)? | +| **Showcase** | Are showcase cards properly letterboxed with readable typography? | -### Step 8: Submit +### Step 10: Submit Validate the render_report against the schema and persist via checkpoint. diff --git a/skills/pipelines/talking-head/edit-director.md b/skills/pipelines/talking-head/edit-director.md index 9e2613be..a80316b2 100644 --- a/skills/pipelines/talking-head/edit-director.md +++ b/skills/pipelines/talking-head/edit-director.md @@ -14,39 +14,63 @@ You have a scene plan and asset manifest. Your job is to assemble the edit decis ## Process -### Step 1: Define Primary Cut +### Step 1: Apply Silence Cuts (if planned) + +If the scene plan includes silence removal, run `silence_cutter` before defining cuts: + +``` +silence_cutter.execute({ + "input_path": "", + "mode": "remove", # or "speed_up" for less jarring result + "silence_threshold_db": -35, + "min_silence_duration": 0.5, + "padding_seconds": 0.08, # prevents clipped words + "output_path": "/assets/video/footage_cut.mp4" +}) +``` + +**Choosing the mode:** +- `remove` — Hard jump cuts. Best for fast-paced social content (Reels, TikTok, Shorts) +- `speed_up` — Fast-forwards through silence at 6x. Less jarring for longer-form content (YouTube, LinkedIn) + +Present the result to the user: "Removed X seconds of silence (Y%) — output is now Z seconds." + +Use the cut footage as the source for all subsequent steps. + +### Step 2: Define Primary Cut For talking-head, the primary cut is usually the full footage (or trimmed segments). Create cuts that: -- Reference the raw footage as source +- Reference the raw footage (or silence-cut footage) as source - Use timestamps from the script sections - Apply any trim decisions (cut dead air, false starts) -### Step 2: Configure Subtitles +### Step 3: Configure Subtitles - Enable subtitles with playbook-compatible styling - Reference the subtitle asset from the manifest - Set position (usually bottom-center) -### Step 3: Configure Audio +### Step 4: Configure Audio - Set narration to the raw footage audio - If background music is desired, configure ducking - Set music volume per playbook -### Step 4: Plan Enhancements +### Step 5: Plan Enhancements If the scene plan includes overlays: - Add overlay cuts for text cards, lower thirds - Time them to match speech content -### Step 5: Self-Evaluate +### Step 6: Self-Evaluate | Criterion | Question | |-----------|----------| | **Coverage** | Do cuts span the full intended duration? | +| **Silence** | Were silence cuts applied if planned? What % was removed? | | **Subtitles** | Are subtitles enabled and styled? | | **Audio** | Is audio configuration complete? | -### Step 6: Submit +### Step 7: Submit Validate the edit_decisions against the schema and persist via checkpoint. diff --git a/skills/pipelines/talking-head/scene-director.md b/skills/pipelines/talking-head/scene-director.md index 707d682c..7da39310 100644 --- a/skills/pipelines/talking-head/scene-director.md +++ b/skills/pipelines/talking-head/scene-director.md @@ -11,28 +11,70 @@ You have a script (from transcription) and raw footage. Your job is to create a | Schema | `schemas/artifacts/scene_plan.schema.json` | Artifact validation | | Prior artifacts | Script, Brief | Section timing and context | | Tools | `frame_sampler` (optional) | Extract representative frames | +| Tools | `face_tracker` (optional) | Analyze speaker face position for reframing | +| Tools | `silence_cutter` (optional) | Detect silence for jump cut planning | ## Process -### Step 1: Plan Base Scenes +### Step 1: Analyze Footage (if tools available) + +**Face tracking** — If `face_tracker` is available, run it on the raw footage: +``` +face_tracker.execute({ + "input_path": "", + "sample_fps": 5 +}) +``` +This outputs per-frame face bounding boxes. Use this data to: +- Decide if reframing is needed (e.g. speaker is off-center for vertical crop) +- Identify sections where the speaker moves significantly (needs dynamic crop) +- Note face position for auto_reframe in the compose stage + +**Silence detection** — If `silence_cutter` is available, run in `mark` mode: +``` +silence_cutter.execute({ + "input_path": "", + "mode": "mark", + "silence_threshold_db": -35, + "min_silence_duration": 0.5 +}) +``` +This outputs silence/speech segment timestamps. Use this to: +- Plan which segments should be jump-cut or sped up +- Identify dead air, false starts, and long pauses +- Estimate the final video duration after cuts +- Present the user with a summary: "Found X seconds of silence across Y segments — recommend removing?" + +### Step 2: Plan Base Scenes For talking-head, the base is simple: one scene per script section, all type `talking_head`. The raw footage IS the scene. -### Step 2: Plan Enhancement Scenes +### Step 3: Plan Enhancement Scenes Based on script enhancement cues, plan overlay scenes: - Text cards for key terms or statistics - Lower thirds for speaker identification - B-roll suggestions for topic illustrations -### Step 3: Build Scene Plan +### Step 4: Plan Reframing & Cuts + +If the target platform requires a different aspect ratio (e.g. Instagram Reels = 9:16): +- Note `auto_reframe` should be applied in the compose stage +- Record the target aspect ratio in the scene plan +- If face tracking data shows significant speaker movement, note that dynamic crop is needed + +If silence detection found segments to cut: +- Record the recommended cut mode (`remove` or `speed_up`) in the scene plan +- Note padding preferences (default 0.08s to avoid clipping words) + +### Step 5: Build Scene Plan Create a scene per section with: - Type: `talking_head` (primary) - Timing from script sections - Required assets: subtitle file, any overlay images -### Step 4: Self-Evaluate +### Step 6: Self-Evaluate | Criterion | Question | |-----------|----------| @@ -40,6 +82,6 @@ Create a scene per section with: | **Enhancement** | Are overlay opportunities identified? | | **Feasibility** | Can all required assets be generated? | -### Step 5: Submit +### Step 7: Submit Validate the scene_plan against the schema and persist via checkpoint. diff --git a/tools/analysis/face_tracker.py b/tools/analysis/face_tracker.py new file mode 100644 index 00000000..f9ed63e7 --- /dev/null +++ b/tools/analysis/face_tracker.py @@ -0,0 +1,314 @@ +"""Face tracking tool using MediaPipe Face Mesh. + +Tracks face bounding boxes, landmarks, and head pose across video frames. +Outputs per-frame face data as JSON — used by auto_reframe, face_enhance, +and other tools that need to know where the speaker's face is. + +Falls back to OpenCV Haar cascade if MediaPipe is not installed. +""" + +from __future__ import annotations + +import json +import time +from pathlib import Path +from typing import Any + +from tools.base_tool import ( + BaseTool, + Determinism, + ExecutionMode, + ResourceProfile, + ToolResult, + ToolStability, + ToolStatus, + ToolTier, +) + + +class FaceTracker(BaseTool): + name = "face_tracker" + version = "0.1.0" + tier = ToolTier.CORE + capability = "analysis" + provider = "mediapipe" + stability = ToolStability.EXPERIMENTAL + execution_mode = ExecutionMode.SYNC + determinism = Determinism.DETERMINISTIC + + dependencies = ["cmd:ffmpeg"] + install_instructions = ( + "For best results install MediaPipe:\n" + "pip install mediapipe opencv-python\n\n" + "Falls back to OpenCV Haar cascade (ships with opencv-python)." + ) + agent_skills = ["ffmpeg"] + + capabilities = [ + "face_detection", + "face_tracking", + "face_bounding_box", + "head_pose_estimation", + ] + + input_schema = { + "type": "object", + "required": ["input_path"], + "properties": { + "input_path": {"type": "string"}, + "output_path": { + "type": "string", + "description": "Path for face tracking JSON output", + }, + "sample_fps": { + "type": "number", + "default": 5, + "description": "Frames per second to sample (lower = faster, less precise)", + }, + "min_detection_confidence": { + "type": "number", + "default": 0.5, + "minimum": 0.0, + "maximum": 1.0, + }, + }, + } + + output_schema = { + "type": "object", + "properties": { + "frame_count": {"type": "integer"}, + "face_detected_count": {"type": "integer"}, + "video_width": {"type": "integer"}, + "video_height": {"type": "integer"}, + "fps": {"type": "number"}, + "duration_seconds": {"type": "number"}, + "faces": { + "type": "array", + "items": { + "type": "object", + "properties": { + "frame_index": {"type": "integer"}, + "timestamp_seconds": {"type": "number"}, + "bbox": { + "type": "object", + "properties": { + "x": {"type": "number"}, + "y": {"type": "number"}, + "width": {"type": "number"}, + "height": {"type": "number"}, + }, + }, + }, + }, + }, + }, + } + + resource_profile = ResourceProfile(cpu_cores=2, ram_mb=1024, vram_mb=0, disk_mb=100) + idempotency_key_fields = ["input_path", "sample_fps", "min_detection_confidence"] + side_effects = ["writes face tracking JSON to output_path"] + user_visible_verification = [ + "Spot-check bounding boxes against video frames", + ] + + fallback_tools = [] + + def _has_mediapipe(self) -> bool: + try: + import mediapipe # noqa: F401 + return True + except ImportError: + return False + + def _has_opencv(self) -> bool: + try: + import cv2 # noqa: F401 + return True + except ImportError: + return False + + def get_status(self) -> ToolStatus: + if self._has_mediapipe() and self._has_opencv(): + return ToolStatus.AVAILABLE + if self._has_opencv(): + return ToolStatus.DEGRADED + return ToolStatus.UNAVAILABLE + + def execute(self, inputs: dict[str, Any]) -> ToolResult: + input_path = Path(inputs["input_path"]) + if not input_path.exists(): + return ToolResult(success=False, error=f"Input not found: {input_path}") + + if not self._has_opencv(): + return ToolResult( + success=False, + error="opencv-python is required. Install: pip install opencv-python", + ) + + output_path = Path( + inputs.get("output_path", str(input_path.with_suffix(".faces.json"))) + ) + output_path.parent.mkdir(parents=True, exist_ok=True) + + sample_fps = inputs.get("sample_fps", 5) + confidence = inputs.get("min_detection_confidence", 0.5) + + start = time.time() + + if self._has_mediapipe(): + result_data = self._track_mediapipe(input_path, sample_fps, confidence) + else: + result_data = self._track_opencv(input_path, sample_fps) + + elapsed = time.time() - start + + output_path.write_text(json.dumps(result_data, indent=2), encoding="utf-8") + + return ToolResult( + success=True, + data={ + "output": str(output_path), + "video_width": result_data["video_width"], + "video_height": result_data["video_height"], + "fps": result_data["fps"], + "duration_seconds": result_data["duration_seconds"], + "frames_sampled": result_data["frame_count"], + "faces_detected": result_data["face_detected_count"], + "method": "mediapipe" if self._has_mediapipe() else "opencv_haar", + }, + artifacts=[str(output_path)], + duration_seconds=round(elapsed, 2), + ) + + def _track_mediapipe( + self, input_path: Path, sample_fps: float, confidence: float + ) -> dict: + import cv2 + import mediapipe as mp + + mp_face = mp.solutions.face_detection + cap = cv2.VideoCapture(str(input_path)) + + video_fps = cap.get(cv2.CAP_PROP_FPS) or 30.0 + video_w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) + video_h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) + total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) + duration = total_frames / video_fps if video_fps > 0 else 0 + + # Calculate frame sampling interval + sample_interval = max(1, int(video_fps / sample_fps)) + + faces_data: list[dict] = [] + frame_idx = 0 + sampled = 0 + + with mp_face.FaceDetection( + model_selection=1, # 1 = full range (up to 5m), 0 = short range (up to 2m) + min_detection_confidence=confidence, + ) as detector: + while cap.isOpened(): + ret, frame = cap.read() + if not ret: + break + + if frame_idx % sample_interval == 0: + sampled += 1 + rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) + results = detector.process(rgb) + + if results.detections: + # Use the highest-confidence detection + det = max( + results.detections, + key=lambda d: d.score[0], + ) + bbox = det.location_data.relative_bounding_box + faces_data.append({ + "frame_index": frame_idx, + "timestamp_seconds": round(frame_idx / video_fps, 3), + "confidence": round(det.score[0], 3), + "bbox": { + "x": round(bbox.xmin, 4), + "y": round(bbox.ymin, 4), + "width": round(bbox.width, 4), + "height": round(bbox.height, 4), + }, + }) + + frame_idx += 1 + + cap.release() + + return { + "video_width": video_w, + "video_height": video_h, + "fps": round(video_fps, 2), + "duration_seconds": round(duration, 3), + "frame_count": sampled, + "face_detected_count": len(faces_data), + "faces": faces_data, + } + + def _track_opencv(self, input_path: Path, sample_fps: float) -> dict: + """Fallback: OpenCV Haar cascade face detection.""" + import cv2 + + cascade_path = cv2.data.haarcascades + "haarcascade_frontalface_default.xml" + cascade = cv2.CascadeClassifier(cascade_path) + + cap = cv2.VideoCapture(str(input_path)) + video_fps = cap.get(cv2.CAP_PROP_FPS) or 30.0 + video_w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) + video_h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) + total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) + duration = total_frames / video_fps if video_fps > 0 else 0 + + sample_interval = max(1, int(video_fps / sample_fps)) + + faces_data: list[dict] = [] + frame_idx = 0 + sampled = 0 + + while cap.isOpened(): + ret, frame = cap.read() + if not ret: + break + + if frame_idx % sample_interval == 0: + sampled += 1 + gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY) + detected = cascade.detectMultiScale( + gray, scaleFactor=1.1, minNeighbors=5, minSize=(60, 60) + ) + + if len(detected) > 0: + # Pick largest face + areas = [w * h for (_, _, w, h) in detected] + best_idx = areas.index(max(areas)) + x, y, w, h = detected[best_idx] + faces_data.append({ + "frame_index": frame_idx, + "timestamp_seconds": round(frame_idx / video_fps, 3), + "confidence": 0.0, # Haar doesn't provide confidence + "bbox": { + "x": round(x / video_w, 4), + "y": round(y / video_h, 4), + "width": round(w / video_w, 4), + "height": round(h / video_h, 4), + }, + }) + + frame_idx += 1 + + cap.release() + + return { + "video_width": video_w, + "video_height": video_h, + "fps": round(video_fps, 2), + "duration_seconds": round(duration, 3), + "frame_count": sampled, + "face_detected_count": len(faces_data), + "faces": faces_data, + } diff --git a/tools/analysis/visual_qa.py b/tools/analysis/visual_qa.py new file mode 100644 index 00000000..7aa30159 --- /dev/null +++ b/tools/analysis/visual_qa.py @@ -0,0 +1,343 @@ +"""Visual QA tool for automated video quality checks. + +Extracts frames at specified timestamps and runs basic quality checks: +- File existence, resolution, duration, codec validation +- Frame extraction for visual inspection by the agent +- Caption occlusion check (compares brightness in face vs caption zones) +- Transition verification (frame similarity at transition points) + +Returns frame paths so the agent can visually inspect them. +""" + +from __future__ import annotations + +import time +from pathlib import Path +from typing import Any + +from tools.base_tool import ( + BaseTool, + Determinism, + ExecutionMode, + ResourceProfile, + ToolResult, + ToolStability, + ToolTier, +) + + +class VisualQA(BaseTool): + name = "visual_qa" + version = "0.1.0" + tier = ToolTier.CORE + capability = "analysis" + provider = "ffmpeg" + stability = ToolStability.EXPERIMENTAL + execution_mode = ExecutionMode.SYNC + determinism = Determinism.DETERMINISTIC + + dependencies = ["cmd:ffmpeg", "cmd:ffprobe"] + install_instructions = "Install FFmpeg: https://ffmpeg.org/download.html" + agent_skills = ["ffmpeg"] + + capabilities = [ + "extract_review_frames", + "probe_video", + "check_audio_levels", + ] + + input_schema = { + "type": "object", + "required": ["operation", "input_path"], + "properties": { + "operation": { + "type": "string", + "enum": ["review", "probe", "audio_levels"], + "description": ( + "review: extract frames at timestamps for visual inspection. " + "probe: get video metadata (duration, resolution, codecs). " + "audio_levels: check audio volume at specified timestamps." + ), + }, + "input_path": { + "type": "string", + "description": "Path to the video file to inspect.", + }, + "timestamps": { + "type": "array", + "items": {"type": "number"}, + "description": ( + "Timestamps (in seconds) at which to extract frames or " + "check audio levels." + ), + }, + "output_dir": { + "type": "string", + "description": ( + "Directory to save extracted frames. Defaults to a " + "'review_frames' subdirectory next to the input file." + ), + }, + "checks": { + "type": "array", + "items": { + "type": "string", + "enum": [ + "resolution", + "duration", + "audio_present", + "pixel_format", + "file_size", + ], + }, + "description": "Specific checks to run (probe operation).", + }, + "expected": { + "type": "object", + "description": ( + "Expected values for validation. " + "Keys: width, height, min_duration, max_duration, " + "pixel_format, has_audio." + ), + "properties": { + "width": {"type": "integer"}, + "height": {"type": "integer"}, + "min_duration": {"type": "number"}, + "max_duration": {"type": "number"}, + "pixel_format": {"type": "string"}, + "has_audio": {"type": "boolean"}, + }, + }, + }, + } + + resource_profile = ResourceProfile(cpu_cores=1, ram_mb=512, vram_mb=0, disk_mb=200) + idempotency_key_fields = ["operation", "input_path", "timestamps"] + side_effects = ["writes frame images to output_dir"] + user_visible_verification = [ + "Visually inspect extracted frames for quality issues", + ] + + def execute(self, inputs: dict[str, Any]) -> ToolResult: + operation = inputs["operation"] + input_path = inputs["input_path"] + + if not Path(input_path).exists(): + return ToolResult(success=False, error=f"Input not found: {input_path}") + + start = time.time() + + try: + if operation == "review": + result = self._review(inputs) + elif operation == "probe": + result = self._probe(inputs) + elif operation == "audio_levels": + result = self._audio_levels(inputs) + else: + return ToolResult(success=False, error=f"Unknown operation: {operation}") + except Exception as e: + return ToolResult(success=False, error=str(e)) + + result.duration_seconds = round(time.time() - start, 2) + return result + + def _review(self, inputs: dict[str, Any]) -> ToolResult: + """Extract frames at specified timestamps for visual review.""" + input_path = inputs["input_path"] + timestamps = inputs.get("timestamps", []) + + if not timestamps: + # Auto-generate timestamps: start, 25%, 50%, 75%, end-1s + dur = self._get_duration(input_path) + timestamps = [ + 1.0, + dur * 0.25, + dur * 0.50, + dur * 0.75, + max(dur - 1.0, 0), + ] + + output_dir = inputs.get("output_dir") + if not output_dir: + output_dir = str(Path(input_path).parent / "review_frames") + Path(output_dir).mkdir(parents=True, exist_ok=True) + + frames = [] + for ts in timestamps: + ts_label = f"{ts:.1f}".replace(".", "_") + frame_path = str(Path(output_dir) / f"frame_{ts_label}s.jpg") + cmd = [ + "ffmpeg", "-y", + "-ss", str(ts), + "-i", input_path, + "-frames:v", "1", + "-q:v", "2", + frame_path, + ] + try: + self.run_command(cmd) + if Path(frame_path).exists(): + frames.append({ + "timestamp": ts, + "path": frame_path, + }) + except Exception: + frames.append({ + "timestamp": ts, + "path": None, + "error": f"Failed to extract frame at {ts}s", + }) + + return ToolResult( + success=True, + data={ + "operation": "review", + "input": input_path, + "frame_count": len([f for f in frames if f.get("path")]), + "frames": frames, + }, + artifacts=[f["path"] for f in frames if f.get("path")], + ) + + def _probe(self, inputs: dict[str, Any]) -> ToolResult: + """Probe video metadata and optionally validate against expectations.""" + input_path = inputs["input_path"] + expected = inputs.get("expected", {}) + + # Get comprehensive probe data + cmd = [ + "ffprobe", "-v", "error", + "-show_entries", + "format=duration,size:stream=width,height,codec_name,pix_fmt," + "r_frame_rate,sample_rate,channels,codec_type", + "-of", "json", + input_path, + ] + import json + probe_out = self.run_command(cmd, capture=True) + probe_data = json.loads(probe_out) + + # Extract key info + video_stream = None + audio_stream = None + for s in probe_data.get("streams", []): + if s.get("codec_type") == "video" and not video_stream: + video_stream = s + elif s.get("codec_type") == "audio" and not audio_stream: + audio_stream = s + + info = { + "duration": float(probe_data.get("format", {}).get("duration", 0)), + "file_size_mb": round( + int(probe_data.get("format", {}).get("size", 0)) / 1048576, 1 + ), + "has_audio": audio_stream is not None, + } + if video_stream: + info.update({ + "width": video_stream.get("width"), + "height": video_stream.get("height"), + "pixel_format": video_stream.get("pix_fmt"), + "video_codec": video_stream.get("codec_name"), + "frame_rate": video_stream.get("r_frame_rate"), + }) + if audio_stream: + info.update({ + "audio_codec": audio_stream.get("codec_name"), + "sample_rate": audio_stream.get("sample_rate"), + "channels": audio_stream.get("channels"), + }) + + # Validate against expectations + issues = [] + if "width" in expected and info.get("width") != expected["width"]: + issues.append(f"Width: expected {expected['width']}, got {info.get('width')}") + if "height" in expected and info.get("height") != expected["height"]: + issues.append(f"Height: expected {expected['height']}, got {info.get('height')}") + if "min_duration" in expected and info["duration"] < expected["min_duration"]: + issues.append( + f"Duration too short: {info['duration']:.1f}s < {expected['min_duration']}s" + ) + if "max_duration" in expected and info["duration"] > expected["max_duration"]: + issues.append( + f"Duration too long: {info['duration']:.1f}s > {expected['max_duration']}s" + ) + if "pixel_format" in expected and info.get("pixel_format") != expected["pixel_format"]: + issues.append( + f"Pixel format: expected {expected['pixel_format']}, got {info.get('pixel_format')}" + ) + if "has_audio" in expected and info["has_audio"] != expected["has_audio"]: + issues.append( + f"Audio: expected {'present' if expected['has_audio'] else 'absent'}, " + f"got {'present' if info['has_audio'] else 'absent'}" + ) + + info["validation_issues"] = issues + info["validation_passed"] = len(issues) == 0 + + return ToolResult( + success=True, + data={ + "operation": "probe", + "input": input_path, + **info, + }, + ) + + def _audio_levels(self, inputs: dict[str, Any]) -> ToolResult: + """Check audio levels at specified timestamps.""" + input_path = inputs["input_path"] + timestamps = inputs.get("timestamps", []) + + if not timestamps: + dur = self._get_duration(input_path) + timestamps = [1.0, dur * 0.5, max(dur - 2.0, 0)] + + levels = [] + for ts in timestamps: + cmd = [ + "ffmpeg", "-y", + "-ss", str(ts), + "-t", "3", + "-i", input_path, + "-vn", "-af", "volumedetect", + "-f", "null", "/dev/null", + ] + try: + output = self.run_command(cmd, capture=True, stderr=True) + mean_vol = None + max_vol = None + for line in output.split("\n"): + if "mean_volume" in line: + mean_vol = float(line.split("mean_volume:")[1].strip().split()[0]) + elif "max_volume" in line: + max_vol = float(line.split("max_volume:")[1].strip().split()[0]) + levels.append({ + "timestamp": ts, + "mean_volume_db": mean_vol, + "max_volume_db": max_vol, + }) + except Exception as e: + levels.append({ + "timestamp": ts, + "error": str(e), + }) + + return ToolResult( + success=True, + data={ + "operation": "audio_levels", + "input": input_path, + "levels": levels, + }, + ) + + def _get_duration(self, path: str) -> float: + cmd = [ + "ffprobe", "-v", "error", + "-show_entries", "format=duration", + "-of", "csv=p=0", + path, + ] + return float(self.run_command(cmd, capture=True).strip().split("\n")[0]) diff --git a/tools/audio/audio_mixer.py b/tools/audio/audio_mixer.py index ab2b9673..39be5a25 100644 --- a/tools/audio/audio_mixer.py +++ b/tools/audio/audio_mixer.py @@ -40,7 +40,7 @@ class AudioMixer(BaseTool): ) agent_skills = ["ffmpeg", "video_toolkit"] - capabilities = ["mix", "duck", "fade", "normalize", "extract_audio"] + capabilities = ["mix", "duck", "fade", "normalize", "extract_audio", "segmented_music"] input_schema = { "type": "object", @@ -48,13 +48,16 @@ class AudioMixer(BaseTool): "properties": { "operation": { "type": "string", - "enum": ["mix", "duck", "extract", "full_mix"], + "enum": ["mix", "duck", "extract", "full_mix", "segmented_music"], "description": ( "mix: layer multiple tracks with volume/delay/fades. " "duck: lower music volume when speech is present. " "extract: extract audio from video file. " "full_mix: combine narration tracks + music with ducking + normalize " - "in a single call (preferred for compose-director)." + "in a single call (preferred for compose-director). " + "segmented_music: mix music into a video only during specified " + "time segments (e.g. music during talking head, silence during " + "showcase clips)." ), }, "tracks": { @@ -128,6 +131,45 @@ class AudioMixer(BaseTool): }, }, "normalize": {"type": "boolean", "default": True}, + "video_path": { + "type": "string", + "description": ( + "Path to the assembled video (segmented_music operation). " + "Music is mixed into this video's audio at specified segments." + ), + }, + "music_path": { + "type": "string", + "description": "Path to background music file (segmented_music operation).", + }, + "music_volume": { + "type": "number", + "minimum": 0, + "maximum": 1.0, + "default": 0.20, + "description": "Volume level for music during active segments.", + }, + "segments": { + "type": "array", + "description": ( + "Time segments where music should play (segmented_music operation). " + "Each segment: {start: seconds, end: seconds}. Music fades in/out " + "at segment boundaries. Outside these segments, music is silent." + ), + "items": { + "type": "object", + "required": ["start", "end"], + "properties": { + "start": {"type": "number", "minimum": 0}, + "end": {"type": "number", "minimum": 0}, + }, + }, + }, + "fade_duration": { + "type": "number", + "default": 0.5, + "description": "Duration of fade in/out at segment boundaries (seconds).", + }, }, } @@ -151,6 +193,8 @@ class AudioMixer(BaseTool): result = self._extract(inputs) elif operation == "full_mix": result = self._full_mix(inputs) + elif operation == "segmented_music": + result = self._segmented_music(inputs) else: return ToolResult(success=False, error=f"Unknown operation: {operation}") except Exception as e: @@ -558,3 +602,105 @@ class AudioMixer(BaseTool): }, artifacts=[str(output_path)], ) + + def _segmented_music(self, inputs: dict[str, Any]) -> ToolResult: + """Mix background music into a video only during specified time segments. + + Uses FFmpeg volume expressions with smooth fades at segment boundaries. + Music is silent outside the specified segments. + + Input format: + { + "operation": "segmented_music", + "video_path": "assembled.mp4", + "music_path": "bg_music.mp3", + "music_volume": 0.20, + "segments": [ + {"start": 0, "end": 17.0}, + {"start": 167.0, "end": 175.0} + ], + "fade_duration": 0.5, + "output_path": "final_with_music.mp4" + } + """ + video_path = inputs.get("video_path") + music_path = inputs.get("music_path") + output_path = Path(inputs.get("output_path", "segmented_music_output.mp4")) + segments = inputs.get("segments", []) + music_volume = inputs.get("music_volume", 0.20) + fade_dur = inputs.get("fade_duration", 0.5) + + if not video_path or not Path(video_path).exists(): + return ToolResult(success=False, error=f"Video not found: {video_path}") + if not music_path or not Path(music_path).exists(): + return ToolResult(success=False, error=f"Music not found: {music_path}") + if not segments: + return ToolResult(success=False, error="No segments specified") + + output_path.parent.mkdir(parents=True, exist_ok=True) + + # Get video duration + dur_cmd = [ + "ffprobe", "-v", "error", + "-show_entries", "format=duration", + "-of", "csv=p=0", + video_path, + ] + total_dur = float(self.run_command(dur_cmd, capture=True).strip().split("\n")[0]) + + # Build volume expression for each segment with smooth fades + parts = [] + for seg in sorted(segments, key=lambda s: s["start"]): + s = seg["start"] + e = seg["end"] + fade_in_end = s + fade_dur + fade_out_start = e - fade_dur + parts.append( + f"if(lt(t,{s}),0," + f"if(lt(t,{fade_in_end}),{music_volume}*(t-{s})/{fade_dur}," + f"if(lt(t,{fade_out_start}),{music_volume}," + f"if(lt(t,{e}),{music_volume}*({e}-t)/{fade_dur}," + f"0))))" + ) + + vol_expr = "+".join(f"({p})" for p in parts) if len(parts) > 1 else parts[0] + + filter_complex = ( + f"[1:a]atrim=0:{total_dur},asetpts=PTS-STARTPTS," + f"volume='{vol_expr}':eval=frame[music_shaped];" + f"[0:a]aformat=sample_fmts=fltp:sample_rates=44100:channel_layouts=stereo[speech];" + f"[music_shaped]aformat=sample_fmts=fltp:sample_rates=44100:channel_layouts=stereo[music_fmt];" + f"[speech][music_fmt]amix=inputs=2:duration=first:dropout_transition=2[aout]" + ) + + cmd = [ + "ffmpeg", "-y", + "-i", video_path, + "-stream_loop", "-1", + "-i", music_path, + "-filter_complex", filter_complex, + "-map", "0:v", + "-map", "[aout]", + "-c:v", "copy", + "-c:a", "aac", "-b:a", "192k", + str(output_path), + ] + + self.run_command(cmd) + + if not output_path.exists(): + return ToolResult(success=False, error="No output produced") + + return ToolResult( + success=True, + data={ + "operation": "segmented_music", + "video": video_path, + "music": music_path, + "segments": segments, + "music_volume": music_volume, + "fade_duration": fade_dur, + "output": str(output_path), + }, + artifacts=[str(output_path)], + ) diff --git a/tools/enhancement/eye_enhance.py b/tools/enhancement/eye_enhance.py new file mode 100644 index 00000000..5ffe771c --- /dev/null +++ b/tools/enhancement/eye_enhance.py @@ -0,0 +1,578 @@ +"""Eye enhancement tool using MediaPipe Face Mesh + OpenCV. + +Targets the eye region for talking-head footage: +- Under-eye dark circle brightening +- Eye/iris sharpening and brightening +- Subtle under-eye smoothing + +Uses MediaPipe Face Mesh (468 landmarks) to precisely locate eye regions, +then applies targeted OpenCV adjustments. Processes video frame-by-frame. + +Falls back to FFmpeg region-based filters if MediaPipe is not installed. +""" + +from __future__ import annotations + +import json +import time +from pathlib import Path +from typing import Any + +from tools.base_tool import ( + BaseTool, + Determinism, + ExecutionMode, + ResourceProfile, + ToolResult, + ToolStability, + ToolStatus, + ToolTier, +) + +# MediaPipe Face Mesh landmark indices for eye regions. +# These form polygons around each eye area. +# Lower eyelid landmarks (used to define the under-eye region): +LEFT_LOWER_EYELID = [33, 7, 163, 144, 145, 153, 154, 155, 133] +RIGHT_LOWER_EYELID = [263, 249, 390, 373, 374, 380, 381, 382, 362] + +# Full eye contour (used for iris/eye brightening): +LEFT_EYE = [33, 7, 163, 144, 145, 153, 154, 155, 133, 173, 157, 158, 159, 160, 161, 246] +RIGHT_EYE = [263, 249, 390, 373, 374, 380, 381, 382, 362, 398, 384, 385, 386, 387, 388, 466] + +# Iris landmarks (available when refine_landmarks=True, indices 468-477): +LEFT_IRIS = [468, 469, 470, 471, 472] +RIGHT_IRIS = [473, 474, 475, 476, 477] + + +class EyeEnhance(BaseTool): + name = "eye_enhance" + version = "0.1.0" + tier = ToolTier.ENHANCE + capability = "enhancement" + provider = "mediapipe" + stability = ToolStability.EXPERIMENTAL + execution_mode = ExecutionMode.SYNC + determinism = Determinism.DETERMINISTIC + + dependencies = ["cmd:ffmpeg"] + install_instructions = ( + "For best results install MediaPipe and OpenCV:\n" + "pip install mediapipe opencv-python numpy\n\n" + "Without MediaPipe, falls back to FFmpeg eye-region filter (less precise)." + ) + agent_skills = ["ffmpeg"] + + capabilities = [ + "under_eye_brightening", + "dark_circle_removal", + "eye_sharpening", + "eye_brightening", + ] + + input_schema = { + "type": "object", + "required": ["input_path"], + "properties": { + "input_path": {"type": "string"}, + "output_path": {"type": "string"}, + "operations": { + "type": "array", + "items": { + "type": "string", + "enum": ["dark_circles", "brighten_eyes", "sharpen_eyes"], + }, + "default": ["dark_circles", "brighten_eyes"], + "description": "Which enhancements to apply", + }, + "dark_circle_intensity": { + "type": "number", + "default": 0.4, + "minimum": 0.0, + "maximum": 1.0, + "description": "Strength of dark circle removal (0=none, 1=max)", + }, + "eye_brighten_intensity": { + "type": "number", + "default": 0.3, + "minimum": 0.0, + "maximum": 1.0, + "description": "Strength of eye brightening (0=none, 1=max)", + }, + "sharpen_intensity": { + "type": "number", + "default": 0.3, + "minimum": 0.0, + "maximum": 1.0, + "description": "Strength of eye sharpening (0=none, 1=max)", + }, + "codec": {"type": "string", "default": "libx264"}, + "crf": {"type": "integer", "default": 18}, + }, + } + + resource_profile = ResourceProfile( + cpu_cores=4, ram_mb=2048, vram_mb=0, disk_mb=4000, network_required=False + ) + idempotency_key_fields = [ + "input_path", "operations", "dark_circle_intensity", + "eye_brighten_intensity", "sharpen_intensity", + ] + side_effects = ["writes enhanced video to output_path"] + user_visible_verification = [ + "Compare eyes in before/after — enhancement should be subtle and natural", + "Check for artifacts around eye region (halos, color shifts)", + "Verify enhancement doesn't make eyes look unnatural", + ] + + def _has_mediapipe(self) -> bool: + try: + import mediapipe # noqa: F401 + return True + except ImportError: + return False + + def _has_opencv(self) -> bool: + try: + import cv2 # noqa: F401 + import numpy # noqa: F401 + return True + except ImportError: + return False + + def get_status(self) -> ToolStatus: + if self._has_mediapipe() and self._has_opencv(): + return ToolStatus.AVAILABLE + if self._has_opencv(): + return ToolStatus.DEGRADED + return ToolStatus.UNAVAILABLE + + def execute(self, inputs: dict[str, Any]) -> ToolResult: + input_path = Path(inputs["input_path"]) + if not input_path.exists(): + return ToolResult(success=False, error=f"Input not found: {input_path}") + + operations = inputs.get("operations", ["dark_circles", "brighten_eyes"]) + output_path = Path( + inputs.get("output_path", str(input_path.with_stem(f"{input_path.stem}_eye_enhanced"))) + ) + output_path.parent.mkdir(parents=True, exist_ok=True) + + start = time.time() + + if self._has_mediapipe() and self._has_opencv(): + result = self._enhance_mediapipe(input_path, output_path, inputs) + elif self._has_opencv(): + result = self._enhance_opencv_only(input_path, output_path, inputs) + else: + result = self._enhance_ffmpeg_fallback(input_path, output_path, inputs) + + if not result.success: + return result + + elapsed = time.time() - start + result.duration_seconds = round(elapsed, 2) + return result + + def _enhance_mediapipe( + self, input_path: Path, output_path: Path, inputs: dict[str, Any] + ) -> ToolResult: + """Full MediaPipe Face Mesh + OpenCV pipeline for precise eye enhancement.""" + import cv2 + import numpy as np + import mediapipe as mp + + operations = inputs.get("operations", ["dark_circles", "brighten_eyes"]) + dark_intensity = inputs.get("dark_circle_intensity", 0.4) + brighten_intensity = inputs.get("eye_brighten_intensity", 0.3) + sharpen_intensity = inputs.get("sharpen_intensity", 0.3) + codec_fourcc = inputs.get("codec", "libx264") + crf = inputs.get("crf", 18) + + mp_face_mesh = mp.solutions.face_mesh + + cap = cv2.VideoCapture(str(input_path)) + fps = cap.get(cv2.CAP_PROP_FPS) or 30.0 + width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) + height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) + total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) + + # Write to temp file, then mux audio via FFmpeg + temp_video = output_path.parent / f".{output_path.stem}_temp.mp4" + + fourcc = cv2.VideoWriter_fourcc(*"mp4v") + writer = cv2.VideoWriter(str(temp_video), fourcc, fps, (width, height)) + + frames_processed = 0 + frames_enhanced = 0 + + with mp_face_mesh.FaceMesh( + static_image_mode=False, + max_num_faces=1, + refine_landmarks=True, # Enables iris landmarks (468-477) + min_detection_confidence=0.5, + min_tracking_confidence=0.5, + ) as face_mesh: + while cap.isOpened(): + ret, frame = cap.read() + if not ret: + break + + frames_processed += 1 + rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) + results = face_mesh.process(rgb) + + if results.multi_face_landmarks: + landmarks = results.multi_face_landmarks[0] + frame = self._apply_eye_enhancements( + frame, landmarks, width, height, + operations, dark_intensity, brighten_intensity, sharpen_intensity, + ) + frames_enhanced += 1 + + writer.write(frame) + + cap.release() + writer.release() + + # Mux original audio back with FFmpeg + cmd = [ + "ffmpeg", "-y", + "-i", str(temp_video), + "-i", str(input_path), + "-c:v", "libx264", "-crf", str(crf), "-preset", "fast", + "-c:a", "aac", "-b:a", "192k", + "-map", "0:v:0", "-map", "1:a:0?", + "-shortest", + str(output_path), + ] + + try: + self.run_command(cmd, timeout=600) + except Exception as e: + return ToolResult(success=False, error=f"Audio mux failed: {e}") + finally: + if temp_video.exists(): + temp_video.unlink() + + return ToolResult( + success=True, + data={ + "input": str(input_path), + "output": str(output_path), + "method": "mediapipe_face_mesh", + "frames_processed": frames_processed, + "frames_enhanced": frames_enhanced, + "operations": operations, + }, + artifacts=[str(output_path)], + ) + + def _apply_eye_enhancements( + self, + frame, + landmarks, + width: int, + height: int, + operations: list[str], + dark_intensity: float, + brighten_intensity: float, + sharpen_intensity: float, + ): + """Apply eye enhancements to a single frame using detected landmarks.""" + import cv2 + import numpy as np + + result = frame.copy() + + # Convert landmarks to pixel coordinates + def lm_to_px(indices): + points = [] + for idx in indices: + if idx < len(landmarks.landmark): + lm = landmarks.landmark[idx] + points.append((int(lm.x * width), int(lm.y * height))) + return np.array(points, dtype=np.int32) + + for side in ["left", "right"]: + lower_lid = lm_to_px(LEFT_LOWER_EYELID if side == "left" else RIGHT_LOWER_EYELID) + eye_contour = lm_to_px(LEFT_EYE if side == "left" else RIGHT_EYE) + + if len(lower_lid) < 3 or len(eye_contour) < 3: + continue + + # Under-eye region: expand lower eyelid downward + if "dark_circles" in operations: + result = self._remove_dark_circles( + result, lower_lid, width, height, dark_intensity + ) + + if "brighten_eyes" in operations: + iris = lm_to_px(LEFT_IRIS if side == "left" else RIGHT_IRIS) + result = self._brighten_eyes( + result, eye_contour, iris, brighten_intensity + ) + + if "sharpen_eyes" in operations: + result = self._sharpen_eyes( + result, eye_contour, sharpen_intensity + ) + + return result + + def _remove_dark_circles( + self, frame, lower_lid_points, width: int, height: int, intensity: float + ): + """Brighten the under-eye area to reduce dark circles.""" + import cv2 + import numpy as np + + # Create under-eye region by shifting lower eyelid points downward + under_eye = lower_lid_points.copy() + # Shift down by ~15% of face height (approximate eye-to-cheek distance) + shift = max(5, int(height * 0.025)) + under_eye_shifted = under_eye.copy() + under_eye_shifted[:, 1] += shift + + # Create polygon from lower lid + shifted points (forms a band) + polygon = np.vstack([lower_lid_points, under_eye_shifted[::-1]]) + + # Create soft mask + mask = np.zeros(frame.shape[:2], dtype=np.float32) + cv2.fillPoly(mask, [polygon], 1.0) + # Gaussian blur for soft edges + blur_size = max(15, int(width * 0.02)) | 1 # Ensure odd + mask = cv2.GaussianBlur(mask, (blur_size, blur_size), 0) + + # Apply brightening in LAB color space (perceptually uniform) + lab = cv2.cvtColor(frame, cv2.COLOR_BGR2LAB).astype(np.float32) + # Boost L channel (lightness) in the masked region + boost = 25 * intensity # 0-25 range + lab[:, :, 0] += mask * boost + lab[:, :, 0] = np.clip(lab[:, :, 0], 0, 255) + + # Also reduce saturation slightly (dark circles are often purplish) + hsv = cv2.cvtColor(frame, cv2.COLOR_BGR2HSV).astype(np.float32) + hsv[:, :, 1] -= mask * (20 * intensity) + hsv[:, :, 1] = np.clip(hsv[:, :, 1], 0, 255) + + # Blend: use LAB for brightness, HSV for desaturation + brightened = cv2.cvtColor(lab.astype(np.uint8), cv2.COLOR_LAB2BGR) + desaturated = cv2.cvtColor(hsv.astype(np.uint8), cv2.COLOR_HSV2BGR) + + # Combine: weighted blend of both adjustments + mask_3ch = np.stack([mask] * 3, axis=-1) + combined = frame.astype(np.float32) + combined = combined * (1 - mask_3ch * intensity) + \ + (brightened.astype(np.float32) * 0.6 + desaturated.astype(np.float32) * 0.4) * (mask_3ch * intensity) + + return np.clip(combined, 0, 255).astype(np.uint8) + + def _brighten_eyes(self, frame, eye_contour, iris_points, intensity: float): + """Subtly brighten the eye/sclera area.""" + import cv2 + import numpy as np + + if len(eye_contour) < 3: + return frame + + # Create mask from eye contour + mask = np.zeros(frame.shape[:2], dtype=np.float32) + cv2.fillPoly(mask, [eye_contour], 1.0) + blur_size = max(5, int(frame.shape[1] * 0.005)) | 1 + mask = cv2.GaussianBlur(mask, (blur_size, blur_size), 0) + + # Brighten in LAB space + lab = cv2.cvtColor(frame, cv2.COLOR_BGR2LAB).astype(np.float32) + boost = 15 * intensity + lab[:, :, 0] += mask * boost + lab[:, :, 0] = np.clip(lab[:, :, 0], 0, 255) + + brightened = cv2.cvtColor(lab.astype(np.uint8), cv2.COLOR_LAB2BGR) + + mask_3ch = np.stack([mask] * 3, axis=-1) + result = frame.astype(np.float32) * (1 - mask_3ch * intensity) + \ + brightened.astype(np.float32) * (mask_3ch * intensity) + + return np.clip(result, 0, 255).astype(np.uint8) + + def _sharpen_eyes(self, frame, eye_contour, intensity: float): + """Sharpen the eye region for more detail.""" + import cv2 + import numpy as np + + if len(eye_contour) < 3: + return frame + + # Create mask + mask = np.zeros(frame.shape[:2], dtype=np.float32) + cv2.fillPoly(mask, [eye_contour], 1.0) + # Expand slightly for natural blend + kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (5, 5)) + mask = cv2.dilate(mask, kernel, iterations=1) + blur_size = max(5, int(frame.shape[1] * 0.005)) | 1 + mask = cv2.GaussianBlur(mask, (blur_size, blur_size), 0) + + # Unsharp mask for sharpening + blur = cv2.GaussianBlur(frame, (0, 0), 3) + sharpened = cv2.addWeighted(frame, 1.0 + intensity, blur, -intensity, 0) + + mask_3ch = np.stack([mask] * 3, axis=-1) + result = frame.astype(np.float32) * (1 - mask_3ch) + \ + sharpened.astype(np.float32) * mask_3ch + + return np.clip(result, 0, 255).astype(np.uint8) + + def _enhance_opencv_only( + self, input_path: Path, output_path: Path, inputs: dict[str, Any] + ) -> ToolResult: + """Fallback: OpenCV Haar cascade for face detection + generic eye region enhancement.""" + import cv2 + import numpy as np + + operations = inputs.get("operations", ["dark_circles", "brighten_eyes"]) + dark_intensity = inputs.get("dark_circle_intensity", 0.4) + crf = inputs.get("crf", 18) + + face_cascade = cv2.CascadeClassifier( + cv2.data.haarcascades + "haarcascade_frontalface_default.xml" + ) + eye_cascade = cv2.CascadeClassifier( + cv2.data.haarcascades + "haarcascade_eye.xml" + ) + + cap = cv2.VideoCapture(str(input_path)) + fps = cap.get(cv2.CAP_PROP_FPS) or 30.0 + width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) + height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) + + temp_video = output_path.parent / f".{output_path.stem}_temp.mp4" + fourcc = cv2.VideoWriter_fourcc(*"mp4v") + writer = cv2.VideoWriter(str(temp_video), fourcc, fps, (width, height)) + + frames_processed = 0 + frames_enhanced = 0 + + while cap.isOpened(): + ret, frame = cap.read() + if not ret: + break + + frames_processed += 1 + gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY) + faces = face_cascade.detectMultiScale(gray, 1.1, 5, minSize=(60, 60)) + + if len(faces) > 0: + # Use largest face + areas = [w * h for (_, _, w, h) in faces] + fi = areas.index(max(areas)) + fx, fy, fw, fh = faces[fi] + + # Detect eyes within face region + face_roi_gray = gray[fy:fy + fh, fx:fx + fw] + eyes = eye_cascade.detectMultiScale(face_roi_gray, 1.1, 5, minSize=(20, 20)) + + for (ex, ey, ew, eh) in eyes: + # Under-eye region: below the detected eye box + under_y = fy + ey + eh + under_h = max(5, int(eh * 0.4)) + under_x = fx + ex + under_w = ew + + if under_y + under_h <= height and under_x + under_w <= width: + if "dark_circles" in operations: + # Create soft mask for under-eye + mask = np.zeros((height, width), dtype=np.float32) + cv2.ellipse( + mask, + (under_x + under_w // 2, under_y + under_h // 2), + (under_w // 2, under_h // 2), + 0, 0, 360, 1.0, -1, + ) + mask = cv2.GaussianBlur(mask, (15, 15), 0) + + lab = cv2.cvtColor(frame, cv2.COLOR_BGR2LAB).astype(np.float32) + lab[:, :, 0] += mask * (25 * dark_intensity) + lab[:, :, 0] = np.clip(lab[:, :, 0], 0, 255) + frame = cv2.cvtColor(lab.astype(np.uint8), cv2.COLOR_LAB2BGR) + + frames_enhanced += 1 + + writer.write(frame) + + cap.release() + writer.release() + + # Mux audio + cmd = [ + "ffmpeg", "-y", + "-i", str(temp_video), + "-i", str(input_path), + "-c:v", "libx264", "-crf", str(crf), "-preset", "fast", + "-c:a", "aac", "-b:a", "192k", + "-map", "0:v:0", "-map", "1:a:0?", + "-shortest", + str(output_path), + ] + + try: + self.run_command(cmd, timeout=600) + except Exception as e: + return ToolResult(success=False, error=f"Audio mux failed: {e}") + finally: + if temp_video.exists(): + temp_video.unlink() + + return ToolResult( + success=True, + data={ + "input": str(input_path), + "output": str(output_path), + "method": "opencv_haar_cascade", + "frames_processed": frames_processed, + "frames_enhanced": frames_enhanced, + "operations": operations, + }, + artifacts=[str(output_path)], + ) + + def _enhance_ffmpeg_fallback( + self, input_path: Path, output_path: Path, inputs: dict[str, Any] + ) -> ToolResult: + """Last resort: FFmpeg-only. Applies general face-area enhancement (not eye-specific).""" + crf = inputs.get("crf", 18) + intensity = inputs.get("dark_circle_intensity", 0.4) + + # General brightness + contrast lift for the whole frame + # Not eye-specific but better than nothing + brightness = 0.02 * intensity + contrast = 1.0 + (0.05 * intensity) + + cmd = [ + "ffmpeg", "-y", + "-i", str(input_path), + "-vf", f"eq=brightness={brightness}:contrast={contrast}", + "-c:v", "libx264", "-crf", str(crf), "-preset", "fast", + "-c:a", "copy", + str(output_path), + ] + + try: + self.run_command(cmd, timeout=600) + except Exception as e: + return ToolResult(success=False, error=f"FFmpeg fallback failed: {e}") + + return ToolResult( + success=True, + data={ + "input": str(input_path), + "output": str(output_path), + "method": "ffmpeg_global_brightness", + "operations": ["global_brightness_contrast"], + "note": "Install mediapipe + opencv-python for precise eye-region enhancement", + }, + artifacts=[str(output_path)], + ) + + def estimate_runtime(self, inputs: dict[str, Any]) -> float: + """Eye enhancement is roughly 0.5x-1x realtime depending on resolution.""" + return 90.0 diff --git a/tools/subtitle/subtitle_gen.py b/tools/subtitle/subtitle_gen.py index 6344352b..ac8472b6 100644 --- a/tools/subtitle/subtitle_gen.py +++ b/tools/subtitle/subtitle_gen.py @@ -60,6 +60,15 @@ class SubtitleGen(BaseTool): "enum": ["none", "word_by_word", "karaoke"], "default": "none", }, + "corrections": { + "type": "object", + "description": ( + "Dictionary of word corrections for common ASR misrecognitions. " + "Keys are the wrong word (case-insensitive), values are the " + "correct replacement. Applied before generating subtitles. " + "Example: {\"cloud\": \"Claude\", \"co-pilot\": \"Copilot\"}." + ), + }, }, } @@ -77,9 +86,14 @@ class SubtitleGen(BaseTool): max_chars = inputs.get("max_chars_per_line", 42) highlight_style = inputs.get("highlight_style", "none") output_path = inputs.get("output_path") + corrections = inputs.get("corrections") start = time.time() + # Apply word corrections if provided + if corrections: + segments = self._apply_corrections(segments, corrections) + # Build cues from word-level timestamps cues = self._build_cues(segments, max_words, max_chars) @@ -114,6 +128,43 @@ class SubtitleGen(BaseTool): duration_seconds=round(elapsed, 2), ) + @staticmethod + def _apply_corrections( + segments: list[dict], corrections: dict[str, str] + ) -> list[dict]: + """Apply word-level corrections to transcript segments. + + Handles case-insensitive matching and preserves punctuation. + """ + import copy + + corr = {k.lower(): v for k, v in corrections.items()} + result = copy.deepcopy(segments) + + for seg in result: + words = seg.get("words", []) + for w in words: + raw = w.get("word", "").strip() + # Strip punctuation for lookup, preserve it + stripped = raw.lower().rstrip(".,!?;:'\"") + if stripped in corr: + trailing = raw[len(stripped):] + w["word"] = corr[stripped] + trailing + # Also fix segment-level text + if "text" in seg and words: + seg["text"] = " ".join(w["word"] for w in words) + elif "text" in seg: + for wrong, right in corr.items(): + import re as _re + seg["text"] = _re.sub( + r"\b" + _re.escape(wrong) + r"\b", + right, + seg["text"], + flags=_re.IGNORECASE, + ) + + return result + def _build_cues( self, segments: list[dict], max_words: int, max_chars: int ) -> list[dict]: diff --git a/tools/video/auto_reframe.py b/tools/video/auto_reframe.py new file mode 100644 index 00000000..d054046c --- /dev/null +++ b/tools/video/auto_reframe.py @@ -0,0 +1,537 @@ +"""Auto-reframe tool for aspect ratio conversion with face tracking. + +Converts video between aspect ratios (e.g. 16:9 → 9:16 for Instagram Reels) +while keeping the speaker's face centered in frame. Uses face_tracker data +for smooth, content-aware cropping. + +Primary use: converting talking-head footage shot in landscape to vertical +format for social media (TikTok, Reels, Shorts). + +Approach: MediaPipe/OpenCV face detection → smoothed bounding box trajectory +→ FFmpeg crop filter. No GPU required. +""" + +from __future__ import annotations + +import json +import time +from pathlib import Path +from typing import Any + +from tools.base_tool import ( + BaseTool, + Determinism, + ExecutionMode, + ResourceProfile, + RetryPolicy, + ResumeSupport, + ToolResult, + ToolStability, + ToolTier, +) + + +# Common target aspect ratios +ASPECT_PRESETS = { + "portrait": (9, 16), # Instagram Reels, TikTok, YouTube Shorts + "square": (1, 1), # Instagram Feed + "landscape": (16, 9), # YouTube, LinkedIn + "cinematic": (21, 9), # Ultra-wide + "vertical_4_5": (4, 5), # Instagram portrait post +} + + +class AutoReframe(BaseTool): + name = "auto_reframe" + version = "0.1.0" + tier = ToolTier.CORE + capability = "video_post" + provider = "ffmpeg" + stability = ToolStability.EXPERIMENTAL + execution_mode = ExecutionMode.SYNC + determinism = Determinism.DETERMINISTIC + + dependencies = ["cmd:ffmpeg"] + install_instructions = ( + "FFmpeg is required. For face-tracked reframing, also install:\n" + "pip install mediapipe opencv-python\n\n" + "Without MediaPipe/OpenCV, falls back to center-crop." + ) + agent_skills = ["ffmpeg"] + + capabilities = [ + "aspect_ratio_conversion", + "face_tracked_crop", + "smart_reframe", + "center_crop", + ] + + input_schema = { + "type": "object", + "required": ["input_path"], + "properties": { + "input_path": {"type": "string"}, + "output_path": {"type": "string"}, + "target_aspect": { + "type": "string", + "enum": list(ASPECT_PRESETS.keys()), + "default": "portrait", + "description": "Target aspect ratio preset", + }, + "target_width": { + "type": "integer", + "description": "Explicit target width (overrides preset)", + }, + "target_height": { + "type": "integer", + "description": "Explicit target height (overrides preset)", + }, + "face_tracking_json": { + "type": "string", + "description": "Path to pre-computed face_tracker JSON. If omitted, runs face detection internally.", + }, + "smoothing_window": { + "type": "integer", + "default": 15, + "minimum": 1, + "description": "Number of frames for position smoothing (higher = smoother pan, lower = more responsive)", + }, + "face_padding": { + "type": "number", + "default": 0.4, + "minimum": 0.0, + "maximum": 1.0, + "description": "Extra space around face as fraction of face size (0.4 = 40% padding)", + }, + "sample_fps": { + "type": "number", + "default": 5, + "description": "Face detection sample rate (only used if no face_tracking_json)", + }, + "codec": {"type": "string", "default": "libx264"}, + "crf": {"type": "integer", "default": 18}, + }, + } + + resource_profile = ResourceProfile( + cpu_cores=4, ram_mb=2048, vram_mb=0, disk_mb=4000, network_required=False + ) + retry_policy = RetryPolicy(max_retries=1, retryable_errors=["FFmpeg error"]) + resume_support = ResumeSupport.FROM_START + idempotency_key_fields = [ + "input_path", "target_aspect", "target_width", "target_height", + "smoothing_window", "face_padding", + ] + side_effects = ["writes reframed video to output_path"] + user_visible_verification = [ + "Play reframed output — verify face stays centered and framing is smooth", + "Check that no important content is cropped out", + ] + + def execute(self, inputs: dict[str, Any]) -> ToolResult: + input_path = Path(inputs["input_path"]) + if not input_path.exists(): + return ToolResult(success=False, error=f"Input not found: {input_path}") + + start = time.time() + + # Get source video dimensions + src_w, src_h, src_fps = self._get_video_info(input_path) + if src_w == 0 or src_h == 0: + return ToolResult(success=False, error="Could not read video dimensions") + + # Determine target crop dimensions (in source pixel space) + target_w, target_h = self._compute_crop_size(inputs, src_w, src_h) + + # If source already matches target aspect, no crop needed + if target_w == src_w and target_h == src_h: + return ToolResult( + success=True, + data={"message": "Source already matches target aspect ratio", "output": str(input_path)}, + artifacts=[str(input_path)], + ) + + # Get face tracking data + face_data = self._get_face_data(inputs, input_path, src_fps) + + # Compute per-frame crop positions + if face_data and len(face_data) > 0: + crop_x, crop_y = self._compute_face_tracked_crop( + face_data, src_w, src_h, target_w, target_h, + src_fps, + inputs.get("smoothing_window", 15), + inputs.get("face_padding", 0.4), + ) + method = "face_tracked" + else: + # Fallback: center crop + crop_x = (src_w - target_w) // 2 + crop_y = (src_h - target_h) // 2 + method = "center_crop" + + # Determine output resolution + out_w, out_h = self._compute_output_resolution(inputs, target_w, target_h, src_w, src_h) + + # Build output path + aspect_name = inputs.get("target_aspect", "portrait") + output_path = Path( + inputs.get("output_path", str(input_path.with_stem(f"{input_path.stem}_{aspect_name}"))) + ) + output_path.parent.mkdir(parents=True, exist_ok=True) + + # Render via FFmpeg + codec = inputs.get("codec", "libx264") + crf = inputs.get("crf", 18) + + if method == "face_tracked" and isinstance(crop_x, list): + # Dynamic crop: write crop coordinates to a file and use sendcmd + result = self._render_dynamic_crop( + input_path, output_path, crop_x, crop_y, + target_w, target_h, out_w, out_h, + src_fps, codec, crf, + ) + else: + # Static crop + result = self._render_static_crop( + input_path, output_path, + crop_x, crop_y, target_w, target_h, + out_w, out_h, codec, crf, + ) + + if not result.success: + return result + + elapsed = time.time() - start + + return ToolResult( + success=True, + data={ + "input": str(input_path), + "output": str(output_path), + "source_resolution": f"{src_w}x{src_h}", + "crop_resolution": f"{target_w}x{target_h}", + "output_resolution": f"{out_w}x{out_h}", + "method": method, + "target_aspect": inputs.get("target_aspect", "portrait"), + }, + artifacts=[str(output_path)], + duration_seconds=round(elapsed, 2), + ) + + def _get_video_info(self, path: Path) -> tuple[int, int, float]: + """Get video width, height, fps via ffprobe.""" + cmd = [ + "ffprobe", "-v", "quiet", + "-select_streams", "v:0", + "-show_entries", "stream=width,height,r_frame_rate", + "-of", "json", str(path), + ] + try: + result = self.run_command(cmd) + data = json.loads(result.stdout) + stream = data["streams"][0] + w = int(stream["width"]) + h = int(stream["height"]) + # Parse r_frame_rate (e.g. "30000/1001") + fps_parts = stream["r_frame_rate"].split("/") + fps = float(fps_parts[0]) / float(fps_parts[1]) if len(fps_parts) == 2 else float(fps_parts[0]) + return w, h, fps + except Exception: + return 0, 0, 30.0 + + def _compute_crop_size( + self, inputs: dict[str, Any], src_w: int, src_h: int + ) -> tuple[int, int]: + """Compute crop dimensions in source pixel space that match the target aspect ratio.""" + if "target_width" in inputs and "target_height" in inputs: + # Explicit dimensions — compute crop in source space matching this ratio + tw, th = inputs["target_width"], inputs["target_height"] + else: + aspect_name = inputs.get("target_aspect", "portrait") + tw, th = ASPECT_PRESETS.get(aspect_name, (9, 16)) + + target_ratio = tw / th + src_ratio = src_w / src_h + + if target_ratio > src_ratio: + # Target is wider — crop height + crop_w = src_w + crop_h = int(src_w / target_ratio) + else: + # Target is taller/narrower — crop width + crop_h = src_h + crop_w = int(src_h * target_ratio) + + # Ensure even dimensions (required by most codecs) + crop_w = crop_w - (crop_w % 2) + crop_h = crop_h - (crop_h % 2) + + return crop_w, crop_h + + def _compute_output_resolution( + self, inputs: dict[str, Any], + crop_w: int, crop_h: int, + src_w: int, src_h: int, + ) -> tuple[int, int]: + """Determine final output resolution. Scales to standard sizes.""" + if "target_width" in inputs and "target_height" in inputs: + out_w = inputs["target_width"] + out_h = inputs["target_height"] + else: + aspect_name = inputs.get("target_aspect", "portrait") + if aspect_name == "portrait": + out_w, out_h = 1080, 1920 + elif aspect_name == "square": + out_w, out_h = 1080, 1080 + elif aspect_name == "landscape": + out_w, out_h = 1920, 1080 + elif aspect_name == "cinematic": + out_w, out_h = 2560, 1080 + elif aspect_name == "vertical_4_5": + out_w, out_h = 1080, 1350 + else: + out_w, out_h = crop_w, crop_h + + # Ensure even + out_w = out_w - (out_w % 2) + out_h = out_h - (out_h % 2) + return out_w, out_h + + def _get_face_data( + self, inputs: dict[str, Any], input_path: Path, src_fps: float + ) -> list[dict]: + """Get face tracking data — from pre-computed JSON or by running detection.""" + # Check for pre-computed tracking data + tracking_json = inputs.get("face_tracking_json") + if tracking_json: + p = Path(tracking_json) + if p.exists(): + data = json.loads(p.read_text(encoding="utf-8")) + return data.get("faces", []) + + # Try to run face_tracker internally + try: + from tools.analysis.face_tracker import FaceTracker + tracker = FaceTracker() + if tracker.get_status().name == "UNAVAILABLE": + return [] + sample_fps = inputs.get("sample_fps", 5) + result = tracker.execute({ + "input_path": str(input_path), + "sample_fps": sample_fps, + }) + if result.success and result.data: + # Read the generated JSON + output_file = result.data.get("output") + if output_file: + data = json.loads(Path(output_file).read_text(encoding="utf-8")) + return data.get("faces", []) + except Exception: + pass + + return [] + + def _compute_face_tracked_crop( + self, + faces: list[dict], + src_w: int, src_h: int, + crop_w: int, crop_h: int, + fps: float, + smoothing_window: int, + face_padding: float, + ) -> tuple[list[int], list[int]]: + """Compute smoothed crop positions from face tracking data. + + Returns a single (x, y) if face positions are stable enough, + or lists of per-frame positions for dynamic cropping. + """ + if not faces: + cx = (src_w - crop_w) // 2 + cy = (src_h - crop_h) // 2 + return cx, cy + + # Convert relative bbox centers to pixel positions + face_centers_x = [] + face_centers_y = [] + face_timestamps = [] + + for f in faces: + bbox = f["bbox"] + # Center of face in pixel space + center_x = (bbox["x"] + bbox["width"] / 2) * src_w + center_y = (bbox["y"] + bbox["height"] / 2) * src_h + face_centers_x.append(center_x) + face_centers_y.append(center_y) + face_timestamps.append(f["timestamp_seconds"]) + + # Check if face position is stable (talking head usually is) + x_range = max(face_centers_x) - min(face_centers_x) + y_range = max(face_centers_y) - min(face_centers_y) + + # If face barely moves (<10% of frame), use a single static crop + if x_range < src_w * 0.10 and y_range < src_h * 0.10: + avg_x = sum(face_centers_x) / len(face_centers_x) + avg_y = sum(face_centers_y) / len(face_centers_y) + + # Position crop window centered on face, with bias toward upper third + crop_x = int(avg_x - crop_w / 2) + crop_y = int(avg_y - crop_h * 0.35) # Face in upper 35% of frame + + # Clamp to frame bounds + crop_x = max(0, min(crop_x, src_w - crop_w)) + crop_y = max(0, min(crop_y, src_h - crop_h)) + + return crop_x, crop_y + + # Dynamic crop: smooth the trajectory + smoothed_x = self._smooth_positions(face_centers_x, smoothing_window) + smoothed_y = self._smooth_positions(face_centers_y, smoothing_window) + + # Convert to crop positions (top-left corner), clamped + crop_xs = [] + crop_ys = [] + for sx, sy in zip(smoothed_x, smoothed_y): + cx = int(sx - crop_w / 2) + cy = int(sy - crop_h * 0.35) + cx = max(0, min(cx, src_w - crop_w)) + cy = max(0, min(cy, src_h - crop_h)) + crop_xs.append(cx) + crop_ys.append(cy) + + return crop_xs, crop_ys + + def _smooth_positions(self, values: list[float], window: int) -> list[float]: + """Simple moving average smoothing.""" + smoothed = [] + for i in range(len(values)): + start = max(0, i - window // 2) + end = min(len(values), i + window // 2 + 1) + smoothed.append(sum(values[start:end]) / (end - start)) + return smoothed + + def _render_static_crop( + self, + input_path: Path, output_path: Path, + crop_x: int, crop_y: int, + crop_w: int, crop_h: int, + out_w: int, out_h: int, + codec: str, crf: int, + ) -> ToolResult: + """Render with a static crop position.""" + vf = f"crop={crop_w}:{crop_h}:{crop_x}:{crop_y},scale={out_w}:{out_h}" + + cmd = [ + "ffmpeg", "-y", + "-i", str(input_path), + "-vf", vf, + "-c:v", codec, "-crf", str(crf), "-preset", "fast", + "-c:a", "aac", "-b:a", "192k", + str(output_path), + ] + + try: + self.run_command(cmd, timeout=600) + except Exception as e: + return ToolResult(success=False, error=f"FFmpeg render failed: {e}") + + return ToolResult(success=True) + + def _render_dynamic_crop( + self, + input_path: Path, output_path: Path, + crop_xs: list[int], crop_ys: list[int], + crop_w: int, crop_h: int, + out_w: int, out_h: int, + fps: float, + codec: str, crf: int, + ) -> ToolResult: + """Render with dynamic crop positions that follow the face. + + Uses FFmpeg's sendcmd filter to update crop position over time. + For simplicity and reliability, we interpolate between key positions + using FFmpeg expression-based crop. + """ + # Build a piecewise-linear x(t) and y(t) using FFmpeg expressions + # We'll sample at the face tracking rate and interpolate between points + if not crop_xs: + return ToolResult(success=False, error="No crop positions computed") + + # If very few data points, fall back to static using average + if len(crop_xs) < 3: + avg_x = int(sum(crop_xs) / len(crop_xs)) + avg_y = int(sum(crop_ys) / len(crop_ys)) + return self._render_static_crop( + input_path, output_path, avg_x, avg_y, + crop_w, crop_h, out_w, out_h, codec, crf, + ) + + # Build sendcmd script for crop filter position updates + # Each command sets the crop x,y at the corresponding timestamp + temp_dir = output_path.parent / ".reframe_tmp" + temp_dir.mkdir(parents=True, exist_ok=True) + sendcmd_path = temp_dir / "crop_commands.txt" + + # Approximate timestamps from the face tracking sample rate + # The face data was sampled at sample_fps intervals + sample_interval = 1.0 / (fps / max(1, int(fps / 5))) # Approximate + + lines = [] + for i, (cx, cy) in enumerate(zip(crop_xs, crop_ys)): + ts = i * sample_interval + lines.append(f"{ts:.3f} [enter] crop x {cx};") + lines.append(f"{ts:.3f} [enter] crop y {cy};") + + sendcmd_path.write_text("\n".join(lines), encoding="utf-8") + + # Use crop with sendcmd for dynamic positioning + vf = ( + f"sendcmd=f='{str(sendcmd_path).replace(chr(92), '/')}':flags=enter," + f"crop={crop_w}:{crop_h}:{crop_xs[0]}:{crop_ys[0]}," + f"scale={out_w}:{out_h}" + ) + + cmd = [ + "ffmpeg", "-y", + "-i", str(input_path), + "-vf", vf, + "-c:v", codec, "-crf", str(crf), "-preset", "fast", + "-c:a", "aac", "-b:a", "192k", + str(output_path), + ] + + try: + self.run_command(cmd, timeout=600) + except Exception: + # sendcmd can be finicky — fall back to static crop with average position + avg_x = int(sum(crop_xs) / len(crop_xs)) + avg_y = int(sum(crop_ys) / len(crop_ys)) + result = self._render_static_crop( + input_path, output_path, avg_x, avg_y, + crop_w, crop_h, out_w, out_h, codec, crf, + ) + if result.success: + result.data = result.data or {} + result.data["fallback"] = "sendcmd failed, used static average crop" + return result + finally: + # Clean up + if sendcmd_path.exists(): + sendcmd_path.unlink() + try: + temp_dir.rmdir() + except OSError: + pass + + return ToolResult(success=True) + + def estimate_runtime(self, inputs: dict[str, Any]) -> float: + """Estimate runtime in seconds. Roughly 1x realtime for face tracking + render.""" + return 60.0 # Conservative default + + @staticmethod + def list_presets() -> dict[str, str]: + """Return available aspect ratio presets.""" + return { + name: f"{w}:{h}" + for name, (w, h) in ASPECT_PRESETS.items() + } diff --git a/tools/video/remotion_caption_burn.py b/tools/video/remotion_caption_burn.py new file mode 100644 index 00000000..6824e69d --- /dev/null +++ b/tools/video/remotion_caption_burn.py @@ -0,0 +1,439 @@ +"""Remotion caption burn tool. + +Renders animated word-by-word captions onto a talking-head video using +the Remotion CaptionOverlay component. Falls back to FFmpeg subtitle +burning if Remotion is not available. + +The tool: +1. Converts word-level transcript segments to Remotion WordCaption JSON +2. Writes a props file for the TalkingHead composition +3. Renders via ``npx remotion render`` +4. Returns the captioned video path + +Fallback: if Remotion is unavailable, burns subtitles at the bottom of +the frame using FFmpeg's ``subtitles`` filter with bold styling. +""" + +from __future__ import annotations + +import json +import math +import re +import shutil +import time +from pathlib import Path +from typing import Any + +from tools.base_tool import ( + BaseTool, + Determinism, + ExecutionMode, + ResourceProfile, + ToolResult, + ToolStability, + ToolTier, +) + + +class RemotionCaptionBurn(BaseTool): + name = "remotion_caption_burn" + version = "0.1.0" + tier = ToolTier.CORE + capability = "subtitle" + provider = "remotion" + stability = ToolStability.EXPERIMENTAL + execution_mode = ExecutionMode.SYNC + determinism = Determinism.DETERMINISTIC + + dependencies = ["cmd:ffmpeg", "cmd:ffprobe"] + install_instructions = ( + "Remotion (optional, preferred): npm install in remotion-composer/\n" + "FFmpeg (required for fallback): https://ffmpeg.org/download.html" + ) + agent_skills = ["remotion-best-practices", "ffmpeg"] + + capabilities = [ + "burn_remotion_captions", + "burn_ffmpeg_captions_fallback", + ] + + input_schema = { + "type": "object", + "required": ["input_path", "output_path"], + "properties": { + "input_path": { + "type": "string", + "description": "Path to the input video (enhanced talking-head footage).", + }, + "output_path": { + "type": "string", + "description": "Path for the output video with captions burned in.", + }, + "segments": { + "type": "array", + "description": ( + "Word-level transcript segments from transcriber tool. " + "Each segment has 'words' array with {word, start, end}." + ), + }, + "srt_path": { + "type": "string", + "description": ( + "Path to an SRT file. Used as an alternative to segments. " + "If both provided, segments take priority." + ), + }, + "words_per_page": { + "type": "integer", + "default": 4, + "description": "Words shown at once in the caption overlay.", + }, + "font_size": { + "type": "integer", + "default": 52, + "description": "Font size for captions.", + }, + "highlight_color": { + "type": "string", + "default": "#22D3EE", + "description": "Highlight color for the active word (hex).", + }, + "corrections": { + "type": "object", + "description": ( + "Dictionary of word corrections for common misrecognitions. " + "Keys are the wrong word (case-insensitive), values are the " + "correct replacement. Example: {\"cloud\": \"Claude\"}." + ), + }, + "force_ffmpeg": { + "type": "boolean", + "default": False, + "description": "Force FFmpeg fallback even if Remotion is available.", + }, + }, + } + + resource_profile = ResourceProfile(cpu_cores=4, ram_mb=2048, vram_mb=0, disk_mb=500) + idempotency_key_fields = ["input_path", "segments", "srt_path"] + side_effects = ["writes captioned video to output_path"] + user_visible_verification = [ + "Play the output video and verify captions appear at the bottom of the frame", + "Check that the active word is highlighted in the specified color", + "Verify face is not occluded by caption text", + ] + + # ------------------------------------------------------------------ # + # Remotion detection + # ------------------------------------------------------------------ # + + def _find_remotion_root(self) -> Path | None: + """Find the remotion-composer directory relative to the repo.""" + candidates = [ + Path.cwd() / "remotion-composer", + Path(__file__).resolve().parent.parent.parent / "remotion-composer", + ] + for p in candidates: + if ( + p.is_dir() + and (p / "package.json").exists() + and (p / "node_modules").is_dir() + ): + return p + return None + + def _remotion_available(self) -> bool: + return ( + shutil.which("npx") is not None + and self._find_remotion_root() is not None + ) + + # ------------------------------------------------------------------ # + # Word caption conversion + # ------------------------------------------------------------------ # + + def _segments_to_word_captions( + self, segments: list[dict], corrections: dict[str, str] | None = None + ) -> list[dict]: + """Convert transcriber segments to [{word, startMs, endMs}, ...].""" + captions: list[dict] = [] + corr = {k.lower(): v for k, v in (corrections or {}).items()} + + for seg in segments: + words = seg.get("words", []) + if words: + for w in words: + raw = w["word"].strip() + fixed = corr.get(raw.lower().strip(".,!?;:"), raw) + # Preserve trailing punctuation from original + trailing = "" + if raw and raw[-1] in ".,!?;:": + trailing = raw[-1] + if fixed != raw and not fixed.endswith(trailing): + fixed = fixed + trailing + captions.append({ + "word": fixed, + "startMs": int(w["start"] * 1000), + "endMs": int(w["end"] * 1000), + }) + elif "text" in seg: + text_words = seg["text"].strip().split() + dur = seg["end"] - seg["start"] + per_word = dur / max(len(text_words), 1) + for i, tw in enumerate(text_words): + fixed = corr.get(tw.lower().strip(".,!?;:"), tw) + captions.append({ + "word": fixed, + "startMs": int((seg["start"] + i * per_word) * 1000), + "endMs": int((seg["start"] + (i + 1) * per_word) * 1000), + }) + return captions + + def _srt_to_word_captions( + self, srt_path: str, corrections: dict[str, str] | None = None + ) -> list[dict]: + """Parse SRT file into word captions.""" + content = Path(srt_path).read_text(encoding="utf-8") + blocks = re.split(r"\n\n+", content.strip()) + corr = {k.lower(): v for k, v in (corrections or {}).items()} + captions: list[dict] = [] + + for block in blocks: + lines = block.strip().split("\n") + if len(lines) < 3: + continue + m = re.match( + r"(\d{2}):(\d{2}):(\d{2}),(\d{3})\s*-->\s*" + r"(\d{2}):(\d{2}):(\d{2}),(\d{3})", + lines[1], + ) + if not m: + continue + start_ms = ( + int(m.group(1)) * 3600000 + + int(m.group(2)) * 60000 + + int(m.group(3)) * 1000 + + int(m.group(4)) + ) + end_ms = ( + int(m.group(5)) * 3600000 + + int(m.group(6)) * 60000 + + int(m.group(7)) * 1000 + + int(m.group(8)) + ) + text = " ".join(lines[2:]).strip() + words = text.split() + per_word = (end_ms - start_ms) / max(len(words), 1) + for i, w in enumerate(words): + fixed = corr.get(w.lower().strip(".,!?;:"), w) + captions.append({ + "word": fixed, + "startMs": int(start_ms + i * per_word), + "endMs": int(start_ms + (i + 1) * per_word), + }) + return captions + + # ------------------------------------------------------------------ # + # Remotion render + # ------------------------------------------------------------------ # + + def _render_remotion( + self, + input_path: str, + output_path: str, + captions: list[dict], + words_per_page: int, + font_size: int, + highlight_color: str, + ) -> ToolResult: + root = self._find_remotion_root() + if root is None: + return ToolResult(success=False, error="Remotion root not found") + + # Get video duration in frames + dur_cmd = [ + "ffprobe", "-v", "error", + "-show_entries", "format=duration", + "-of", "csv=p=0", + input_path, + ] + dur_out = self.run_command(dur_cmd, capture=True) + duration_s = float(dur_out.strip().split("\n")[0]) + total_frames = math.ceil(duration_s * 30) + + # Copy video to Remotion public folder + pub_dir = root / "public" / "talking-head" + pub_dir.mkdir(parents=True, exist_ok=True) + video_filename = Path(input_path).name + dest_video = pub_dir / video_filename + shutil.copy2(input_path, dest_video) + + # Build props JSON + props = { + "videoSrc": f"public/talking-head/{video_filename}", + "captions": captions, + "wordsPerPage": words_per_page, + "fontSize": font_size, + "highlightColor": highlight_color, + } + props_dir = root / "public" / "demo-props" + props_dir.mkdir(parents=True, exist_ok=True) + props_file = props_dir / f"caption-burn-{Path(input_path).stem}.json" + props_file.write_text(json.dumps(props, indent=2), encoding="utf-8") + + # Render + render_cmd = [ + "npx", "remotion", "render", + "src/index.tsx", "TalkingHead", + f"--props={props_file.relative_to(root)}", + "--width=1080", "--height=1920", "--fps=30", + f"--frames=0-{total_frames - 1}", + "--codec=h264", "--crf=18", + str(Path(output_path).resolve()), + ] + self.run_command(render_cmd, cwd=str(root)) + + if not Path(output_path).exists(): + return ToolResult(success=False, error="Remotion render produced no output") + + return ToolResult( + success=True, + data={ + "method": "remotion", + "output": output_path, + "duration_seconds": round(duration_s, 2), + "total_frames": total_frames, + "caption_count": len(captions), + "words_per_page": words_per_page, + }, + artifacts=[output_path], + ) + + # ------------------------------------------------------------------ # + # FFmpeg fallback + # ------------------------------------------------------------------ # + + def _render_ffmpeg( + self, + input_path: str, + output_path: str, + captions: list[dict], + ) -> ToolResult: + """Fall back to FFmpeg subtitle burning at bottom of frame.""" + # Generate temporary SRT from word captions + tmp_srt = Path(output_path).parent / f"_tmp_captions_{int(time.time())}.srt" + tmp_srt.parent.mkdir(parents=True, exist_ok=True) + + srt_lines = [] + idx = 1 + # Group into pages of ~4 words + page_size = 4 + for i in range(0, len(captions), page_size): + page = captions[i : i + page_size] + text = " ".join(c["word"] for c in page) + start = page[0]["startMs"] + end = page[-1]["endMs"] + srt_lines.append(str(idx)) + srt_lines.append( + f"{self._ms_to_srt(start)} --> {self._ms_to_srt(end)}" + ) + srt_lines.append(text) + srt_lines.append("") + idx += 1 + + tmp_srt.write_text("\n".join(srt_lines), encoding="utf-8") + + # Escape path for FFmpeg subtitles filter (Windows colon issue) + srt_escaped = str(tmp_srt).replace("\\", "/").replace(":", "\\:") + + cmd = [ + "ffmpeg", "-y", + "-i", input_path, + "-vf", ( + f"subtitles='{srt_escaped}'" + ":force_style='FontName=Segoe UI,FontSize=24,Bold=1," + "PrimaryColour=&H00FFFFFF,OutlineColour=&H00000000," + "Outline=3,Shadow=2,Alignment=2,MarginV=100'" + ), + "-c:v", "libx264", "-preset", "fast", "-crf", "18", + "-pix_fmt", "yuv420p", + "-c:a", "copy", + output_path, + ] + self.run_command(cmd) + + # Clean up temp SRT + try: + tmp_srt.unlink() + except OSError: + pass + + if not Path(output_path).exists(): + return ToolResult(success=False, error="FFmpeg subtitle burn produced no output") + + return ToolResult( + success=True, + data={ + "method": "ffmpeg_fallback", + "output": output_path, + "caption_count": len(captions), + "note": "Used FFmpeg fallback. Install Remotion for animated captions.", + }, + artifacts=[output_path], + ) + + @staticmethod + def _ms_to_srt(ms: int) -> str: + h = ms // 3600000 + m = (ms % 3600000) // 60000 + s = (ms % 60000) // 1000 + rem = ms % 1000 + return f"{h:02d}:{m:02d}:{s:02d},{rem:03d}" + + # ------------------------------------------------------------------ # + # Main execute + # ------------------------------------------------------------------ # + + def execute(self, inputs: dict[str, Any]) -> ToolResult: + input_path = inputs["input_path"] + output_path = inputs["output_path"] + corrections = inputs.get("corrections") + force_ffmpeg = inputs.get("force_ffmpeg", False) + words_per_page = inputs.get("words_per_page", 4) + font_size = inputs.get("font_size", 52) + highlight_color = inputs.get("highlight_color", "#22D3EE") + + if not Path(input_path).exists(): + return ToolResult(success=False, error=f"Input video not found: {input_path}") + + Path(output_path).parent.mkdir(parents=True, exist_ok=True) + start = time.time() + + # Build word captions from segments or SRT + segments = inputs.get("segments") + srt_path = inputs.get("srt_path") + + if segments: + captions = self._segments_to_word_captions(segments, corrections) + elif srt_path: + captions = self._srt_to_word_captions(srt_path, corrections) + else: + return ToolResult( + success=False, + error="Provide either 'segments' (from transcriber) or 'srt_path'.", + ) + + if not captions: + return ToolResult(success=False, error="No caption words extracted.") + + # Choose render method + if not force_ffmpeg and self._remotion_available(): + result = self._render_remotion( + input_path, output_path, captions, + words_per_page, font_size, highlight_color, + ) + else: + result = self._render_ffmpeg(input_path, output_path, captions) + + result.duration_seconds = round(time.time() - start, 2) + return result diff --git a/tools/video/showcase_card.py b/tools/video/showcase_card.py new file mode 100644 index 00000000..ec0f3f4d --- /dev/null +++ b/tools/video/showcase_card.py @@ -0,0 +1,226 @@ +"""Showcase card tool wrapping FFmpeg. + +Creates a presentation-ready 9:16 card from a source video: letterboxes +the content, adds a bold title at the top, a subtitle description at the +bottom, and a dark background. Designed for Instagram Reels / TikTok +showcase segments. +""" + +from __future__ import annotations + +import time +from pathlib import Path +from typing import Any + +from tools.base_tool import ( + BaseTool, + Determinism, + ExecutionMode, + ResourceProfile, + ToolResult, + ToolStability, + ToolTier, +) + + +class ShowcaseCard(BaseTool): + name = "showcase_card" + version = "0.1.0" + tier = ToolTier.CORE + capability = "video_post" + provider = "ffmpeg" + stability = ToolStability.EXPERIMENTAL + execution_mode = ExecutionMode.SYNC + determinism = Determinism.DETERMINISTIC + + dependencies = ["cmd:ffmpeg", "cmd:ffprobe"] + install_instructions = "Install FFmpeg: https://ffmpeg.org/download.html" + agent_skills = ["ffmpeg", "video_toolkit"] + + capabilities = ["create_showcase_card"] + + input_schema = { + "type": "object", + "required": ["input_path", "output_path", "title"], + "properties": { + "input_path": { + "type": "string", + "description": "Path to the source video.", + }, + "output_path": { + "type": "string", + "description": "Path for the output showcase card video.", + }, + "title": { + "type": "string", + "description": "Bold title text displayed at the top of the card.", + }, + "subtitle": { + "type": "string", + "default": "", + "description": "Subtitle text displayed at the bottom of the card.", + }, + "output_width": { + "type": "integer", + "default": 1080, + "description": "Output width in pixels.", + }, + "output_height": { + "type": "integer", + "default": 1920, + "description": "Output height in pixels.", + }, + "background_color": { + "type": "string", + "default": "0x0A0F1A", + "description": "Background color in hex (FFmpeg format, e.g. 0x0A0F1A).", + }, + "title_font": { + "type": "string", + "default": "segoeuib.ttf", + "description": "Font file for the title. Uses system font lookup.", + }, + "title_font_size": { + "type": "integer", + "default": 52, + "description": "Font size for the title.", + }, + "subtitle_font_size": { + "type": "integer", + "default": 28, + "description": "Font size for the subtitle.", + }, + "title_color": { + "type": "string", + "default": "white", + "description": "Title text color.", + }, + "watermark": { + "type": "string", + "default": "", + "description": "Optional watermark text overlaid on the video (e.g. brand name).", + }, + }, + } + + resource_profile = ResourceProfile(cpu_cores=2, ram_mb=1024, vram_mb=0, disk_mb=500) + idempotency_key_fields = ["input_path", "title", "subtitle"] + side_effects = ["writes showcase card video to output_path"] + user_visible_verification = [ + "Play output and verify title, subtitle, and video are positioned correctly", + "Verify the video content is fully visible (not cropped)", + ] + + def execute(self, inputs: dict[str, Any]) -> ToolResult: + input_path = inputs["input_path"] + output_path = inputs["output_path"] + title = inputs["title"] + subtitle = inputs.get("subtitle", "") + out_w = inputs.get("output_width", 1080) + out_h = inputs.get("output_height", 1920) + bg_color = inputs.get("background_color", "0x0A0F1A") + title_font = inputs.get("title_font", "segoeuib.ttf") + title_font_size = inputs.get("title_font_size", 52) + subtitle_font_size = inputs.get("subtitle_font_size", 28) + title_color = inputs.get("title_color", "white") + watermark = inputs.get("watermark", "") + + if not Path(input_path).exists(): + return ToolResult(success=False, error=f"Input not found: {input_path}") + + Path(output_path).parent.mkdir(parents=True, exist_ok=True) + start = time.time() + + # Get source dimensions + probe_cmd = [ + "ffprobe", "-v", "error", + "-select_streams", "v:0", + "-show_entries", "stream=width,height", + "-of", "csv=p=0", + input_path, + ] + probe_out = self.run_command(probe_cmd, capture=True).strip() + src_w, src_h = [int(x.strip()) for x in probe_out.split(",")[:2]] + + # Calculate letterbox dimensions — fit source into output width, + # center vertically in the frame. + scale_factor = out_w / src_w + scaled_h = int(src_h * scale_factor) + # Ensure even dimensions + scaled_h = scaled_h if scaled_h % 2 == 0 else scaled_h + 1 + pad_y = (out_h - scaled_h) // 2 + + # Build filter chain + filters = [ + f"scale={out_w}:{scaled_h}", + f"pad={out_w}:{out_h}:0:{pad_y}:color={bg_color}", + ] + + # Title text at top + title_escaped = title.replace("'", "\\'").replace(":", "\\:") + filters.append( + f"drawtext=text='{title_escaped}'" + f":fontfile='{title_font}'" + f":fontsize={title_font_size}" + f":fontcolor={title_color}" + f":borderw=3:bordercolor=black" + f":x=(w-text_w)/2:y=60" + ) + + # Subtitle text at bottom + if subtitle: + sub_escaped = subtitle.replace("'", "\\'").replace(":", "\\:") + filters.append( + f"drawtext=text='{sub_escaped}'" + f":fontfile='segoeui.ttf'" + f":fontsize={subtitle_font_size}" + f":fontcolor=white@0.85" + f":x=(w-text_w)/2:y=h-100" + ) + + # Watermark centered on video + if watermark: + wm_escaped = watermark.replace("'", "\\'").replace(":", "\\:") + filters.append( + f"drawtext=text='{wm_escaped}'" + f":fontfile='segoeui.ttf'" + f":fontsize=36" + f":fontcolor=white@0.3" + f":x=(w-text_w)/2:y=(h-text_h)/2" + ) + + vf = ",".join(filters) + + cmd = [ + "ffmpeg", "-y", + "-i", input_path, + "-vf", vf, + "-c:v", "libx264", "-preset", "fast", "-crf", "18", + "-pix_fmt", "yuv420p", + "-c:a", "aac", "-b:a", "192k", + output_path, + ] + + try: + self.run_command(cmd) + except Exception as e: + return ToolResult(success=False, error=f"FFmpeg failed: {e}") + + if not Path(output_path).exists(): + return ToolResult(success=False, error="No output produced") + + elapsed = round(time.time() - start, 2) + + return ToolResult( + success=True, + data={ + "output": output_path, + "source_resolution": f"{src_w}x{src_h}", + "output_resolution": f"{out_w}x{out_h}", + "title": title, + "subtitle": subtitle, + "letterbox_y_offset": pad_y, + }, + artifacts=[output_path], + duration_seconds=elapsed, + ) diff --git a/tools/video/silence_cutter.py b/tools/video/silence_cutter.py new file mode 100644 index 00000000..922b1d64 --- /dev/null +++ b/tools/video/silence_cutter.py @@ -0,0 +1,481 @@ +"""Silence cutter tool for automatic jump cuts. + +Detects silent segments in talking-head footage and removes them, +creating tight jump cuts. Uses FFmpeg's silencedetect filter — no +external dependencies beyond FFmpeg. + +Modes: + - remove: Cut out silent segments entirely (jump cut) + - speed_up: Speed up silent segments instead of cutting (less jarring) + - mark: Don't cut — just output silence timestamps for manual review +""" + +from __future__ import annotations + +import json +import re +import time +from pathlib import Path +from typing import Any + +from tools.base_tool import ( + BaseTool, + Determinism, + ExecutionMode, + ResourceProfile, + RetryPolicy, + ResumeSupport, + ToolResult, + ToolStability, + ToolTier, +) + + +class SilenceCutter(BaseTool): + name = "silence_cutter" + version = "0.1.0" + tier = ToolTier.CORE + capability = "video_post" + provider = "ffmpeg" + stability = ToolStability.EXPERIMENTAL + execution_mode = ExecutionMode.SYNC + determinism = Determinism.DETERMINISTIC + + dependencies = ["cmd:ffmpeg"] + install_instructions = "Install FFmpeg: https://ffmpeg.org/download.html" + agent_skills = ["ffmpeg"] + + capabilities = [ + "silence_detection", + "jump_cut", + "silence_removal", + "silence_speedup", + ] + + input_schema = { + "type": "object", + "required": ["input_path"], + "properties": { + "input_path": {"type": "string"}, + "output_path": {"type": "string"}, + "mode": { + "type": "string", + "enum": ["remove", "speed_up", "mark"], + "default": "remove", + "description": "remove=jump cut, speed_up=fast-forward silence, mark=detect only", + }, + "silence_threshold_db": { + "type": "number", + "default": -35, + "description": "Audio level below this (in dB) is considered silence. Lower = more sensitive.", + }, + "min_silence_duration": { + "type": "number", + "default": 0.5, + "minimum": 0.1, + "description": "Minimum silence duration in seconds to trigger a cut", + }, + "padding_seconds": { + "type": "number", + "default": 0.08, + "minimum": 0.0, + "description": "Seconds of silence to keep on each side of speech (prevents clipped words)", + }, + "silence_speed_factor": { + "type": "number", + "default": 6.0, + "minimum": 1.5, + "maximum": 100.0, + "description": "Speed multiplier for silent segments (only used in speed_up mode)", + }, + "codec": {"type": "string", "default": "libx264"}, + "crf": {"type": "integer", "default": 18}, + }, + } + + resource_profile = ResourceProfile( + cpu_cores=4, ram_mb=2048, vram_mb=0, disk_mb=4000, network_required=False + ) + retry_policy = RetryPolicy(max_retries=1, retryable_errors=["FFmpeg error"]) + resume_support = ResumeSupport.FROM_START + idempotency_key_fields = [ + "input_path", "mode", "silence_threshold_db", + "min_silence_duration", "padding_seconds", + ] + side_effects = ["writes cut video to output_path"] + user_visible_verification = [ + "Watch output for unnaturally clipped words at cut points", + "Compare duration: output should be noticeably shorter than input", + ] + + def execute(self, inputs: dict[str, Any]) -> ToolResult: + input_path = Path(inputs["input_path"]) + if not input_path.exists(): + return ToolResult(success=False, error=f"Input not found: {input_path}") + + mode = inputs.get("mode", "remove") + start = time.time() + + # Step 1: Detect silence segments + threshold_db = inputs.get("silence_threshold_db", -35) + min_dur = inputs.get("min_silence_duration", 0.5) + padding = inputs.get("padding_seconds", 0.08) + + silences = self._detect_silence(input_path, threshold_db, min_dur) + + if not silences: + elapsed = time.time() - start + return ToolResult( + success=True, + data={ + "message": "No silence detected — video unchanged", + "silence_segments": 0, + "input": str(input_path), + "output": str(input_path), + }, + artifacts=[str(input_path)], + duration_seconds=round(elapsed, 2), + ) + + # Get total duration + total_duration = self._get_duration(input_path) + + # Step 2: Compute speech segments (inverse of silence) + speech_segments = self._compute_speech_segments( + silences, total_duration, padding + ) + + # Step 3: Handle based on mode + if mode == "mark": + elapsed = time.time() - start + output_json = Path( + inputs.get("output_path", str(input_path.with_suffix(".silence.json"))) + ) + result_data = { + "silences": silences, + "speech_segments": speech_segments, + "total_duration": total_duration, + "silence_duration": sum(s["duration"] for s in silences), + "speech_duration": sum(s["end"] - s["start"] for s in speech_segments), + } + output_json.parent.mkdir(parents=True, exist_ok=True) + output_json.write_text(json.dumps(result_data, indent=2), encoding="utf-8") + return ToolResult( + success=True, + data={ + "mode": "mark", + "silence_segments": len(silences), + "speech_segments": len(speech_segments), + "silence_duration_seconds": round(result_data["silence_duration"], 2), + "speech_duration_seconds": round(result_data["speech_duration"], 2), + "output": str(output_json), + }, + artifacts=[str(output_json)], + duration_seconds=round(elapsed, 2), + ) + + output_path = Path( + inputs.get("output_path", str(input_path.with_stem(f"{input_path.stem}_cut"))) + ) + output_path.parent.mkdir(parents=True, exist_ok=True) + codec = inputs.get("codec", "libx264") + crf = inputs.get("crf", 18) + + if mode == "speed_up": + speed_factor = inputs.get("silence_speed_factor", 6.0) + result = self._render_speed_up( + input_path, output_path, silences, speech_segments, + total_duration, speed_factor, codec, crf, + ) + else: + result = self._render_jump_cut( + input_path, output_path, speech_segments, codec, crf, + ) + + if not result.success: + return result + + elapsed = time.time() - start + + silence_dur = sum(s["duration"] for s in silences) + speech_dur = sum(s["end"] - s["start"] for s in speech_segments) + + return ToolResult( + success=True, + data={ + "mode": mode, + "input": str(input_path), + "output": str(output_path), + "input_duration": round(total_duration, 2), + "output_duration": round(speech_dur, 2) if mode == "remove" else None, + "silence_removed_seconds": round(silence_dur, 2), + "silence_segments": len(silences), + "speech_segments": len(speech_segments), + "time_saved_percent": round(silence_dur / total_duration * 100, 1) if total_duration > 0 else 0, + }, + artifacts=[str(output_path)], + duration_seconds=round(elapsed, 2), + ) + + def _detect_silence( + self, input_path: Path, threshold_db: float, min_duration: float + ) -> list[dict]: + """Detect silent segments using FFmpeg silencedetect filter.""" + cmd = [ + "ffmpeg", + "-i", str(input_path), + "-af", f"silencedetect=noise={threshold_db}dB:d={min_duration}", + "-f", "null", "-", + ] + + try: + result = self.run_command(cmd, timeout=300) + output = result.stderr + except Exception as e: + # FFmpeg writes to stderr even on success for filters + output = str(e) + + # Parse silencedetect output + # Format: [silencedetect @ ...] silence_start: 1.234 + # [silencedetect @ ...] silence_end: 2.567 | silence_duration: 1.333 + starts = re.findall(r"silence_start:\s*([\d.]+)", output) + ends = re.findall(r"silence_end:\s*([\d.]+)", output) + durations = re.findall(r"silence_duration:\s*([\d.]+)", output) + + silences = [] + for i in range(min(len(starts), len(ends))): + silences.append({ + "start": float(starts[i]), + "end": float(ends[i]), + "duration": float(durations[i]) if i < len(durations) else float(ends[i]) - float(starts[i]), + }) + + return silences + + def _get_duration(self, input_path: Path) -> float: + """Get video duration via ffprobe.""" + cmd = [ + "ffprobe", "-v", "quiet", + "-show_entries", "format=duration", + "-of", "json", str(input_path), + ] + try: + result = self.run_command(cmd) + data = json.loads(result.stdout) + return float(data["format"]["duration"]) + except Exception: + return 0.0 + + def _compute_speech_segments( + self, silences: list[dict], total_duration: float, padding: float + ) -> list[dict]: + """Compute speech segments as the inverse of silence segments, with padding.""" + segments = [] + cursor = 0.0 + + for silence in silences: + speech_end = silence["start"] + padding + if speech_end > cursor: + segments.append({"start": cursor, "end": min(speech_end, total_duration)}) + cursor = max(cursor, silence["end"] - padding) + + # Final segment after last silence + if cursor < total_duration: + segments.append({"start": cursor, "end": total_duration}) + + # Merge very short gaps (segments < 0.05s apart) + merged = [] + for seg in segments: + if seg["end"] - seg["start"] < 0.01: + continue # Skip tiny segments + if merged and seg["start"] - merged[-1]["end"] < 0.05: + merged[-1]["end"] = seg["end"] + else: + merged.append(seg) + + return merged + + def _render_jump_cut( + self, + input_path: Path, output_path: Path, + speech_segments: list[dict], + codec: str, crf: int, + ) -> ToolResult: + """Remove silence by concatenating speech segments.""" + if not speech_segments: + return ToolResult(success=False, error="No speech segments found") + + temp_dir = output_path.parent / ".silence_cut_tmp" + temp_dir.mkdir(parents=True, exist_ok=True) + + try: + # Cut each speech segment + seg_files = [] + for i, seg in enumerate(speech_segments): + seg_path = temp_dir / f"seg_{i:04d}.mp4" + cmd = [ + "ffmpeg", "-y", + "-i", str(input_path), + "-ss", f"{seg['start']:.3f}", + "-to", f"{seg['end']:.3f}", + "-c:v", codec, "-crf", str(crf), "-preset", "fast", + "-c:a", "aac", "-b:a", "192k", + # Force keyframe at start for clean cuts + "-force_key_frames", f"{seg['start']:.3f}", + str(seg_path), + ] + self.run_command(cmd, timeout=120) + if seg_path.exists() and seg_path.stat().st_size > 0: + seg_files.append(seg_path) + + if not seg_files: + return ToolResult(success=False, error="No segments were successfully cut") + + # Concat all segments + list_path = temp_dir / "concat_list.txt" + with open(list_path, "w", encoding="utf-8") as f: + for sf in seg_files: + safe_path = str(sf.resolve()).replace("\\", "/") + f.write(f"file '{safe_path}'\n") + + cmd = [ + "ffmpeg", "-y", + "-f", "concat", "-safe", "0", + "-i", str(list_path), + "-c", "copy", + str(output_path), + ] + self.run_command(cmd, timeout=120) + + return ToolResult(success=True) + except Exception as e: + return ToolResult(success=False, error=f"Jump cut render failed: {e}") + finally: + # Clean up temp files + for f in temp_dir.glob("*"): + try: + f.unlink() + except OSError: + pass + try: + temp_dir.rmdir() + except OSError: + pass + + def _render_speed_up( + self, + input_path: Path, output_path: Path, + silences: list[dict], speech_segments: list[dict], + total_duration: float, + speed_factor: float, + codec: str, crf: int, + ) -> ToolResult: + """Speed up silent segments instead of removing them. + + This is less jarring than jump cuts — the viewer sees a brief + fast-forward during pauses. + """ + temp_dir = output_path.parent / ".silence_speed_tmp" + temp_dir.mkdir(parents=True, exist_ok=True) + + try: + # Build a timeline of segments: speech at 1x, silence at Nx + all_segments = [] + + for seg in speech_segments: + all_segments.append({"start": seg["start"], "end": seg["end"], "speed": 1.0}) + + for sil in silences: + all_segments.append({"start": sil["start"], "end": sil["end"], "speed": speed_factor}) + + # Sort by start time and merge overlaps + all_segments.sort(key=lambda s: s["start"]) + + # Process each segment + seg_files = [] + for i, seg in enumerate(all_segments): + seg_path = temp_dir / f"seg_{i:04d}.mp4" + duration = seg["end"] - seg["start"] + if duration < 0.05: + continue + + if seg["speed"] == 1.0: + # Normal speed + cmd = [ + "ffmpeg", "-y", + "-i", str(input_path), + "-ss", f"{seg['start']:.3f}", + "-to", f"{seg['end']:.3f}", + "-c:v", codec, "-crf", str(crf), "-preset", "fast", + "-c:a", "aac", "-b:a", "192k", + str(seg_path), + ] + else: + # Speed up + pts = 1.0 / seg["speed"] + atempo_chain = self._build_atempo_chain(seg["speed"]) + cmd = [ + "ffmpeg", "-y", + "-i", str(input_path), + "-ss", f"{seg['start']:.3f}", + "-to", f"{seg['end']:.3f}", + "-filter:v", f"setpts={pts:.4f}*PTS", + "-filter:a", atempo_chain, + "-c:v", codec, "-crf", str(crf), "-preset", "fast", + "-c:a", "aac", "-b:a", "192k", + str(seg_path), + ] + + self.run_command(cmd, timeout=120) + if seg_path.exists() and seg_path.stat().st_size > 0: + seg_files.append(seg_path) + + if not seg_files: + return ToolResult(success=False, error="No segments rendered") + + # Concat + list_path = temp_dir / "concat_list.txt" + with open(list_path, "w", encoding="utf-8") as f: + for sf in seg_files: + safe_path = str(sf.resolve()).replace("\\", "/") + f.write(f"file '{safe_path}'\n") + + cmd = [ + "ffmpeg", "-y", + "-f", "concat", "-safe", "0", + "-i", str(list_path), + "-c", "copy", + str(output_path), + ] + self.run_command(cmd, timeout=120) + + return ToolResult(success=True) + except Exception as e: + return ToolResult(success=False, error=f"Speed-up render failed: {e}") + finally: + for f in temp_dir.glob("*"): + try: + f.unlink() + except OSError: + pass + try: + temp_dir.rmdir() + except OSError: + pass + + @staticmethod + def _build_atempo_chain(factor: float) -> str: + """Build atempo filter chain. atempo accepts [0.5, 100.0].""" + filters = [] + remaining = factor + while remaining > 100.0: + filters.append("atempo=100.0") + remaining /= 100.0 + while remaining < 0.5: + filters.append("atempo=0.5") + remaining /= 0.5 + filters.append(f"atempo={remaining:.4f}") + return ",".join(filters) + + def estimate_runtime(self, inputs: dict[str, Any]) -> float: + return 45.0