diff --git a/src-tauri/src/services/clip_exporter.rs b/src-tauri/src/services/clip_exporter.rs index d49e0b6..24b9c87 100644 --- a/src-tauri/src/services/clip_exporter.rs +++ b/src-tauri/src/services/clip_exporter.rs @@ -1,6 +1,795 @@ -use crate::models::{Clip, CutMode}; +use crate::models::{CaptionStyle, Clip, CutMode}; +use crate::services::dependency_manager::ffmpeg_bin; +use std::io::{BufRead, BufReader, Read as _}; use std::path::Path; -use std::process::Command; +use std::process::{Command, Stdio}; + +/// Strip YouTube auto-generated VTT karaoke tags and positioning metadata +/// that confuse ffmpeg's VTT parser / mov_text conversion. +/// Returns the path to a cleaned temp VTT file. +pub fn sanitize_vtt_for_ffmpeg(caption_path: &str, temp_dir: &str) -> Result { + let content = std::fs::read_to_string(caption_path) + .map_err(|e| format!("Failed to read VTT file '{}': {e}", caption_path))?; + + std::fs::create_dir_all(temp_dir) + .map_err(|e| format!("Failed to create temp dir for sanitized VTT: {e}"))?; + + let out_path = Path::new(temp_dir).join("sanitized.vtt"); + let mut output = String::with_capacity(content.len()); + + for line in content.lines() { + if line.contains("-->") { + // Strip positioning metadata (align:start position:0% etc.) from timestamp lines + if let Some(arrow_end) = line.find("-->") { + let after_arrow = &line[arrow_end + 3..]; + // Find end of the second timestamp (digits, colons, dots/commas) + let ts_end = after_arrow + .find(|c: char| !c.is_ascii_digit() && c != ':' && c != '.' && c != ',' && c != ' ') + .unwrap_or(after_arrow.len()); + let cleaned = format!("{}{}", &line[..arrow_end + 3], &after_arrow[..ts_end].trim_end()); + output.push_str(&cleaned); + } else { + output.push_str(line); + } + } else { + // Strip inline tags: , , and inline timestamps like <00:00:00.599> + let cleaned = strip_vtt_tags(line); + output.push_str(&cleaned); + } + output.push('\n'); + } + + std::fs::write(&out_path, &output) + .map_err(|e| format!("Failed to write sanitized VTT: {e}"))?; + + Ok(out_path.to_string_lossy().to_string()) +} + +/// Sanitize a VTT file AND trim it to a clip's time range, shifting timestamps +/// to start from 0. Used for mux mode where the video uses input seeking +/// (output PTS starts at ~0) so subtitle timestamps must also start from 0. +pub fn trim_and_sanitize_vtt( + caption_path: &str, + temp_dir: &str, + start_time: f64, + end_time: f64, +) -> Result { + let content = std::fs::read_to_string(caption_path) + .map_err(|e| format!("Failed to read VTT file '{}': {e}", caption_path))?; + + std::fs::create_dir_all(temp_dir) + .map_err(|e| format!("Failed to create temp dir for trimmed VTT: {e}"))?; + + let out_path = Path::new(temp_dir).join("trimmed.vtt"); + let mut output = String::from("WEBVTT\n\n"); + + // Split into blocks on blank lines, process each cue + let blocks = content.replace("\r\n", "\n"); + let blocks: Vec<&str> = blocks.split("\n\n").collect(); + + for block in &blocks { + let lines: Vec<&str> = block.trim().lines().collect(); + // Find the timestamp line + let ts_idx = lines.iter().position(|l| l.contains("-->")); + let ts_idx = match ts_idx { + Some(i) => i, + None => continue, + }; + + let ts_line = lines[ts_idx]; + // Parse the two timestamps from the line + let (cue_start, cue_end) = match parse_vtt_timestamp_line(ts_line) { + Some(pair) => pair, + None => continue, + }; + + // Skip cues outside the clip range + if cue_end <= start_time || cue_start >= end_time { + continue; + } + + // Clamp and shift to start from 0 + let shifted_start = (cue_start - start_time).max(0.0); + let shifted_end = (cue_end - start_time).min(end_time - start_time); + + if shifted_end <= shifted_start { + continue; + } + + // Write the cue with shifted timestamps and sanitized text + output.push_str(&format_vtt_timestamp(shifted_start)); + output.push_str(" --> "); + output.push_str(&format_vtt_timestamp(shifted_end)); + output.push('\n'); + + // Collect text lines (everything after the timestamp line), sanitize tags + for &line in &lines[ts_idx + 1..] { + let cleaned = strip_vtt_tags(line); + let cleaned = cleaned.trim(); + if !cleaned.is_empty() { + output.push_str(cleaned); + output.push('\n'); + } + } + output.push('\n'); + } + + std::fs::write(&out_path, &output) + .map_err(|e| format!("Failed to write trimmed VTT: {e}"))?; + + Ok(out_path.to_string_lossy().to_string()) +} + +/// Parse a VTT timestamp line like "00:01:23.456 --> 00:02:34.567 align:start" +/// Returns (start_seconds, end_seconds) or None if parsing fails. +fn parse_vtt_timestamp_line(line: &str) -> Option<(f64, f64)> { + let arrow_pos = line.find("-->")?; + let before = line[..arrow_pos].trim(); + let after_arrow = &line[arrow_pos + 3..]; + // The end timestamp ends at the first non-timestamp character + let end_ts_str = after_arrow + .trim_start() + .split(|c: char| !c.is_ascii_digit() && c != ':' && c != '.' && c != ',') + .next()?; + + Some((parse_vtt_ts(before)?, parse_vtt_ts(end_ts_str)?)) +} + +/// Parse a single VTT timestamp "HH:MM:SS.mmm" or "MM:SS.mmm" into seconds. +fn parse_vtt_ts(ts: &str) -> Option { + let normalized = ts.replace(',', "."); + let parts: Vec<&str> = normalized.split(':').collect(); + match parts.len() { + 3 => { + let h: f64 = parts[0].parse().ok()?; + let m: f64 = parts[1].parse().ok()?; + let s: f64 = parts[2].parse().ok()?; + Some(h * 3600.0 + m * 60.0 + s) + } + 2 => { + let m: f64 = parts[0].parse().ok()?; + let s: f64 = parts[1].parse().ok()?; + Some(m * 60.0 + s) + } + _ => None, + } +} + +/// Format seconds as a VTT timestamp "HH:MM:SS.mmm". +fn format_vtt_timestamp(secs: f64) -> String { + let total_ms = (secs * 1000.0).round() as u64; + let ms = total_ms % 1000; + let total_s = total_ms / 1000; + let s = total_s % 60; + let total_m = total_s / 60; + let m = total_m % 60; + let h = total_m / 60; + format!("{:02}:{:02}:{:02}.{:03}", h, m, s, ms) +} + +fn strip_vtt_tags(line: &str) -> String { + let mut result = String::with_capacity(line.len()); + let mut in_tag = false; + let mut chars = line.chars().peekable(); + + while let Some(c) = chars.next() { + if c == '<' { + in_tag = true; + } else if c == '>' { + in_tag = false; + } else if !in_tag { + result.push(c); + } + } + + result +} + +/// Convert a hex color "#RRGGBB" to ASS color format "&H00BBGGRR". +/// ASS uses BGR byte order with an alpha prefix byte (00 = fully opaque). +fn hex_to_ass_color(hex: &str) -> String { + let hex = hex.trim_start_matches('#'); + if hex.len() >= 6 { + let r = &hex[0..2]; + let g = &hex[2..4]; + let b = &hex[4..6]; + format!("&H00{}{}{}", b.to_uppercase(), g.to_uppercase(), r.to_uppercase()) + } else { + "&H00FFFFFF".to_string() + } +} + +/// Convert a background opacity (0.0=transparent, 1.0=opaque) to ASS BackColour. +/// ASS alpha: 00=opaque, FF=transparent (inverted from CSS). +fn opacity_to_ass_back_colour(opacity: f64) -> String { + let alpha = ((1.0 - opacity.clamp(0.0, 1.0)) * 255.0).round() as u8; + format!("&H{:02X}000000", alpha) +} + +/// Build an ASS force_style string from user caption settings. +fn caption_style_to_force_style(style: &CaptionStyle) -> String { + let font_size = (style.font_size as f64 * 2.0).round() as u32; + let primary_colour = hex_to_ass_color(&style.text_color); + let back_colour = opacity_to_ass_back_colour(style.background_opacity); + let (outline, shadow) = if style.text_outline { + ("2", "1") + } else { + ("0", "0") + }; + // ASS Alignment: 2 = bottom-center, 8 = top-center + let alignment = if style.position == "top" { "8" } else { "2" }; + let margin_v = if style.position == "top" { "40" } else { "40" }; + + format!( + "Fontsize={},PrimaryColour={},BackColour={},Outline={},Shadow={},Alignment={},MarginV={},BorderStyle=4", + font_size, primary_colour, back_colour, outline, shadow, alignment, margin_v + ) +} + +/// Convert a hex color "#RRGGBB" to ASS color with a specific alpha. +/// Returns "&H" format. +fn hex_to_ass_color_with_alpha(hex: &str, alpha: u8) -> String { + let hex = hex.trim_start_matches('#'); + if hex.len() >= 6 { + let r = &hex[0..2]; + let g = &hex[2..4]; + let b = &hex[4..6]; + format!("&H{:02X}{}{}{}", alpha, b.to_uppercase(), g.to_uppercase(), r.to_uppercase()) + } else { + format!("&H{:02X}FFFFFF", alpha) + } +} + +/// Format seconds as an ASS timestamp "H:MM:SS.cc" (centiseconds, not milliseconds). +fn format_ass_timestamp(secs: f64) -> String { + let total_cs = (secs * 100.0).round() as u64; + let cs = total_cs % 100; + let total_s = total_cs / 100; + let s = total_s % 60; + let total_m = total_s / 60; + let m = total_m % 60; + let h = total_m / 60; + format!("{}:{:02}:{:02}.{:02}", h, m, s, cs) +} + +/// Parse word-level timing data from YouTube-style VTT cue text. +/// Returns a Vec of (word_text, start_time_seconds) pairs. +/// +/// YouTube format: `the<00:00:00.599> gym<00:00:00.840> I` +fn parse_vtt_word_timings(raw_text: &str, cue_start: f64) -> Vec<(String, f64)> { + if !raw_text.contains("") { + return Vec::new(); + } + + let mut segments: Vec<(String, f64)> = Vec::new(); + + // Find the first timestamp tag position + let first_ts_pos = find_timestamp_tag_pos(raw_text); + + // Extract leading text (before any timestamp tag) + if let Some(pos) = first_ts_pos { + if pos > 0 { + let leading = strip_vtt_tags(&raw_text[..pos]).trim().to_string(); + if !leading.is_empty() { + segments.push((leading, cue_start)); + } + } + } + + // Parse word pairs + let mut search_from = 0; + while search_from < raw_text.len() { + // Find next timestamp tag like <00:00:05.320> + if let Some((ts_val, ts_end)) = find_next_timestamp(raw_text, search_from) { + // After the timestamp, expect ... + let after_ts = &raw_text[ts_end..]; + if let Some(c_start) = after_ts.find("") { + let content_start = c_start + 3; + if let Some(c_end) = after_ts[content_start..].find("") { + let word = after_ts[content_start..content_start + c_end].trim().to_string(); + if !word.is_empty() { + if let Some(time) = parse_vtt_ts(&ts_val) { + segments.push((word, time)); + } + } + search_from = ts_end + content_start + c_end + 4; // past + continue; + } + } + search_from = ts_end; + } else { + break; + } + } + + segments +} + +/// Find the byte position of the first `` tag in text. +fn find_timestamp_tag_pos(text: &str) -> Option { + let mut i = 0; + let bytes = text.as_bytes(); + while i < bytes.len() { + if bytes[i] == b'<' && i + 1 < bytes.len() && bytes[i + 1].is_ascii_digit() { + // Check if this looks like a timestamp tag + if let Some(end) = text[i..].find('>') { + let inner = &text[i + 1..i + end]; + if inner.contains(':') && inner.contains('.') { + return Some(i); + } + } + } + i += 1; + } + None +} + +/// Find the next `` timestamp tag starting from `from`. +/// Returns (timestamp_string, byte_position_after_closing_bracket). +fn find_next_timestamp(text: &str, from: usize) -> Option<(String, usize)> { + let slice = &text[from..]; + let mut i = 0; + let bytes = slice.as_bytes(); + while i < bytes.len() { + if bytes[i] == b'<' && i + 1 < bytes.len() && bytes[i + 1].is_ascii_digit() { + if let Some(end_rel) = slice[i..].find('>') { + let inner = &slice[i + 1..i + end_rel]; + if inner.contains(':') && (inner.contains('.') || inner.contains(',')) { + return Some((inner.to_string(), from + i + end_rel + 1)); + } + } + } + i += 1; + } + None +} + +/// A single spoken line extracted from a YouTube VTT cue. +/// Used to build the rolling two-line subtitle display. +struct SpokenLine { + plain_text: String, + raw_text: String, + start_time: f64, + end_time: f64, + has_karaoke: bool, + is_non_speech: bool, +} + +/// Extract a flat sequence of spoken lines from YouTube auto-generated VTT content. +/// Skips zero-duration transition cues and extracts only the active (karaoke) line +/// from two-line cues, ignoring the context line (which is reconstructed from the +/// previous SpokenLine during ASS generation). +fn extract_spoken_lines(content: &str) -> Vec { + let normalized = content.replace("\r\n", "\n"); + let blocks: Vec<&str> = normalized.split("\n\n").collect(); + let mut lines = Vec::new(); + + for block in &blocks { + let block_lines: Vec<&str> = block.trim().lines().collect(); + let ts_idx = block_lines.iter().position(|l| l.contains("-->")); + let ts_idx = match ts_idx { + Some(i) => i, + None => continue, + }; + + let ts_line = block_lines[ts_idx]; + let (cue_start, cue_end) = match parse_vtt_timestamp_line(ts_line) { + Some(pair) => pair, + None => continue, + }; + + // Skip zero-duration transition cues (YouTube uses e.g. 00:02.629 --> 00:02.639) + if (cue_end - cue_start) < 0.05 { + continue; + } + + let text_lines: Vec<&str> = block_lines[ts_idx + 1..].iter().copied().collect(); + if text_lines.is_empty() { + continue; + } + + // Find the karaoke line (contains tags) + let karaoke_line = text_lines.iter().find(|l| l.contains("")); + + if let Some(raw) = karaoke_line { + let plain = strip_vtt_tags(raw).trim().to_string(); + if !plain.is_empty() { + lines.push(SpokenLine { + plain_text: plain, + raw_text: raw.to_string(), + start_time: cue_start, + end_time: cue_end, + has_karaoke: true, + is_non_speech: false, + }); + } + } else { + // No karaoke — could be plain text or non-speech like [Music] + let all_text: Vec = text_lines.iter() + .map(|l| strip_vtt_tags(l).trim().to_string()) + .filter(|l| !l.is_empty()) + .collect(); + let joined = all_text.join(" "); + if joined.is_empty() { + continue; + } + let is_non_speech = joined.starts_with('[') && joined.ends_with(']'); + lines.push(SpokenLine { + plain_text: joined.clone(), + raw_text: joined, + start_time: cue_start, + end_time: cue_end, + has_karaoke: false, + is_non_speech, + }); + } + } + + lines +} + +/// Build a karaoke text string with \k tags from a raw VTT line's word timings. +fn build_karaoke_text(raw: &str, cue_start: f64, cue_end: f64) -> String { + let words = parse_vtt_word_timings(raw, cue_start); + if words.len() >= 2 { + let mut parts = Vec::new(); + for (i, (word, start)) in words.iter().enumerate() { + let next_start = if i + 1 < words.len() { + words[i + 1].1 + } else { + cue_end + }; + let duration_cs = ((next_start - start) * 100.0).round().max(1.0) as u64; + parts.push(format!("{{\\k{}}}{}", duration_cs, word)); + } + parts.join(" ") + } else { + strip_vtt_tags(raw) + } +} + +/// Group spoken lines into 2-line pages for the teleprompter display. +/// If the gap between two consecutive lines exceeds `gap_threshold` seconds, +/// the pair is split into separate single-line pages. +/// Non-speech lines (e.g., [Music]) always get their own page. +fn group_into_pages(spoken_lines: &[SpokenLine], gap_threshold: f64) -> Vec> { + let mut pages: Vec> = Vec::new(); + let mut i = 0; + + while i < spoken_lines.len() { + let line = &spoken_lines[i]; + + if line.is_non_speech { + pages.push(vec![i]); + i += 1; + continue; + } + + if i + 1 < spoken_lines.len() { + let next = &spoken_lines[i + 1]; + let gap = next.start_time - line.end_time; + + if gap <= gap_threshold && !next.is_non_speech { + pages.push(vec![i, i + 1]); + i += 2; + continue; + } + } + + pages.push(vec![i]); + i += 1; + } + + pages +} + +/// Build karaoke text for a 1-or-2-line page, stitching word timings +/// across lines with \N as the visual line break. The \k durations flow +/// continuously so karaoke highlighting progresses top-to-bottom. +fn build_page_karaoke_text(lines: &[&SpokenLine]) -> String { + struct WordEntry { + text: String, + start: f64, + is_line_break_before: bool, + } + + let mut entries: Vec = Vec::new(); + + for (line_idx, line) in lines.iter().enumerate() { + let is_new_line = line_idx > 0; + + if line.has_karaoke { + let words = parse_vtt_word_timings(&line.raw_text, line.start_time); + if words.len() >= 2 { + for (w_idx, (word, start)) in words.iter().enumerate() { + entries.push(WordEntry { + text: word.clone(), + start: *start, + is_line_break_before: is_new_line && w_idx == 0, + }); + } + } else { + entries.push(WordEntry { + text: strip_vtt_tags(&line.raw_text), + start: line.start_time, + is_line_break_before: is_new_line, + }); + } + } else { + entries.push(WordEntry { + text: line.plain_text.clone(), + start: line.start_time, + is_line_break_before: is_new_line, + }); + } + } + + if entries.is_empty() { + return String::new(); + } + + let page_end = lines.last().unwrap().end_time; + + let mut parts: Vec = Vec::new(); + for (i, entry) in entries.iter().enumerate() { + let next_start = if i + 1 < entries.len() { + entries[i + 1].start + } else { + page_end + }; + let duration_cs = ((next_start - entry.start) * 100.0).round().max(1.0) as u64; + + let prefix = if entry.is_line_break_before { + "\\N" + } else if i > 0 { + " " + } else { + "" + }; + parts.push(format!("{}{{\\k{}}}{}", prefix, duration_cs, entry.text)); + } + + parts.join("") +} + +/// Position parameters for the rolling subtitle layout. +struct RollingLayout { + cx: i32, + y_bottom: i32, + line_height: i32, + scroll_ms: i32, +} + +/// Convert a word-timed VTT file to an ASS file with rolling two-line display. +pub fn vtt_to_ass_with_karaoke( + caption_path: &str, + temp_dir: &str, + style: Option<&CaptionStyle>, +) -> Result { + let content = std::fs::read_to_string(caption_path) + .map_err(|e| format!("Failed to read VTT file '{}': {e}", caption_path))?; + + std::fs::create_dir_all(temp_dir) + .map_err(|e| format!("Failed to create temp dir for ASS: {e}"))?; + + let out_path = Path::new(temp_dir).join("karaoke.ass"); + + let font_size = style.map_or(45, |s| (s.font_size as f64 * 2.5).round() as u32); + let primary_colour = style.map_or("&H00FFFFFF".to_string(), |s| hex_to_ass_color(&s.text_color)); + let secondary_colour = "&H73CCCCCC".to_string(); + let outline_colour = "&H00000000".to_string(); + let back_colour = style.map_or("&H80000000".to_string(), |s| opacity_to_ass_back_colour(s.background_opacity)); + let text_outline = style.map_or(true, |s| s.text_outline); + let (border_style, outline_val, shadow_val) = if text_outline { + (4, 2, 0) + } else { + (3, 0, 0) + }; + let margin_v: i32 = 40; + + let line_height = font_size as i32 + outline_val * 2 + 4; + let layout = RollingLayout { + cx: 960, + y_bottom: 1080 - margin_v, + line_height, + scroll_ms: 350, + }; + + let mut output = String::new(); + + output.push_str("[Script Info]\n"); + output.push_str("ScriptType: v4.00+\n"); + output.push_str("PlayResX: 1920\n"); + output.push_str("PlayResY: 1080\n"); + output.push_str("WrapStyle: 0\n"); + output.push_str("ScaledBorderAndShadow: yes\n"); + output.push('\n'); + + output.push_str("[V4+ Styles]\n"); + output.push_str("Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding\n"); + output.push_str(&format!( + "Style: Default,Arial,{},{},{},{},{},0,0,0,0,100,100,0,0,{},{},{},2,20,20,{},1\n", + font_size, primary_colour, secondary_colour, outline_colour, back_colour, + border_style, outline_val, shadow_val, margin_v + )); + output.push('\n'); + + output.push_str("[Events]\n"); + output.push_str("Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text\n"); + + let spoken_lines = extract_spoken_lines(&content); + let gap_threshold = 2.0; + + for (i, line) in spoken_lines.iter().enumerate() { + if line.is_non_speech { + output.push_str(&format!( + "Dialogue: 0,{},{},Default,,0,0,0,,{{\\an2\\pos({},{})}}{}\n", + format_ass_timestamp(line.start_time), + format_ass_timestamp(line.end_time), + layout.cx, layout.y_bottom, + line.plain_text + )); + continue; + } + + let next_line = spoken_lines.get(i + 1); + let next_next_line = spoken_lines.get(i + 2); + + let phase1_start = line.start_time; + let phase1_end = next_line + .filter(|nl| !nl.is_non_speech) + .map_or(line.end_time, |nl| nl.start_time); + + let karaoke_text = if line.has_karaoke { + build_karaoke_text(&line.raw_text, line.start_time, line.end_time) + } else { + line.plain_text.clone() + }; + + let y1_from = layout.y_bottom; + let y1_to = layout.y_bottom - layout.line_height; + + output.push_str(&format!( + "Dialogue: 0,{},{},Default,,0,0,0,,{{\\an2\\move({},{},{},{},0,{})}}{}\n", + format_ass_timestamp(phase1_start), + format_ass_timestamp(phase1_end), + layout.cx, y1_from, layout.cx, y1_to, + layout.scroll_ms, + karaoke_text + )); + + let has_gap = next_line.map_or(true, |nl| nl.start_time - line.end_time > gap_threshold); + if has_gap { + let hold_end = line.end_time + 1.0; + let phase3_start = hold_end; + let phase3_end = phase3_start + 0.4; + let y_ctx = layout.y_bottom - layout.line_height; + + output.push_str(&format!( + "Dialogue: 0,{},{},Default,,0,0,0,,{{\\an2\\pos({},{})\\fad(0,400)}}{}\n", + format_ass_timestamp(phase3_start), + format_ass_timestamp(phase3_end), + layout.cx, y_ctx, + line.plain_text + )); + continue; + } + + let phase2_start = phase1_end; + let phase2_end = if let Some(nnl) = next_next_line.filter(|nnl| !nnl.is_non_speech) { + let nn_gap = nnl.start_time - next_line.unwrap().end_time; + if nn_gap > gap_threshold { + next_line.unwrap().end_time + 1.0 + } else { + nnl.start_time + } + } else { + next_line.map_or(phase2_start + 2.0, |nl| nl.end_time) + }; + + if phase2_end > phase2_start { + let y2_from = layout.y_bottom - layout.line_height; + let y2_to = layout.y_bottom - 2 * layout.line_height; + + output.push_str(&format!( + "Dialogue: 0,{},{},Default,,0,0,0,,{{\\an2\\move({},{},{},{},0,{})}}{}\n", + format_ass_timestamp(phase2_start), + format_ass_timestamp(phase2_end), + layout.cx, y2_from, layout.cx, y2_to, + layout.scroll_ms, + line.plain_text + )); + + let phase3_start = phase2_end; + let phase3_end = phase3_start + 0.4; + let y3 = layout.y_bottom - 2 * layout.line_height; + + output.push_str(&format!( + "Dialogue: 0,{},{},Default,,0,0,0,,{{\\an2\\pos({},{})\\fad(0,400)}}{}\n", + format_ass_timestamp(phase3_start), + format_ass_timestamp(phase3_end), + layout.cx, y3, + line.plain_text + )); + } + } + + std::fs::write(&out_path, &output) + .map_err(|e| format!("Failed to write ASS file: {e}"))?; + + Ok(out_path.to_string_lossy().to_string()) +} + +/// Parse `out_time_ms=` from ffmpeg's `-progress` output. +/// Returns the time in seconds. +fn parse_progress_time_us(line: &str) -> Option { + let val = line.strip_prefix("out_time_ms=")?; + let us: i64 = val.trim().parse().ok()?; + if us < 0 { + return None; + } + Some(us as f64 / 1_000_000.0) +} + +/// Run an ffmpeg command with real-time progress reporting. +/// Uses ffmpeg's `-progress pipe:1 -nostats` flag to get machine-readable +/// `\n`-delimited progress on stdout (avoids the `\r`-only stderr issue). +/// Calls `on_progress(percent)` with values 0.0..=1.0 as encoding proceeds. +fn run_ffmpeg_with_progress( + args: &[String], + duration: f64, + on_progress: &dyn Fn(f64), +) -> Result<(), String> { + // Inject -progress pipe:1 -nostats before the output file (last arg) + let mut full_args = args.to_vec(); + if let Some(output_pos) = full_args.len().checked_sub(1) { + full_args.insert(output_pos, "-nostats".to_string()); + full_args.insert(output_pos, "pipe:1".to_string()); + full_args.insert(output_pos, "-progress".to_string()); + } + + let mut child = Command::new(ffmpeg_bin()) + .args(&full_args) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .spawn() + .map_err(|e| format!("Failed to run ffmpeg: {e}"))?; + + // Read progress from stdout (-progress pipe:1 outputs \n-delimited key=value) + let stderr_handle = child.stderr.take().map(|stderr| { + std::thread::spawn(move || { + let mut buf = String::new(); + let mut reader = BufReader::new(stderr); + let _ = reader.read_to_string(&mut buf); + buf + }) + }); + + if let Some(stdout) = child.stdout.take() { + let reader = BufReader::new(stdout); + for line in reader.lines().map_while(Result::ok) { + if let Some(time_secs) = parse_progress_time_us(&line) { + let percent = if duration > 0.0 { + (time_secs / duration).clamp(0.0, 1.0) + } else { + 0.0 + }; + on_progress(percent); + } + } + } + + let status = child.wait().map_err(|e| format!("ffmpeg wait failed: {e}"))?; + + if !status.success() { + let stderr_text = stderr_handle + .and_then(|h| h.join().ok()) + .unwrap_or_default(); + return Err(format!("ffmpeg export failed: {}", stderr_text)); + } + on_progress(1.0); + Ok(()) +} pub fn build_ffmpeg_args( clip: &Clip, @@ -45,6 +834,142 @@ pub fn build_ffmpeg_args( args } +/// Build ffmpeg args for exporting a clip with subtitles muxed as a track. +/// Works with both lossless and precise cut modes. +/// `caption_path` must be a pre-trimmed VTT (timestamps shifted to start from 0) +/// produced by `trim_and_sanitize_vtt()`. +pub fn build_ffmpeg_args_mux_subs( + clip: &Clip, + source: &str, + output: &str, + cut_mode: &CutMode, + caption_path: &str, +) -> Vec { + let ext = Path::new(output) + .extension() + .map(|e| e.to_string_lossy().to_ascii_lowercase()) + .unwrap_or_default(); + + let sub_codec = match ext.as_str() { + "mp4" | "m4v" => "mov_text", + "mkv" => "srt", + _ => "mov_text", + }; + + let duration = clip.end_time - clip.start_time; + + // Use INPUT seeking on the video (-ss/-t BEFORE -i) for speed (especially + // lossless). The pre-trimmed VTT already has timestamps starting from 0 + // to match the video output PTS. + let mut args = vec![ + "-y".to_string(), + "-ss".to_string(), + format!("{}", clip.start_time), + "-t".to_string(), + format!("{}", duration), + "-i".to_string(), + source.to_string(), + "-i".to_string(), + caption_path.to_string(), + "-map".to_string(), + "0:v".to_string(), + "-map".to_string(), + "0:a".to_string(), + "-map".to_string(), + "1:s".to_string(), + ]; + + match cut_mode { + CutMode::Lossless => { + args.extend([ + "-c:v".to_string(), + "copy".to_string(), + "-c:a".to_string(), + "copy".to_string(), + ]); + } + CutMode::Precise { format: _ } => { + let (video_codec, audio_codec) = match ext.as_str() { + "webm" => ("libvpx-vp9", "libopus"), + "mp4" | "mkv" => ("libx264", "aac"), + _ => ("libx264", "aac"), + }; + args.extend([ + "-c:v".to_string(), + video_codec.to_string(), + "-c:a".to_string(), + audio_codec.to_string(), + ]); + } + } + + args.extend(["-c:s".to_string(), sub_codec.to_string()]); + args.push(output.to_string()); + args +} + +/// Build ffmpeg args for exporting a clip with subtitles burned into the video. +/// Always re-encodes. If the user selected lossless, we override to H.264/AAC. +/// `caption_path` should be an ASS file (from `vtt_to_ass_with_karaoke()`) with +/// style and karaoke tags baked in. Uses the `ass=` filter (not `subtitles=`) +/// to preserve `\kf` karaoke timing through libass. +pub fn build_ffmpeg_args_burnin_subs( + clip: &Clip, + source: &str, + output: &str, + cut_mode: &CutMode, + caption_path: &str, + _caption_style: Option<&CaptionStyle>, +) -> Vec { + let ext = Path::new(output) + .extension() + .map(|e| e.to_string_lossy().to_ascii_lowercase()) + .unwrap_or_default(); + + let (video_codec, audio_codec) = match cut_mode { + CutMode::Precise { format: _ } => match ext.as_str() { + "webm" => ("libvpx-vp9", "libopus"), + "mp4" | "mkv" => ("libx264", "aac"), + _ => ("libx264", "aac"), + }, + // Lossless override — burn-in requires re-encoding + CutMode::Lossless => ("libx264", "aac"), + }; + + // Escape the path for the ffmpeg ass filter (colons and backslashes) + let escaped_path = caption_path + .replace('\\', "\\\\") + .replace(':', "\\:") + .replace("'", "\\'"); + + let duration = clip.end_time - clip.start_time; + + // Use the `ass=` filter, NOT `subtitles=`. The `subtitles` filter routes + // through libavformat and can flatten karaoke \kf tags. The `ass` filter + // passes the file directly to libass, preserving all override tags. + // Style is baked into the ASS [V4+ Styles] header, no force_style needed. + let vf = format!("ass={}", escaped_path); + + // Use output seeking (-ss/-t AFTER -i) so the ass filter reads + // the original timestamps and they match the output time range. + vec![ + "-y".to_string(), + "-i".to_string(), + source.to_string(), + "-ss".to_string(), + format!("{}", clip.start_time), + "-t".to_string(), + format!("{}", duration), + "-vf".to_string(), + vf, + "-c:v".to_string(), + video_codec.to_string(), + "-c:a".to_string(), + audio_codec.to_string(), + output.to_string(), + ] +} + pub fn expand_tilde_path(path: &str) -> String { if path == "~" { return dirs::home_dir() @@ -125,19 +1050,45 @@ pub fn export_single_clip( source: &str, output: &str, cut_mode: &CutMode, + on_progress: &dyn Fn(f64), ) -> Result<(), String> { let args = build_ffmpeg_args(clip, source, output, cut_mode); - let output_result = Command::new("ffmpeg") - .args(&args) - .output() - .map_err(|e| format!("Failed to run ffmpeg: {e}"))?; + let duration = clip.end_time - clip.start_time; + run_ffmpeg_with_progress(&args, duration, on_progress) +} - if !output_result.status.success() { - let stderr = String::from_utf8_lossy(&output_result.stderr); - return Err(format!("ffmpeg export failed: {stderr}")); - } +pub fn export_single_clip_with_subs( + clip: &Clip, + source: &str, + output: &str, + cut_mode: &CutMode, + caption_path: &str, + burn_in: bool, + caption_style: Option<&CaptionStyle>, + on_progress: &dyn Fn(f64), +) -> Result<(), String> { + let temp_dir = std::env::temp_dir() + .join("video-clipper-subs-sanitize") + .to_string_lossy() + .to_string(); - Ok(()) + let args = if burn_in { + // Convert VTT to ASS with karaoke \kf tags and style baked into the header + let ass_path = vtt_to_ass_with_karaoke(caption_path, &temp_dir, caption_style)?; + // Style is in the ASS header, so pass None for caption_style + build_ffmpeg_args_burnin_subs(clip, source, output, cut_mode, &ass_path, None) + } else { + let trimmed = trim_and_sanitize_vtt(caption_path, &temp_dir, clip.start_time, clip.end_time)?; + build_ffmpeg_args_mux_subs(clip, source, output, cut_mode, &trimmed) + }; + + eprintln!( + "[video-clipper:export] with subs (burn_in={}) — ffmpeg {}", + burn_in, + args.join(" ") + ); + let duration = clip.end_time - clip.start_time; + run_ffmpeg_with_progress(&args, duration, on_progress) } pub fn export_merged( @@ -146,16 +1097,24 @@ pub fn export_merged( output: &str, cut_mode: &CutMode, temp_dir: &str, + on_progress: &dyn Fn(usize, f64), ) -> Result<(), String> { let mut temp_files = Vec::new(); let ext = get_extension(source, cut_mode); + let total = clips.len(); for (i, clip) in clips.iter().enumerate() { let temp_path = format!("{}/merge_part_{}.{}", temp_dir, i, ext); - export_single_clip(clip, source, &temp_path, cut_mode)?; + let clip_idx = i; + export_single_clip(clip, source, &temp_path, cut_mode, &|pct| { + on_progress(clip_idx, pct); + })?; temp_files.push(temp_path); } + // Concat step — no granular progress, signal start + on_progress(total, 0.0); + let concat_path = format!("{}/concat_list.txt", temp_dir); let concat_content: String = temp_files .iter() @@ -165,7 +1124,7 @@ pub fn export_merged( std::fs::write(&concat_path, concat_content) .map_err(|e| format!("Failed to write concat file: {e}"))?; - let result = Command::new("ffmpeg") + let result = Command::new(ffmpeg_bin()) .args([ "-y", "-f", @@ -191,6 +1150,82 @@ pub fn export_merged( } let _ = std::fs::remove_file(&concat_path); + on_progress(total, 1.0); + Ok(()) +} + +/// Merged export with subtitle support. Each segment is exported with subs, +/// then the segments are concatenated. +pub fn export_merged_with_subs( + clips: &[Clip], + source: &str, + output: &str, + cut_mode: &CutMode, + temp_dir: &str, + caption_path: &str, + burn_in: bool, + caption_style: Option<&CaptionStyle>, + on_progress: &dyn Fn(usize, f64), +) -> Result<(), String> { + let mut temp_files = Vec::new(); + let ext = if burn_in { + match cut_mode { + CutMode::Precise { format } => format.clone(), + CutMode::Lossless => "mp4".to_string(), + } + } else { + get_extension(source, cut_mode) + }; + let total = clips.len(); + + for (i, clip) in clips.iter().enumerate() { + let temp_path = format!("{}/merge_part_{}.{}", temp_dir, i, ext); + let clip_idx = i; + export_single_clip_with_subs( + clip, source, &temp_path, cut_mode, caption_path, burn_in, caption_style, + &|pct| { on_progress(clip_idx, pct); }, + )?; + temp_files.push(temp_path); + } + + on_progress(total, 0.0); + + let concat_path = format!("{}/concat_list.txt", temp_dir); + let concat_content: String = temp_files + .iter() + .map(|p| format!("file '{}'", p)) + .collect::>() + .join("\n"); + std::fs::write(&concat_path, concat_content) + .map_err(|e| format!("Failed to write concat file: {e}"))?; + + let result = Command::new(ffmpeg_bin()) + .args([ + "-y", + "-f", + "concat", + "-safe", + "0", + "-i", + &concat_path, + "-c", + "copy", + output, + ]) + .output() + .map_err(|e| format!("Failed to run ffmpeg concat: {e}"))?; + + if !result.status.success() { + let stderr = String::from_utf8_lossy(&result.stderr); + return Err(format!("ffmpeg merge failed: {stderr}")); + } + + for f in &temp_files { + let _ = std::fs::remove_file(f); + } + let _ = std::fs::remove_file(&concat_path); + + on_progress(total, 1.0); Ok(()) } @@ -236,6 +1271,178 @@ mod tests { assert!(webm_args.contains(&"libopus".to_string())); } + #[test] + fn test_build_ffmpeg_args_mux_subs_lossless() { + let clip = Clip { + id: "1".to_string(), + start_time: 10.0, + end_time: 20.0, + label: "Clip 1".to_string(), + color: "#000".to_string(), + }; + let args = build_ffmpeg_args_mux_subs( + &clip, + "input.mp4", + "output.mp4", + &CutMode::Lossless, + "/tmp/subs.vtt", + ); + assert!(args.contains(&"-map".to_string())); + assert!(args.contains(&"1:s".to_string())); + assert!(args.contains(&"mov_text".to_string())); + assert!(args.contains(&"copy".to_string())); + // Input seeking: -ss must come BEFORE -i + let ss_pos = args.iter().position(|a| a == "-ss").unwrap(); + let i_pos = args.iter().position(|a| a == "-i").unwrap(); + assert!(ss_pos < i_pos, "-ss must come before -i for input seeking"); + // Uses -t duration, not -to + assert!(args.contains(&"-t".to_string())); + assert!(!args.contains(&"-to".to_string())); + assert!(args.contains(&"10".to_string())); // duration = 20 - 10 = 10 + } + + #[test] + fn test_build_ffmpeg_args_mux_subs_precise() { + let clip = Clip { + id: "1".to_string(), + start_time: 5.0, + end_time: 15.0, + label: "Clip 1".to_string(), + color: "#000".to_string(), + }; + let mode = CutMode::Precise { + format: "mp4".to_string(), + }; + let args = build_ffmpeg_args_mux_subs( + &clip, + "input.mp4", + "output.mp4", + &mode, + "/tmp/subs.vtt", + ); + assert!(args.contains(&"-map".to_string())); + assert!(args.contains(&"1:s".to_string())); + assert!(args.contains(&"libx264".to_string())); + assert!(args.contains(&"-t".to_string())); + } + + #[test] + fn test_build_ffmpeg_args_burnin_subs_precise() { + let clip = Clip { + id: "1".to_string(), + start_time: 5.0, + end_time: 15.0, + label: "Clip 1".to_string(), + color: "#000".to_string(), + }; + let mode = CutMode::Precise { + format: "mp4".to_string(), + }; + let args = build_ffmpeg_args_burnin_subs( + &clip, + "input.mp4", + "output.mp4", + &mode, + "/tmp/subs.vtt", + None, + ); + // Uses the ass= filter (not subtitles=) for karaoke preservation + let vf_arg = args.iter().find(|a| a.starts_with("ass=")).unwrap(); + assert!(vf_arg.contains("subs.vtt")); + assert!(args.contains(&"libx264".to_string())); + // Output seeking: -ss must come AFTER -i + let i_pos = args.iter().position(|a| a == "-i").unwrap(); + let ss_pos = args.iter().position(|a| a == "-ss").unwrap(); + assert!(ss_pos > i_pos, "-ss must come after -i for output seeking"); + // Uses -t duration, not -to + assert!(args.contains(&"-t".to_string())); + assert!(!args.contains(&"-to".to_string())); + } + + #[test] + fn test_build_ffmpeg_args_burnin_subs_lossless_override() { + let clip = Clip { + id: "1".to_string(), + start_time: 0.0, + end_time: 10.0, + label: "Clip 1".to_string(), + color: "#000".to_string(), + }; + let args = build_ffmpeg_args_burnin_subs( + &clip, + "input.mp4", + "output.mp4", + &CutMode::Lossless, + "/tmp/subs.vtt", + None, + ); + assert!(args.contains(&"libx264".to_string())); + assert!(args.contains(&"aac".to_string())); + let vf_arg = args.iter().find(|a| a.starts_with("ass=")).unwrap(); + assert!(vf_arg.contains("subs.vtt")); + } + + #[test] + fn test_sanitize_vtt_strips_tags() { + let result = strip_vtt_tags("the<00:00:00.599> gym<00:00:00.840> I"); + assert_eq!(result, "the gym I"); + } + + #[test] + fn test_parse_vtt_timestamp_line() { + let result = parse_vtt_timestamp_line("00:01:23.456 --> 00:02:34.567 align:start position:0%"); + assert!(result.is_some()); + let (start, end) = result.unwrap(); + assert!((start - 83.456).abs() < 0.001); + assert!((end - 154.567).abs() < 0.001); + } + + #[test] + fn test_format_vtt_timestamp() { + assert_eq!(format_vtt_timestamp(0.0), "00:00:00.000"); + assert_eq!(format_vtt_timestamp(83.456), "00:01:23.456"); + assert_eq!(format_vtt_timestamp(3661.5), "01:01:01.500"); + } + + #[test] + fn test_trim_and_sanitize_vtt() { + let temp = std::env::temp_dir().join("test-trim-vtt"); + let _ = std::fs::create_dir_all(&temp); + + let vtt_content = "\ +WEBVTT + +00:00:10.000 --> 00:00:15.000 align:start position:0% +the<00:00:10.500> first cue + +00:00:20.000 --> 00:00:25.000 +second cue + +00:00:30.000 --> 00:00:35.000 +third cue +"; + let input_path = temp.join("input.vtt"); + std::fs::write(&input_path, vtt_content).unwrap(); + + // Trim to 18s-28s — should include the second cue only + let result = trim_and_sanitize_vtt( + input_path.to_str().unwrap(), + temp.to_str().unwrap(), + 18.0, + 28.0, + ); + assert!(result.is_ok()); + let trimmed = std::fs::read_to_string(result.unwrap()).unwrap(); + // Second cue shifted: 20-18=2s start, 25-18=7s end + assert!(trimmed.contains("00:00:02.000 --> 00:00:07.000")); + // First and third cues should NOT be present + assert!(!trimmed.contains("first")); + assert!(!trimmed.contains("third")); + assert!(trimmed.contains("second cue")); + + let _ = std::fs::remove_dir_all(&temp); + } + #[test] fn test_expand_tilde_path() { let expanded = expand_tilde_path("~/Videos"); @@ -269,4 +1476,99 @@ mod tests { }; assert_eq!(get_extension("video.webm", &mode), "mkv"); } + + #[test] + fn test_build_page_karaoke_single_line_with_karaoke() { + let line = SpokenLine { + plain_text: "hello world".to_string(), + raw_text: "hello<00:00:01.000> world".to_string(), + start_time: 0.5, + end_time: 1.5, + has_karaoke: true, + is_non_speech: false, + }; + let result = build_page_karaoke_text(&[&line]); + assert!(result.contains("{\\k50}hello")); + assert!(result.contains("{\\k50}world")); + assert!(!result.contains("\\N")); + } + + #[test] + fn test_build_page_karaoke_two_lines_stitched() { + let line1 = SpokenLine { + plain_text: "hello world".to_string(), + raw_text: "hello<00:00:01.000> world".to_string(), + start_time: 0.5, + end_time: 1.5, + has_karaoke: true, + is_non_speech: false, + }; + let line2 = SpokenLine { + plain_text: "foo bar".to_string(), + raw_text: "foo<00:00:02.500> bar".to_string(), + start_time: 2.0, + end_time: 3.0, + has_karaoke: true, + is_non_speech: false, + }; + let result = build_page_karaoke_text(&[&line1, &line2]); + assert!(result.contains("\\N")); + assert!(result.contains("{\\k100}world")); + assert!(result.contains("{\\k50}foo")); + assert!(result.contains("{\\k50}bar")); + } + + #[test] + fn test_build_page_karaoke_no_karaoke_tags() { + let line = SpokenLine { + plain_text: "just plain text".to_string(), + raw_text: "just plain text".to_string(), + start_time: 1.0, + end_time: 3.0, + has_karaoke: false, + is_non_speech: false, + }; + let result = build_page_karaoke_text(&[&line]); + assert!(result.contains("{\\k200}just plain text")); + } + + #[test] + fn test_group_into_pages_even() { + let lines = vec![ + SpokenLine { plain_text: "a".into(), raw_text: "a".into(), start_time: 0.0, end_time: 1.0, has_karaoke: false, is_non_speech: false }, + SpokenLine { plain_text: "b".into(), raw_text: "b".into(), start_time: 1.0, end_time: 2.0, has_karaoke: false, is_non_speech: false }, + SpokenLine { plain_text: "c".into(), raw_text: "c".into(), start_time: 2.0, end_time: 3.0, has_karaoke: false, is_non_speech: false }, + SpokenLine { plain_text: "d".into(), raw_text: "d".into(), start_time: 3.0, end_time: 4.0, has_karaoke: false, is_non_speech: false }, + ]; + let pages = group_into_pages(&lines, 2.0); + assert_eq!(pages.len(), 2); + assert_eq!(pages[0], vec![0, 1]); + assert_eq!(pages[1], vec![2, 3]); + } + + #[test] + fn test_group_into_pages_gap_splits() { + let lines = vec![ + SpokenLine { plain_text: "a".into(), raw_text: "a".into(), start_time: 0.0, end_time: 1.0, has_karaoke: false, is_non_speech: false }, + SpokenLine { plain_text: "b".into(), raw_text: "b".into(), start_time: 5.0, end_time: 6.0, has_karaoke: false, is_non_speech: false }, + ]; + let pages = group_into_pages(&lines, 2.0); + assert_eq!(pages.len(), 2); + assert_eq!(pages[0], vec![0]); + assert_eq!(pages[1], vec![1]); + } + + #[test] + fn test_group_into_pages_odd_count() { + let lines = vec![ + SpokenLine { plain_text: "a".into(), raw_text: "a".into(), start_time: 0.0, end_time: 1.0, has_karaoke: false, is_non_speech: false }, + SpokenLine { plain_text: "b".into(), raw_text: "b".into(), start_time: 1.0, end_time: 2.0, has_karaoke: false, is_non_speech: false }, + SpokenLine { plain_text: "c".into(), raw_text: "c".into(), start_time: 2.0, end_time: 3.0, has_karaoke: false, is_non_speech: false }, + ]; + let pages = group_into_pages(&lines, 2.0); + assert_eq!(pages.len(), 2); + assert_eq!(pages[0], vec![0, 1]); + assert_eq!(pages[1], vec![2]); + } + }