import os import subprocess def transcribe_audio(video_path): """ Transcribe audio from a video file using faster-whisper. Returns transcript in the same format as main.py for compatibility. """ from faster_whisper import WhisperModel print(f"🎙️ Transcribing audio from: {video_path}") # Run on CPU with INT8 quantization for speed model = WhisperModel("base", device="cpu", compute_type="int8") segments, info = model.transcribe(video_path, word_timestamps=True) transcript = { "segments": [], "language": info.language } for segment in segments: seg_data = { "start": segment.start, "end": segment.end, "text": segment.text, "words": [] } if segment.words: for word in segment.words: seg_data["words"].append({ "word": word.word.strip(), "start": word.start, "end": word.end }) transcript["segments"].append(seg_data) print(f"✅ Transcription complete. Language: {info.language}") return transcript def generate_srt_from_video(video_path, output_path, max_chars=20, max_duration=2.0): """ Transcribe a video and generate SRT directly. Used for dubbed videos that don't have a pre-existing transcript. """ transcript = transcribe_audio(video_path) # Get video duration to use as clip_end import cv2 cap = cv2.VideoCapture(video_path) fps = cap.get(cv2.CAP_PROP_FPS) frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) duration = frame_count / fps if fps else 0 cap.release() return generate_srt(transcript, 0, duration, output_path, max_chars, max_duration) import re def load_swears(swears_path): if not swears_path or not os.path.exists(swears_path): return set() with open(swears_path, "r", encoding="utf-8") as f: return {w.strip().lower() for w in f if w.strip()} def censor_word(word): if len(word) <= 2: return "*" * len(word) return word[0] + "*" * (len(word) - 2) + word[-1] def _merge_overlapping(ranges, pad_seconds=0.05): if not ranges: return [] # Apply padding padded_ranges = [] for s, e in ranges: padded_ranges.append((max(0.0, s - pad_seconds), e + pad_seconds)) padded_ranges = sorted(padded_ranges) merged = [padded_ranges[0]] for s, e in padded_ranges[1:]: ps, pe = merged[-1] if s <= pe: merged[-1] = (ps, max(pe, e)) else: merged.append((s, e)) return merged def generate_srt(transcript, clip_start, clip_end, output_path, max_chars=20, max_duration=2.0, swears_path=None, return_mute_ranges=False, speed_factor=1.0): """ Generates an SRT file from the transcript for a specific time range. Groups words into short lines suitable for vertical video. If swears_path is provided, censors swear words in the text and extracts mute ranges. """ swears = load_swears(swears_path) if swears_path else set() raw_mute_ranges = [] words = [] # 1. Extract and flatten words within range for segment in transcript.get('segments', []): for word_info in segment.get('words', []): # Check overlap if word_info['end'] > clip_start and word_info['start'] < clip_end: word_copy = dict(word_info) word_text = word_copy['word'] cleaned_word = re.sub(r'[^\w]', '', word_text.lower()) if cleaned_word in swears: # Record absolute mute range relative to clip start, scaled by speed_factor rel_start = (max(0.0, word_copy['start'] - clip_start)) / speed_factor rel_end = (max(0.0, word_copy['end'] - clip_start)) / speed_factor raw_mute_ranges.append((rel_start, rel_end)) # Censor word word_copy['word'] = censor_word(word_text) words.append(word_copy) if not words: if return_mute_ranges: return False, [] return False import json srt_content = "" index = 1 blocks = [] current_block = [] block_start = None for i, word in enumerate(words): # Adjust times relative to clip, scaled by speed_factor start = (max(0, word['start'] - clip_start)) / speed_factor end = (max(0, word['end'] - clip_start)) / speed_factor # Word info copy with relative times w_rel = { 'word': word['word'], 'start': start, 'end': end } if not current_block: current_block.append(w_rel) block_start = start else: current_text_len = sum(len(w['word']) + 1 for w in current_block) duration = end - block_start if current_text_len + len(w_rel['word']) > max_chars or duration > max_duration: blocks.append(current_block) block_end = current_block[-1]['end'] text = " ".join([w['word'] for w in current_block]).strip() srt_content += format_srt_block(index, block_start, block_end, text) index += 1 current_block = [w_rel] block_start = start else: current_block.append(w_rel) # Final block if current_block: blocks.append(current_block) block_end = current_block[-1]['end'] text = " ".join([w['word'] for w in current_block]).strip() srt_content += format_srt_block(index, block_start, block_end, text) with open(output_path, 'w', encoding='utf-8') as f: f.write(srt_content) # Write words json file words_json_path = os.path.splitext(output_path)[0] + ".words.json" with open(words_json_path, 'w', encoding='utf-8') as f: json.dump(blocks, f, indent=2) mute_ranges = _merge_overlapping(raw_mute_ranges) if return_mute_ranges: return True, mute_ranges return True def format_srt_block(index, start, end, text): def format_time(seconds): hours = int(seconds // 3600) minutes = int((seconds % 3600) // 60) secs = int(seconds % 60) millis = int((seconds - int(seconds)) * 1000) return f"{hours:02d}:{minutes:02d}:{secs:02d},{millis:03d}" return f"{index}\n{format_time(start)} --> {format_time(end)}\n{text}\n\n" def hex_to_ass_color(hex_color, opacity=1.0): """Convert #RRGGBB to ASS &HAABBGGRR format. opacity: 0.0=transparent, 1.0=opaque""" hex_color = hex_color.lstrip('#') if len(hex_color) != 6: hex_color = "FFFFFF" r = int(hex_color[0:2], 16) g = int(hex_color[2:4], 16) b = int(hex_color[4:6], 16) alpha = round((1.0 - opacity) * 255) return f"&H{alpha:02X}{b:02X}{g:02X}{r:02X}" def burn_subtitles(video_path, srt_path, output_path, alignment=2, fontsize=16, font_name="Verdana", font_color="#FFFFFF", border_color="#000000", border_width=2, bg_color="#000000", bg_opacity=0.0, fonts_dir=None): """ Burns subtitles into the video using FFmpeg by converting SRT to a styled ASS file. Supports outline mode, box mode, and active word-level highlight box if a words JSON is available. """ import pysubs2 import os import json # 1. Load subtitles subs = pysubs2.load(srt_path, encoding="utf-8") # 2. Configure project resolutions to match vertical video standards subs.info["PlayResX"] = "1080" subs.info["PlayResY"] = "1920" subs.info["ScaledBorderAndShadow"] = "yes" # 3. Configure Default style style = subs.styles["Default"] style.fontname = font_name style.fontsize = int(fontsize * 6.66) style.bold = True # Position mapping ass_alignment = 2 align_lower = str(alignment).lower() if align_lower == 'top': ass_alignment = 8 elif align_lower == 'middle': ass_alignment = 5 elif align_lower == 'bottom': ass_alignment = 2 style.alignment = ass_alignment style.marginv = 200 # Convert colors def hex_to_rgb(hex_str): hex_str = hex_str.lstrip('#') if len(hex_str) != 6: hex_str = "FFFFFF" return int(hex_str[0:2], 16), int(hex_str[2:4], 16), int(hex_str[4:6], 16) pr, pg, pb = hex_to_rgb(font_color) or_, og, ob = hex_to_rgb(border_color) br, bg, bb = hex_to_rgb(bg_color) # Check if we have word-level highlights words_json_path = os.path.splitext(srt_path)[0] + ".words.json" has_words = os.path.exists(words_json_path) if has_words: # Highlight Mode: # Default style is plain white text with black outline (no background box) style.borderstyle = 1 style.primarycolor = pysubs2.Color(pr, pg, pb, 0) style.outlinecolor = pysubs2.Color(or_, og, ob, 0) style.backcolor = pysubs2.Color(0, 0, 0, 255) # transparent shadow style.outline = border_width style.shadow = 0 # Define Highlight style (pink box, black outline inside the box) highlight_style = style.copy() highlight_style.borderstyle = 3 # Opaque box highlight_style.outlinecolor = pysubs2.Color(br, bg, bb, round((1.0 - bg_opacity) * 255)) highlight_style.backcolor = pysubs2.Color(or_, og, ob, 0) # Text outline inside the box highlight_style.outline = border_width # Box padding subs.styles["Highlight"] = highlight_style # Load words and rebuild events try: with open(words_json_path, 'r', encoding='utf-8') as f: blocks = json.load(f) subs.events.clear() for block in blocks: if not block: continue block_start = block[0]['start'] block_end = block[-1]['end'] n_words = len(block) for idx in range(n_words): if idx == 0: event_start = block_start else: event_start = block[idx]['start'] if idx == n_words - 1: event_end = block_end else: event_end = block[idx+1]['start'] text_parts = [] for j, w in enumerate(block): word_str = w['word'] if j == idx: text_parts.append(f"{{\\rHighlight}}{word_str}{{\\r}}") else: text_parts.append(word_str) event_text = " ".join(text_parts) start_ms = int(event_start * 1000) end_ms = int(event_end * 1000) subs.events.append(pysubs2.SSAEvent(start=start_ms, end=end_ms, text=event_text)) except Exception as e: print(f"⚠️ Failed to parse words JSON: {e}. Falling back to standard ASS.") has_words = False if not has_words: # Fallback to standard full-block style (original style behavior) style.primarycolor = pysubs2.Color(pr, pg, pb, 0) if bg_opacity > 0: style.borderstyle = 3 # Opaque box style.outlinecolor = pysubs2.Color(br, bg, bb, round((1.0 - bg_opacity) * 255)) style.backcolor = pysubs2.Color(or_, og, ob, 0) style.outline = border_width style.shadow = 0 else: style.borderstyle = 1 style.outlinecolor = pysubs2.Color(or_, og, ob, 0) style.backcolor = pysubs2.Color(0, 0, 0, 255) style.outline = border_width style.shadow = 0 # Save styled ASS file ass_path = os.path.splitext(srt_path)[0] + ".ass" subs.save(ass_path, format_="ass") # 4. Burn using FFmpeg safe_ass_path = ass_path.replace('\\', '/').replace(':', '\\:') subtitles_filter = f"subtitles='{safe_ass_path}'" if fonts_dir: safe_fonts_dir = fonts_dir.replace('\\', '/').replace(':', '\\:') subtitles_filter += f":fontsdir='{safe_fonts_dir}'" cmd = [ 'ffmpeg', '-y', '-i', video_path, '-vf', subtitles_filter, '-c:a', 'copy', '-c:v', 'libx264', '-preset', 'fast', '-crf', '23', output_path ] print(f"🎬 Burning subtitles using ASS file: {' '.join(cmd)}") result = subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE) # Clean up files if os.path.exists(ass_path): os.remove(ass_path) if os.path.exists(words_json_path): os.remove(words_json_path) if result.returncode != 0: print(f"❌ FFmpeg Subtitle Error: {result.stderr.decode()}") raise Exception(f"FFmpeg failed: {result.stderr.decode()}") return True