feat: implement vertical crop, local LLM deepseek-v4-flash, multi-threading, word-level pink box highlight subtitles, and Nextcloud scan sync
This commit is contained in:
@@ -0,0 +1,374 @@
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
|
||||
def transcribe_audio(video_path):
|
||||
"""
|
||||
Transcribe audio from a video file using faster-whisper.
|
||||
Returns transcript in the same format as main.py for compatibility.
|
||||
"""
|
||||
from faster_whisper import WhisperModel
|
||||
|
||||
print(f"🎙️ Transcribing audio from: {video_path}")
|
||||
|
||||
# Run on CPU with INT8 quantization for speed
|
||||
model = WhisperModel("base", device="cpu", compute_type="int8")
|
||||
|
||||
segments, info = model.transcribe(video_path, word_timestamps=True)
|
||||
|
||||
transcript = {
|
||||
"segments": [],
|
||||
"language": info.language
|
||||
}
|
||||
|
||||
for segment in segments:
|
||||
seg_data = {
|
||||
"start": segment.start,
|
||||
"end": segment.end,
|
||||
"text": segment.text,
|
||||
"words": []
|
||||
}
|
||||
if segment.words:
|
||||
for word in segment.words:
|
||||
seg_data["words"].append({
|
||||
"word": word.word.strip(),
|
||||
"start": word.start,
|
||||
"end": word.end
|
||||
})
|
||||
transcript["segments"].append(seg_data)
|
||||
|
||||
print(f"✅ Transcription complete. Language: {info.language}")
|
||||
return transcript
|
||||
|
||||
|
||||
def generate_srt_from_video(video_path, output_path, max_chars=20, max_duration=2.0):
|
||||
"""
|
||||
Transcribe a video and generate SRT directly.
|
||||
Used for dubbed videos that don't have a pre-existing transcript.
|
||||
"""
|
||||
transcript = transcribe_audio(video_path)
|
||||
|
||||
# Get video duration to use as clip_end
|
||||
import cv2
|
||||
cap = cv2.VideoCapture(video_path)
|
||||
fps = cap.get(cv2.CAP_PROP_FPS)
|
||||
frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
|
||||
duration = frame_count / fps if fps else 0
|
||||
cap.release()
|
||||
|
||||
return generate_srt(transcript, 0, duration, output_path, max_chars, max_duration)
|
||||
|
||||
|
||||
import re
|
||||
|
||||
def load_swears(swears_path):
|
||||
if not swears_path or not os.path.exists(swears_path):
|
||||
return set()
|
||||
with open(swears_path, "r", encoding="utf-8") as f:
|
||||
return {w.strip().lower() for w in f if w.strip()}
|
||||
|
||||
def censor_word(word):
|
||||
if len(word) <= 2:
|
||||
return "*" * len(word)
|
||||
return word[0] + "*" * (len(word) - 2) + word[-1]
|
||||
|
||||
def _merge_overlapping(ranges, pad_seconds=0.05):
|
||||
if not ranges:
|
||||
return []
|
||||
# Apply padding
|
||||
padded_ranges = []
|
||||
for s, e in ranges:
|
||||
padded_ranges.append((max(0.0, s - pad_seconds), e + pad_seconds))
|
||||
|
||||
padded_ranges = sorted(padded_ranges)
|
||||
merged = [padded_ranges[0]]
|
||||
for s, e in padded_ranges[1:]:
|
||||
ps, pe = merged[-1]
|
||||
if s <= pe:
|
||||
merged[-1] = (ps, max(pe, e))
|
||||
else:
|
||||
merged.append((s, e))
|
||||
return merged
|
||||
|
||||
def generate_srt(transcript, clip_start, clip_end, output_path, max_chars=20, max_duration=2.0, swears_path=None, return_mute_ranges=False):
|
||||
"""
|
||||
Generates an SRT file from the transcript for a specific time range.
|
||||
Groups words into short lines suitable for vertical video.
|
||||
If swears_path is provided, censors swear words in the text and extracts mute ranges.
|
||||
"""
|
||||
swears = load_swears(swears_path) if swears_path else set()
|
||||
raw_mute_ranges = []
|
||||
|
||||
words = []
|
||||
# 1. Extract and flatten words within range
|
||||
for segment in transcript.get('segments', []):
|
||||
for word_info in segment.get('words', []):
|
||||
# Check overlap
|
||||
if word_info['end'] > clip_start and word_info['start'] < clip_end:
|
||||
word_copy = dict(word_info)
|
||||
word_text = word_copy['word']
|
||||
cleaned_word = re.sub(r'[^\w]', '', word_text.lower())
|
||||
|
||||
if cleaned_word in swears:
|
||||
# Record absolute mute range relative to clip start
|
||||
rel_start = max(0.0, word_copy['start'] - clip_start)
|
||||
rel_end = max(0.0, word_copy['end'] - clip_start)
|
||||
raw_mute_ranges.append((rel_start, rel_end))
|
||||
# Censor word
|
||||
word_copy['word'] = censor_word(word_text)
|
||||
|
||||
words.append(word_copy)
|
||||
|
||||
if not words:
|
||||
if return_mute_ranges:
|
||||
return False, []
|
||||
return False
|
||||
|
||||
import json
|
||||
|
||||
srt_content = ""
|
||||
index = 1
|
||||
|
||||
blocks = []
|
||||
current_block = []
|
||||
block_start = None
|
||||
|
||||
for i, word in enumerate(words):
|
||||
# Adjust times relative to clip
|
||||
start = max(0, word['start'] - clip_start)
|
||||
end = max(0, word['end'] - clip_start)
|
||||
|
||||
# Word info copy with relative times
|
||||
w_rel = {
|
||||
'word': word['word'],
|
||||
'start': start,
|
||||
'end': end
|
||||
}
|
||||
|
||||
if not current_block:
|
||||
current_block.append(w_rel)
|
||||
block_start = start
|
||||
else:
|
||||
current_text_len = sum(len(w['word']) + 1 for w in current_block)
|
||||
duration = end - block_start
|
||||
|
||||
if current_text_len + len(w_rel['word']) > max_chars or duration > max_duration:
|
||||
blocks.append(current_block)
|
||||
block_end = current_block[-1]['end']
|
||||
text = " ".join([w['word'] for w in current_block]).strip()
|
||||
srt_content += format_srt_block(index, block_start, block_end, text)
|
||||
index += 1
|
||||
|
||||
current_block = [w_rel]
|
||||
block_start = start
|
||||
else:
|
||||
current_block.append(w_rel)
|
||||
|
||||
# Final block
|
||||
if current_block:
|
||||
blocks.append(current_block)
|
||||
block_end = current_block[-1]['end']
|
||||
text = " ".join([w['word'] for w in current_block]).strip()
|
||||
srt_content += format_srt_block(index, block_start, block_end, text)
|
||||
|
||||
with open(output_path, 'w', encoding='utf-8') as f:
|
||||
f.write(srt_content)
|
||||
|
||||
# Write words json file
|
||||
words_json_path = os.path.splitext(output_path)[0] + ".words.json"
|
||||
with open(words_json_path, 'w', encoding='utf-8') as f:
|
||||
json.dump(blocks, f, indent=2)
|
||||
|
||||
mute_ranges = _merge_overlapping(raw_mute_ranges)
|
||||
|
||||
if return_mute_ranges:
|
||||
return True, mute_ranges
|
||||
return True
|
||||
|
||||
def format_srt_block(index, start, end, text):
|
||||
def format_time(seconds):
|
||||
hours = int(seconds // 3600)
|
||||
minutes = int((seconds % 3600) // 60)
|
||||
secs = int(seconds % 60)
|
||||
millis = int((seconds - int(seconds)) * 1000)
|
||||
return f"{hours:02d}:{minutes:02d}:{secs:02d},{millis:03d}"
|
||||
|
||||
return f"{index}\n{format_time(start)} --> {format_time(end)}\n{text}\n\n"
|
||||
|
||||
def hex_to_ass_color(hex_color, opacity=1.0):
|
||||
"""Convert #RRGGBB to ASS &HAABBGGRR format. opacity: 0.0=transparent, 1.0=opaque"""
|
||||
hex_color = hex_color.lstrip('#')
|
||||
if len(hex_color) != 6:
|
||||
hex_color = "FFFFFF"
|
||||
r = int(hex_color[0:2], 16)
|
||||
g = int(hex_color[2:4], 16)
|
||||
b = int(hex_color[4:6], 16)
|
||||
alpha = round((1.0 - opacity) * 255)
|
||||
return f"&H{alpha:02X}{b:02X}{g:02X}{r:02X}"
|
||||
|
||||
|
||||
def burn_subtitles(video_path, srt_path, output_path, alignment=2, fontsize=16,
|
||||
font_name="Verdana", font_color="#FFFFFF",
|
||||
border_color="#000000", border_width=2,
|
||||
bg_color="#000000", bg_opacity=0.0, fonts_dir=None):
|
||||
"""
|
||||
Burns subtitles into the video using FFmpeg by converting SRT to a styled ASS file.
|
||||
Supports outline mode, box mode, and active word-level highlight box if a words JSON is available.
|
||||
"""
|
||||
import pysubs2
|
||||
import os
|
||||
import json
|
||||
|
||||
# 1. Load subtitles
|
||||
subs = pysubs2.load(srt_path, encoding="utf-8")
|
||||
|
||||
# 2. Configure project resolutions to match vertical video standards
|
||||
subs.info["PlayResX"] = "1080"
|
||||
subs.info["PlayResY"] = "1920"
|
||||
subs.info["ScaledBorderAndShadow"] = "yes"
|
||||
|
||||
# 3. Configure Default style
|
||||
style = subs.styles["Default"]
|
||||
style.fontname = font_name
|
||||
style.fontsize = int(fontsize * 6.66)
|
||||
style.bold = True
|
||||
|
||||
# Position mapping
|
||||
ass_alignment = 2
|
||||
align_lower = str(alignment).lower()
|
||||
if align_lower == 'top':
|
||||
ass_alignment = 8
|
||||
elif align_lower == 'middle':
|
||||
ass_alignment = 5
|
||||
elif align_lower == 'bottom':
|
||||
ass_alignment = 2
|
||||
|
||||
style.alignment = ass_alignment
|
||||
style.marginv = 200
|
||||
|
||||
# Convert colors
|
||||
def hex_to_rgb(hex_str):
|
||||
hex_str = hex_str.lstrip('#')
|
||||
if len(hex_str) != 6:
|
||||
hex_str = "FFFFFF"
|
||||
return int(hex_str[0:2], 16), int(hex_str[2:4], 16), int(hex_str[4:6], 16)
|
||||
|
||||
pr, pg, pb = hex_to_rgb(font_color)
|
||||
or_, og, ob = hex_to_rgb(border_color)
|
||||
br, bg, bb = hex_to_rgb(bg_color)
|
||||
|
||||
# Check if we have word-level highlights
|
||||
words_json_path = os.path.splitext(srt_path)[0] + ".words.json"
|
||||
has_words = os.path.exists(words_json_path)
|
||||
|
||||
if has_words:
|
||||
# Highlight Mode:
|
||||
# Default style is plain white text with black outline (no background box)
|
||||
style.borderstyle = 1
|
||||
style.primarycolor = pysubs2.Color(pr, pg, pb, 0)
|
||||
style.outlinecolor = pysubs2.Color(or_, og, ob, 0)
|
||||
style.backcolor = pysubs2.Color(0, 0, 0, 255) # transparent shadow
|
||||
style.outline = border_width
|
||||
style.shadow = 0
|
||||
|
||||
# Define Highlight style (pink box, black outline inside the box)
|
||||
highlight_style = style.copy()
|
||||
highlight_style.borderstyle = 3 # Opaque box
|
||||
highlight_style.outlinecolor = pysubs2.Color(br, bg, bb, round((1.0 - bg_opacity) * 255))
|
||||
highlight_style.backcolor = pysubs2.Color(or_, og, ob, 0) # Text outline inside the box
|
||||
highlight_style.outline = border_width # Box padding
|
||||
|
||||
subs.styles["Highlight"] = highlight_style
|
||||
|
||||
# Load words and rebuild events
|
||||
try:
|
||||
with open(words_json_path, 'r', encoding='utf-8') as f:
|
||||
blocks = json.load(f)
|
||||
|
||||
subs.events.clear()
|
||||
for block in blocks:
|
||||
if not block:
|
||||
continue
|
||||
block_start = block[0]['start']
|
||||
block_end = block[-1]['end']
|
||||
n_words = len(block)
|
||||
|
||||
for idx in range(n_words):
|
||||
if idx == 0:
|
||||
event_start = block_start
|
||||
else:
|
||||
event_start = block[idx]['start']
|
||||
|
||||
if idx == n_words - 1:
|
||||
event_end = block_end
|
||||
else:
|
||||
event_end = block[idx+1]['start']
|
||||
|
||||
text_parts = []
|
||||
for j, w in enumerate(block):
|
||||
word_str = w['word']
|
||||
if j == idx:
|
||||
text_parts.append(f"{{\\rHighlight}}{word_str}{{\\r}}")
|
||||
else:
|
||||
text_parts.append(word_str)
|
||||
event_text = " ".join(text_parts)
|
||||
|
||||
start_ms = int(event_start * 1000)
|
||||
end_ms = int(event_end * 1000)
|
||||
|
||||
subs.events.append(pysubs2.SSAEvent(start=start_ms, end=end_ms, text=event_text))
|
||||
except Exception as e:
|
||||
print(f"⚠️ Failed to parse words JSON: {e}. Falling back to standard ASS.")
|
||||
has_words = False
|
||||
|
||||
if not has_words:
|
||||
# Fallback to standard full-block style (original style behavior)
|
||||
style.primarycolor = pysubs2.Color(pr, pg, pb, 0)
|
||||
if bg_opacity > 0:
|
||||
style.borderstyle = 3 # Opaque box
|
||||
style.outlinecolor = pysubs2.Color(br, bg, bb, round((1.0 - bg_opacity) * 255))
|
||||
style.backcolor = pysubs2.Color(or_, og, ob, 0)
|
||||
style.outline = border_width
|
||||
style.shadow = 0
|
||||
else:
|
||||
style.borderstyle = 1
|
||||
style.outlinecolor = pysubs2.Color(or_, og, ob, 0)
|
||||
style.backcolor = pysubs2.Color(0, 0, 0, 255)
|
||||
style.outline = border_width
|
||||
style.shadow = 0
|
||||
|
||||
# Save styled ASS file
|
||||
ass_path = os.path.splitext(srt_path)[0] + ".ass"
|
||||
subs.save(ass_path, format_="ass")
|
||||
|
||||
# 4. Burn using FFmpeg
|
||||
safe_ass_path = ass_path.replace('\\', '/').replace(':', '\\:')
|
||||
subtitles_filter = f"subtitles='{safe_ass_path}'"
|
||||
if fonts_dir:
|
||||
safe_fonts_dir = fonts_dir.replace('\\', '/').replace(':', '\\:')
|
||||
subtitles_filter += f":fontsdir='{safe_fonts_dir}'"
|
||||
|
||||
cmd = [
|
||||
'ffmpeg', '-y',
|
||||
'-i', video_path,
|
||||
'-vf', subtitles_filter,
|
||||
'-c:a', 'copy',
|
||||
'-c:v', 'libx264', '-preset', 'fast', '-crf', '23',
|
||||
output_path
|
||||
]
|
||||
|
||||
print(f"🎬 Burning subtitles using ASS file: {' '.join(cmd)}")
|
||||
result = subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE)
|
||||
|
||||
# Clean up files
|
||||
if os.path.exists(ass_path):
|
||||
os.remove(ass_path)
|
||||
if os.path.exists(words_json_path):
|
||||
os.remove(words_json_path)
|
||||
|
||||
if result.returncode != 0:
|
||||
print(f"❌ FFmpeg Subtitle Error: {result.stderr.decode()}")
|
||||
raise Exception(f"FFmpeg failed: {result.stderr.decode()}")
|
||||
|
||||
return True
|
||||
|
||||
Reference in New Issue
Block a user