#!/usr/bin/env python3
"""
SRT subtitle tool - three modes:
1. GENERATE - Whisper transcribes the video and creates a perfectly-synced SRT.
2. SYNC - ffsubsync syncs an existing SRT, Whisper cross-checks the result.
3. BATCH - sync all video+SRT pairs in the current directory.
GPU is used automatically if CUDA (NVIDIA) or MPS (Apple Silicon) is detected.
openai-whisper and ffsubsync are installed automatically if missing.
Flags:
--translate Mode 2 outputs English regardless of source language
--lang CODE Source language hint (e.g. fr, id, es) — speeds up detection
--lang-auto Auto-detect language (default when --translate is used)
--extract-all FILE Non-interactive: extract all subtitle tracks and exit
Requirements:
Python 3, ffmpeg in PATH.
"""
import os, sys, re, subprocess, struct, difflib, urllib.request, urllib.parse, json, glob
from statistics import median
# ---------- .env loader -------------------------------------------------------
def _load_env():
"""Parse KEY=value lines from .env in cwd or script directory."""
import pathlib
for candidate in [pathlib.Path('.env'),
pathlib.Path(__file__).resolve().parent / '.env']:
try:
for line in candidate.read_text().splitlines():
line = line.strip()
if not line or line.startswith('#') or '=' not in line:
continue
k, _, v = line.partition('=')
k = k.strip()
v = v.strip().strip('"').strip("'")
if k and k not in os.environ:
os.environ[k] = v
except FileNotFoundError:
pass
_load_env()
# =============================================================================
# TMDB API key — set here OR put TMDB_API_KEY=your_key in a .env file
# Get a free key at https://www.themoviedb.org/settings/api
TMDB_API_KEY = os.environ.get('TMDB_API_KEY', '')
# =============================================================================
# ---------- Path setup -------------------------------------------------------
def _extend_path():
import site, pathlib
candidates = []
try:
candidates.append(site.getusersitepackages())
except Exception:
pass
home = str(pathlib.Path.home())
candidates += glob.glob(
os.path.join(home, '.local', 'lib', 'python*', 'site-packages')
)
for p in candidates:
if p and os.path.isdir(p) and p not in sys.path:
sys.path.insert(0, p)
_extend_path()
# ---------- ffsubsync finder -------------------------------------------------
def _find_ffsubsync():
import shutil, pathlib
found = shutil.which('ffsubsync')
if found:
return found
local_bin = os.path.join(str(pathlib.Path.home()), '.local', 'bin', 'ffsubsync')
if os.path.isfile(local_bin):
return local_bin
return None
# ---------- Optional dependency detection ------------------------------------
try:
import whisper as _whisper
WHISPER_AVAILABLE = True
except ImportError:
_whisper = None
WHISPER_AVAILABLE = False
FFSUBSYNC_AVAILABLE = _find_ffsubsync() is not None
TS_RE = re.compile(
r'(\d{1,2}:\d{2}:\d{2}[,\.]\d{1,3})\s*-->\s*(\d{1,2}:\d{2}:\d{2}[,\.]\d{1,3})'
)
VIDEO_EXTS = ('.mp4','.mkv','.mov','.avi','.ts','.m2ts','.webm','.flv','.wmv','.mpg','.mpeg')
# ---------- Startup diagnostic -----------------------------------------------
def _check_deps():
print("--- dependency check ---")
try:
import whisper as _w, inspect
print(f" whisper : found at {os.path.dirname(inspect.getfile(_w))}")
except ImportError:
print(" whisper : NOT found")
exe = _find_ffsubsync()
print(f" ffsubsync : {'found at ' + exe if exe else 'NOT found'}")
try:
r = subprocess.run(['ffmpeg', '-version'], stdout=subprocess.PIPE, stderr=subprocess.PIPE)
line = r.stdout.decode(errors='ignore').splitlines()[0]
print(f" ffmpeg : {line}")
except FileNotFoundError:
print(" ffmpeg : NOT found - required!")
local_paths = [p for p in sys.path if 'local' in p or 'site' in p]
if local_paths:
print(" sys.path (local/site entries):")
for p in local_paths:
print(f" {p}")
print("------------------------")
_check_deps()
# ---------- Tunable constants ------------------------------------------------
WHISPER_MODEL = "large-v3-turbo"
WHISPER_LANGUAGE = "en"
WHISPER_TASK = "transcribe" # or "translate" (→ English output)
WHISPER_MODELS = {
"1": ("tiny", "~39 MB - very fast, low accuracy"),
"2": ("base", "~74 MB - fast, basic accuracy"),
"3": ("small", "~244 MB - good for simple audio"),
"4": ("medium", "~769 MB - better accuracy, slower"),
"5": ("large-v3-turbo", "~809 MB - best speed/accuracy balance (recommended)"),
"6": ("large-v3", "~1.5 GB - highest accuracy, slowest"),
}
WHISPER_MODEL_SIZES = {
"tiny": "39 MB", "base": "74 MB", "small": "244 MB",
"medium": "769 MB", "large-v3-turbo": "809 MB", "large-v3": "1.5 GB",
}
WHISPER_PROMPT = (
"Transcript with proper punctuation, capitalization, and grammar. "
"Mark all sung lyrics and songs with ♪ symbols at the start and end. "
"Use italics tags for off-screen or narrator dialogue."
)
START_SKIP_S = 0
ANALYZE_S = 600 # 10 minutes of audio for alignment
MIN_WORD_LEN = 4
OFFSET_AGREE_THRESHOLD = 1.5 # seconds - warn if ffsubsync and Whisper differ more than this
STOP_WORDS = {
'the','and','you','that','was','for','are','with','his','they','this',
'have','from','not','but','had','her','she','him','been','has','its',
'who','did','get','may','now','can','our','out','all','yes','no',
'what','just','will','your','when','them','than','then','some','into',
'said','more','also','very','here','well','like','even','back','much',
}
MAX_OFFSET_S = 90.0
RESOLUTION_S = 0.1
RESAMPLE_HZ = 100
SPEECH_LO = 300
SPEECH_HI = 3400
CHUNK_SIZE = max(1, int(RESAMPLE_HZ * RESOLUTION_S))
_NOISE_RE = re.compile(
r'\b(720p|1080p|2160p|4k|uhd|webrip|web|bluray|bdrip|dvdrip|hdtv|dl'
r'|x264|x265|hevc|avc|h264|h265|aac|dts|ac3|nf|amzn|hulu|dsnp|atvp'
r'|hmax|pcok|repack|proper|extended|theatrical|directors?cut|remux'
r'|episode|episodes?)\b',
re.IGNORECASE
)
_SXXEXX_RE = re.compile(r'\bS(\d{1,2})E(\d{1,2})\b', re.IGNORECASE)
_SEASON_DIR_RE = re.compile(r'^[Ss]eason[\s._-]*\d+$')
_BRACKET_RE = re.compile(r'^\s*\[[^\]]*\]\s*') # leading [SubGroup] tags
_TEXT_SUB_CODECS = {'subrip', 'srt', 'ass', 'ssa', 'mov_text',
'webvtt', 'microdvd', 'text', 'dvb_teletext'}
_IMAGE_SUB_CODECS = {'dvd_subtitle', 'hdmv_pgs_subtitle',
'dvb_subtitle', 'dvbsub', 'pgssub', 'xsub'}
# ---------- Auto-install helpers ---------------------------------------------
def _find_pip():
for cmd in (['pip3'], ['pip']):
try:
if subprocess.run(cmd + ['--version'],
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL).returncode == 0:
return cmd
except FileNotFoundError:
pass
for py in [sys.executable, 'python3', 'python']:
try:
if subprocess.run([py, '-m', 'pip', '--version'],
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL).returncode == 0:
return [py, '-m', 'pip']
except FileNotFoundError:
pass
# Try bootstrapping pip via ensurepip
try:
if subprocess.run([sys.executable, '-m', 'ensurepip', '--upgrade'],
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL).returncode == 0:
if subprocess.run([sys.executable, '-m', 'pip', '--version'],
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL).returncode == 0:
return [sys.executable, '-m', 'pip']
except Exception:
pass
# Last resort: apt-get
print(" pip not found - attempting: sudo apt-get install python3-pip ...")
try:
if subprocess.run(['sudo', 'apt-get', 'install', '-y', 'python3-pip'],
timeout=120).returncode == 0:
for cmd in (['pip3'], [sys.executable, '-m', 'pip']):
try:
if subprocess.run(cmd + ['--version'],
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL).returncode == 0:
return cmd
except FileNotFoundError:
pass
except Exception:
pass
return None
def _pip_install(package):
pip = _find_pip()
if pip is None:
print(f" Cannot find pip. Try manually: pip3 install {package}")
return False
for flags in [[], ['--user']]:
if subprocess.run(pip + ['install'] + flags + [package]).returncode == 0:
_extend_path()
return True
print(" Standard and --user installs failed.")
if input(" Try --break-system-packages? [y/N]: ").strip().lower() == 'y':
if subprocess.run(pip + ['install', '--break-system-packages',
package]).returncode == 0:
_extend_path()
return True
return False
def ensure_whisper():
global _whisper, WHISPER_AVAILABLE
if WHISPER_AVAILABLE:
return True
print("\nopenai-whisper is not installed.")
if input("Install it now? [y/N]: ").strip().lower() != 'y':
print("Skipping - will fall back to audio energy method.")
return False
print("Installing openai-whisper...")
if not _pip_install('openai-whisper'):
print("Installation failed.")
return False
import importlib
importlib.invalidate_caches()
try:
import whisper as _w
_whisper = _w
WHISPER_AVAILABLE = True
print("Installed successfully.\n")
return True
except ImportError:
print("Installed but import failed - try restarting the script.")
return False
def ensure_ffsubsync():
global FFSUBSYNC_AVAILABLE
if FFSUBSYNC_AVAILABLE:
return True
print("\nffsubsync is not installed (recommended for syncing existing SRTs).")
if input("Install it now? [y/N]: ").strip().lower() != 'y':
return False
print("Installing ffsubsync...")
if not _pip_install('ffsubsync'):
print("Installation failed.")
return False
import importlib
importlib.invalidate_caches()
if _find_ffsubsync():
FFSUBSYNC_AVAILABLE = True
print("ffsubsync installed successfully.")
return True
print("Installed but ffsubsync not found - try restarting the script.")
return False
def ensure_easyocr():
try:
import easyocr # noqa: F401
return True
except ImportError:
pass
print("\neasyocr not installed (needed to scan video frames for a title card).")
if input("Install it now? (~200 MB package, ~170 MB model download on first use) [y/N]: ").strip().lower() != 'y':
return False
print("Installing easyocr...")
if not _pip_install('easyocr'):
print("Installation failed.")
return False
import importlib
importlib.invalidate_caches()
try:
import easyocr # noqa: F401
return True
except ImportError:
print("Installed but import failed - try restarting the script.")
return False
def ensure_ccextractor():
"""Return ccextractor command, or None if unavailable."""
for cmd in ['ccextractor', 'ccextractorwin', 'ccx']:
try:
if subprocess.run([cmd, '--version'],
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL).returncode in (0, 1):
return cmd
except FileNotFoundError:
pass
print("\nccextractor not found (needed for CC and some DVD subtitles).")
if input("Try to install via apt-get? [y/N]: ").strip().lower() != 'y':
print(" Install manually: https://ccextractor.org")
return None
try:
if subprocess.run(['sudo', 'apt-get', 'install', '-y', 'ccextractor'],
timeout=120).returncode == 0:
return 'ccextractor'
except Exception:
pass
print(" apt-get failed. Install manually: https://ccextractor.org")
return None
def ensure_pgsreader():
try:
import pgsreader # noqa: F401
return True
except ImportError:
pass
print("\npgsreader not installed (needed for Blu-ray PGS subtitles).")
if input("Install it now? [y/N]: ").strip().lower() != 'y':
return False
if not _pip_install('pgsreader'):
return False
import importlib
importlib.invalidate_caches()
try:
import pgsreader # noqa: F401
return True
except ImportError:
print("Installed but import failed - try restarting the script.")
return False
def ensure_mkvtoolnix():
"""Return True if mkvmerge is available, offering to install if not."""
import shutil, platform
if shutil.which('mkvmerge'):
return True
print("\nmkvmerge not found — needed to embed subtitles into MKV files.")
system = platform.system()
if system == 'Darwin':
if input(" Try to install via brew? [y/N]: ").strip().lower() == 'y':
try:
if subprocess.run(['brew', 'install', 'mkvtoolnix'],
timeout=300).returncode == 0:
return bool(shutil.which('mkvmerge'))
except Exception:
pass
print(" Install manually: brew install mkvtoolnix")
else:
if input(" Try to install via apt-get? [y/N]: ").strip().lower() == 'y':
try:
if subprocess.run(['sudo', 'apt-get', 'install', '-y', 'mkvtoolnix'],
timeout=120).returncode == 0:
return bool(shutil.which('mkvmerge'))
except Exception:
pass
print(" Install manually:")
print(" Debian/Ubuntu : sudo apt install mkvtoolnix")
print(" Arch : sudo pacman -S mkvtoolnix-cli")
print(" Other : https://mkvtoolnix.download/")
return False
def ensure_vobsub2srt():
"""Return True if vobsub2srt is available, offering to install if not."""
import shutil
if shutil.which('vobsub2srt'):
return True
print("\nvobsub2srt not found — needed for DVD VOB subtitle OCR to SRT.")
if input(" Try to install via apt-get? [y/N]: ").strip().lower() == 'y':
try:
if subprocess.run(['sudo', 'apt-get', 'install', '-y', 'vobsub2srt'],
timeout=120).returncode == 0:
return bool(shutil.which('vobsub2srt'))
except Exception:
pass
print(" Install manually: sudo apt install vobsub2srt")
print(" Alternative GUI : https://github.com/SubtitleEdit/subtitleedit")
return False
# ---------- GPU detection ----------------------------------------------------
def get_device():
try:
import torch
if torch.cuda.is_available():
print(f" GPU detected: {torch.cuda.get_device_name(0)} (CUDA)")
return "cuda"
if hasattr(torch.backends, 'mps') and torch.backends.mps.is_available():
print(" GPU detected: Apple Silicon (MPS)")
return "mps"
except Exception:
pass
print(" No GPU detected - running on CPU.")
return "cpu"
def load_whisper_model(model_name):
device = get_device()
size = WHISPER_MODEL_SIZES.get(model_name, '?')
print(f" Loading Whisper '{model_name}' model "
f"(first run downloads ~{size} to ~/.cache/whisper)...")
try:
return _whisper.load_model(model_name, device=device), device
except Exception as e:
if 'out of memory' in str(e).lower() and device != 'cpu':
print(" GPU out of memory - clearing cache and retrying on CPU...")
try:
import torch
torch.cuda.empty_cache()
torch.cuda.synchronize()
except Exception:
pass
return _whisper.load_model(model_name, device='cpu'), 'cpu'
raise
# ---------- File listing / selection -----------------------------------------
def list_files(exts, label):
exts = (exts,) if isinstance(exts, str) else exts
files = [f for f in sorted(os.listdir('.')) if f.lower().endswith(exts)]
if not files:
print(f"No {label} files found in current directory.")
else:
for i, f in enumerate(files, 1):
print(f"{i}: {f}")
return files
def pick_file(files, prompt, allow_skip=False):
skip_hint = " or Enter to skip" if allow_skip else ""
while True:
choice = input(prompt + skip_hint + " (0 to cancel): ").strip()
if choice == '0':
return None
if choice == "" and allow_skip:
return ""
if choice == "":
for i, f in enumerate(files, 1):
print(f"{i}: {f}")
continue
if choice.isdigit():
idx = int(choice)
if 1 <= idx <= len(files):
return files[idx - 1]
print("Invalid number.")
continue
if os.path.isfile(choice):
return choice
print("File not found.")
# ---------- SRT parsing / writing --------------------------------------------
def srt_to_seconds(t):
h, m, rest = t.split(':')
s, ms = rest.split(',')
return int(h)*3600 + int(m)*60 + int(s) + int(ms)/1000.0
def seconds_to_srt(t):
t = max(0.0, t)
h = int(t) // 3600
m = (int(t) // 60) % 60
s = int(t) % 60
ms = int(round((t - int(t)) * 1000))
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
def _read_srt_text(path):
"""Read an SRT file, auto-detecting encoding and stripping BOM."""
for enc in ('utf-8-sig', 'utf-16', 'cp1252', 'latin-1'):
try:
text = open(path, encoding=enc).read()
# utf-16 files decoded correctly won't have lone surrogates
return text
except (UnicodeDecodeError, UnicodeError):
continue
return open(path, encoding='utf-8', errors='replace').read()
def _normalise_srt_ts(text):
"""Accept HH:MM:SS.mmm or H:MM:SS,mm etc. — normalise to HH:MM:SS,mmm."""
def _fix(m):
ts = m.group(0)
ts = ts.replace('.', ',')
ms_part = ts.rsplit(',', 1)[1]
ts = ts.rsplit(',', 1)[0] + ',' + ms_part.ljust(3, '0')[:3]
return ts
return re.sub(r'\d{1,2}:\d{2}:\d{2}[,\.]\d{1,3}', _fix, text)
_HI_LINE_RE = re.compile(r'^\s*[\(\[].+[\)\]]\s*$') # lines that are ONLY a bracketed description
def _is_hi_subtitle(path):
"""Return True if >25% of text lines look like HI sound descriptions."""
entries = parse_srt_full(path, limit=80)
if not entries:
return False
total = hi = 0
for _, _, text in entries:
for line in text.splitlines():
line = line.strip()
if not line:
continue
total += 1
if _HI_LINE_RE.match(line):
hi += 1
return total > 0 and (hi / total) > 0.25
def _strip_hi_for_sync(src_path, dst_path):
"""Write a copy of src_path with description-only entries removed.
Entries that mix dialogue with descriptions are kept (stripped to dialogue only).
Returns True if any entries were removed/modified."""
text = _normalise_srt_ts(_read_srt_text(src_path))
blocks = re.split(r'\n\s*\n', text.strip())
out = []
changed = False
for block in blocks:
lines = block.strip().splitlines()
ts_idx = next((i for i, l in enumerate(lines) if TS_RE.search(l)), None)
if ts_idx is None:
out.append(block)
continue
text_lines = [l for l in lines[ts_idx + 1:] if l.strip()]
dialogue = [l for l in text_lines if not _HI_LINE_RE.match(l)]
if not text_lines:
out.append(block)
elif not dialogue:
# entry is entirely sound descriptions — drop it
changed = True
else:
if len(dialogue) < len(text_lines):
changed = True
out.append('\n'.join(lines[:ts_idx + 1] + dialogue))
with open(dst_path, 'w', encoding='utf-8') as f:
f.write('\n\n'.join(out))
return changed
def parse_srt_full(path, limit=9999):
entries = []
try:
text = _normalise_srt_ts(_read_srt_text(path))
except Exception:
return entries
for block in re.split(r'\n\s*\n', text.strip()):
lines = block.strip().splitlines()
for i, line in enumerate(lines):
m = TS_RE.search(line)
if m:
start = srt_to_seconds(m.group(1))
end = srt_to_seconds(m.group(2))
body = re.sub(r'<[^>]+>', '', ' '.join(lines[i+1:]).strip())
entries.append((start, end, body))
break
if len(entries) >= limit:
break
return entries
def normalize_word(w):
return re.sub(r"[^a-z0-9']", '', w.lower())
def srt_to_word_times(entries):
result = []
for start, _end, text in entries:
for raw in text.split():
w = normalize_word(raw)
if len(w) >= MIN_WORD_LEN and w not in STOP_WORDS:
result.append((w, start))
return result
def shift_srt(inpath, outpath, offset):
text = _normalise_srt_ts(_read_srt_text(inpath))
with open(outpath, 'w', encoding='utf-8') as fout, \
__import__('io').StringIO(text) as fin:
for line in fin:
m = TS_RE.search(line)
if m:
s = srt_to_seconds(m.group(1)) + offset
e = srt_to_seconds(m.group(2)) + offset
fout.write(f"{seconds_to_srt(s)} --> {seconds_to_srt(e)}\n")
else:
fout.write(line)
def parse_offset(s):
try:
return float(s)
except Exception:
return None
# ---------- Filename / show info parsing -------------------------------------
def extract_show_info(filepath, extra_paths=None):
"""
Extract (show_name, SxxExx) by checking, in order:
1. The video filename
2. Any extra_paths (e.g. matching SRT filename)
3. Directory path components (handles SxxExx in a folder name)
4. Plex-style layout: .../Show Name/Season NN/file
5. Immediate parent directory name as a last resort
"""
show = ''
episode = ''
def _parse_name(path):
base = os.path.splitext(os.path.basename(path))[0]
base = _BRACKET_RE.sub('', base) # strip leading [SubGroup]
base = re.sub(r'[._]', ' ', base)
m = _SXXEXX_RE.search(base)
if m:
s = _NOISE_RE.sub('', base[:m.start()]).strip()
return re.sub(r'\s+', ' ', s).strip(), m.group(0).upper()
s = _NOISE_RE.sub('', base).strip()
return re.sub(r'\s+', ' ', s).strip(), ''
for path in [filepath] + (extra_paths or []):
s, e = _parse_name(path)
if not show and s:
show = s
if not episode and e:
episode = e
if show and episode:
break
if not show or not episode:
parts = os.path.normpath(os.path.abspath(filepath)).split(os.sep)
for part in reversed(parts[:-1]):
part_clean = re.sub(r'[._]', ' ', part)
m = _SXXEXX_RE.search(part_clean)
if m:
if not episode:
episode = m.group(0).upper()
if not show:
s = _NOISE_RE.sub('', part_clean[:m.start()]).strip()
show = re.sub(r'\s+', ' ', s).strip()
if not show:
for i, part in enumerate(parts):
if _SEASON_DIR_RE.match(part) and i > 0:
show = re.sub(r'[._]', ' ', parts[i - 1]).strip()
show = re.sub(r'\s+', ' ', show).strip()
break
if not show:
parent = os.path.basename(os.path.dirname(os.path.abspath(filepath)))
if parent not in ('', '.') and not _SEASON_DIR_RE.match(parent):
show = re.sub(r'[._]', ' ', parent).strip()
show = re.sub(r'\s+', ' ', show).strip()
return show, episode
# ---------- Text post-processing ---------------------------------------------
def postprocess_text(text):
text = text.strip()
if not text:
return text
# OCR misreads \u266a as $. Strip $ embedded inside words; replace remaining
# $ (not before a digit) with \u266a so music-note lines are handled correctly.
text = re.sub(r'(?<=[A-Za-z])\$(?=[A-Za-z])', '', text)
text = re.sub(r'\$(?!\d)', '\u266a', text)
music_rx = re.compile(
r'\[\s*(music|singing|song|humming|instrumental|melody)\s*\]',
re.IGNORECASE
)
has_music = bool(music_rx.search(text)) or '\u266a' in text
text = music_rx.sub('\u266a', text)
text = re.sub(r'\[[^\]]{1,40}\]', '', text).strip()
text = re.sub(r' +', ' ', text).strip()
if has_music:
core = re.sub(r'[\u266a]+', '', text).strip()
text = f'\u266a {core} \u266a' if core else '\u266a'
if text.startswith('\u266a'):
after = text[1:].lstrip()
if after and after[0].islower():
text = '\u266a ' + after[0].upper() + after[1:]
elif text and text[0].islower():
text = text[0].upper() + text[1:]
return text
# ---------- SRT vocabulary extraction ----------------------------------------
def extract_srt_vocab(srt_path, max_words=60):
entries = parse_srt_full(srt_path)
proper = {}
for _, _, text in entries:
words = text.split()
for i, raw in enumerate(words):
w = re.sub(r"[^a-zA-Z']", '', raw)
if not w:
continue
if i > 0 and w[0].isupper() and w.lower() not in STOP_WORDS:
proper[w] = proper.get(w, 0) + 1
return sorted(proper, key=lambda w: -proper[w])[:max_words]
def build_prompt(video_path, srt_path=None):
show, episode = extract_show_info(video_path)
prompt = WHISPER_PROMPT
if show:
prompt += f" This is '{show}'"
prompt += f", {episode}." if episode else "."
if srt_path and os.path.isfile(srt_path):
vocab = extract_srt_vocab(srt_path)
if vocab:
prompt += f" Vocabulary: {', '.join(vocab)}."
return prompt
# ---------- Mode 2: Generate SRT from scratch --------------------------------
def _detect_language(video_path, model):
"""Sample 30 s of audio and return (code, confidence, display_name)."""
import numpy as np
raw = subprocess.run([
'ffmpeg', '-hide_banner', '-loglevel', 'error',
'-i', video_path, '-t', '30',
'-vn', '-ac', '1', '-ar', '16000', '-f', 'f32le', 'pipe:1'
], stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=60).stdout
n = len(raw) // 4
if n == 0:
return None, None, None
audio = np.frombuffer(raw, dtype=np.float32).copy()
audio = _whisper.pad_or_trim(audio)
n_mels = getattr(getattr(model, 'dims', None), 'n_mels', 80)
mel = _whisper.log_mel_spectrogram(audio, n_mels=n_mels).to(model.device)
_, probs = model.detect_language(mel)
code = max(probs, key=probs.get)
conf = probs[code]
names = getattr(_whisper.tokenizer, 'LANGUAGES', {})
name = names.get(code, code).title()
return code, conf, name
def generate_srt(video_path, output_path, model_name, srt_path=None,
task='transcribe', language=None, _model=None):
if _model is None:
_model, _ = load_whisper_model(model_name)
show, episode = extract_show_info(video_path)
if show:
print(f" Detected show: '{show}'" + (f" Episode: {episode}" if episode else ""))
if srt_path:
print(f" Vocabulary seeded from: {os.path.basename(srt_path)}")
if task == 'translate':
hint = f" (source: {language})" if language else " (auto-detect source)"
print(f" Translating to English{hint} - lines will appear as recognised...")
else:
print(" Transcribing - lines will appear as they are recognised...")
result = _model.transcribe(video_path,
initial_prompt=build_prompt(video_path, srt_path),
language=language,
task=task,
verbose=True)
segs = result.get('segments', [])
idx = 0
with open(output_path, 'w', encoding='utf-8') as f:
for seg in segs:
txt = postprocess_text(seg['text'])
if not txt:
continue
idx += 1
f.write(f"{idx}\n")
f.write(f"{seconds_to_srt(seg['start'])} --> {seconds_to_srt(seg['end'])}\n")
f.write(f"{txt}\n\n")
return idx, output_path
def scan_title_card(video_path, start=20, duration=160, interval=5):
"""
Extract frames from the video and OCR them to find on-screen episode title cards.
Returns list of (text, frame_count, timestamp_seconds) sorted by frame count.
"""
try:
import easyocr
except ImportError:
print(" easyocr not available.")
return []
import tempfile, glob
end = start + duration
print(f" Extracting frames ({start}s – {end}s, one every {interval}s)...")
seen = {} # lower-normalised key -> (original_case, count, first_timestamp)
with tempfile.TemporaryDirectory() as tmpdir:
frame_pattern = os.path.join(tmpdir, 'frame_%04d.png')
r = subprocess.run([
'ffmpeg', '-hide_banner', '-loglevel', 'error',
'-ss', str(start), '-i', video_path,
'-t', str(duration),
'-vf', f'fps=1/{interval},scale=1280:-1',
frame_pattern
], timeout=120)
frames = sorted(glob.glob(os.path.join(tmpdir, 'frame_*.png')))
if not frames:
print(" No frames extracted.")
return []
print(f" Running OCR on {len(frames)} frames"
f" (first run downloads ~170 MB model)...")
reader = easyocr.Reader(['en'], verbose=False)
for frame_idx, frame_path in enumerate(frames):
ts = start + frame_idx * interval
try:
results = reader.readtext(frame_path, detail=1, paragraph=False)
frame_seen = set()
for (_, text, conf) in results:
text = text.strip()
if conf < 0.4:
continue
words = text.split()
if not (2 <= len(words) <= 8) or not (4 <= len(text) <= 60):
continue
if re.search(r'[©®@]|\d{2}:\d{2}|www\.', text):
continue
key = re.sub(r'\s+', ' ', text).lower()
if key not in frame_seen:
frame_seen.add(key)
if key in seen:
seen[key] = (seen[key][0], seen[key][1] + 1, seen[key][2])
else:
seen[key] = (text, 1, ts)
except Exception:
continue
return sorted(seen.values(), key=lambda x: -x[1])
def _timed_input(prompt, timeout=15):
"""Print prompt and wait for Enter; auto-continues after timeout seconds."""
import select as _sel
print(prompt, end='', flush=True)
ready, _, _ = _sel.select([sys.stdin], [], [], timeout)
if ready:
sys.stdin.readline()
else:
print(f" (timed out after {timeout}s)")
def _preview_frame(video_path, timestamp):
"""Extract the frame at timestamp and open it in the system image viewer."""
import tempfile
fd, png = tempfile.mkstemp(suffix='.png', prefix='cc_preview_')
os.close(fd)
try:
subprocess.run([
'ffmpeg', '-hide_banner', '-loglevel', 'error',
'-ss', str(timestamp), '-i', video_path,
'-frames:v', '1', '-y', png
], timeout=30, check=True)
viewer = 'open' if sys.platform == 'darwin' else 'xdg-open'
subprocess.Popen([viewer, png])
_timed_input(" (Press Enter to continue, auto-closes in 15s...)", timeout=15)
except Exception as e:
print(f" Preview failed: {e}")
finally:
try:
os.unlink(png)
except Exception:
pass
def _sync_pass(video_path, whisper_out, final_out, ffsubsync_ok):
"""Run ffsubsync on whisper_out → final_out. Returns path of best result."""
if not ffsubsync_ok:
print(" ffsubsync not available, skipping timing pass.")
return whisper_out
ok, offset = sync_with_ffsubsync(video_path, whisper_out, final_out)
if ok:
if offset is not None:
print(f" Timing adjusted by {offset:+.3f} s")
return final_out
print(" ffsubsync timing pass failed - using Whisper output as-is.")
return whisper_out
def generate_and_sync(video_path, model_name, srt_path=None, ffsubsync_ok=False):
"""Load model once, detect language, ask user, then transcribe/translate/both."""
global WHISPER_TASK, WHISPER_LANGUAGE
base = os.path.splitext(video_path)[0]
model, _ = load_whisper_model(model_name)
# --- Language detection ---
print("\n Detecting language from first 30 seconds...")
lang_code, conf, lang_name = _detect_language(video_path, model)
if lang_code:
print(f" Detected: {lang_name} ({lang_code}) {conf*100:.0f}% confidence")
else:
print(" Language detection failed — defaulting to current setting.")
lang_code = WHISPER_LANGUAGE
is_english = lang_code in ('en', None)
# --- Skip choice if --translate was passed explicitly ---
if WHISPER_TASK == 'translate' and not is_english:
task = 'translate'
src_lang = lang_code
do_orig = False
do_en = True
elif is_english:
task = 'transcribe'
src_lang = lang_code
do_orig = True
do_en = False
else:
# Non-English detected — ask what to generate
print(f"\n Source language: {lang_name}. What would you like?")
print(f" 1: {lang_name} SRT - transcribe in original language")
print( " 2: English SRT - translate to English")
print(f" 3: Both - {lang_name} + English SRT")
print( " 0: Cancel")
while True:
ch = input(" Choose [2]: ").strip() or '2'
if ch in ('0', '1', '2', '3'):
break
print(" Enter 0-3.")
if ch == '0':
return None
do_orig = ch in ('1', '3')
do_en = ch in ('2', '3')
src_lang = lang_code
outputs = []
# --- Original language pass ---
if do_orig:
suffix = f'-whisper-{src_lang}' if src_lang and src_lang != 'en' else '-whisper'
w_out = f"{base}{suffix}.srt"
f_out = f"{base}{suffix}-synced.srt"
print(f"\nWhisper transcription → {os.path.basename(w_out)}")
n, _ = generate_srt(video_path, w_out, model_name,
srt_path=srt_path, task='transcribe',
language=src_lang, _model=model)
print(f" {n} segments written.")
print(f"\nffsubsync timing pass → {os.path.basename(f_out)}")
outputs.append(_sync_pass(video_path, w_out, f_out, ffsubsync_ok))
# --- English translation pass ---
if do_en:
w_out = f"{base}-whisper-en.srt"
f_out = f"{base}-whisper-en-synced.srt"
print(f"\nWhisper translation → English → {os.path.basename(w_out)}")
n, _ = generate_srt(video_path, w_out, model_name,
srt_path=srt_path, task='translate',
language=src_lang, _model=model)
print(f" {n} segments written.")
print(f"\nffsubsync timing pass → {os.path.basename(f_out)}")
outputs.append(_sync_pass(video_path, w_out, f_out, ffsubsync_ok))
return outputs[-1] if outputs else None
# ---------- ffsubsync --------------------------------------------------------
def sync_with_ffsubsync(video_path, srt_path, output_path):
exe = _find_ffsubsync()
if not exe:
return False, None
import tempfile
# HI subtitles (hearing impaired) have many [sound] descriptions that
# don't correspond to speech, wrecking VAD-based cross-correlation.
# Sync on a dialogue-only copy; apply the resulting offset to the original.
hi = _is_hi_subtitle(srt_path)
if hi:
print(" Detected HI (hearing-impaired) subtitle — stripping sound "
"descriptions for sync pass, will reapply to original.")
fd, stripped_path = tempfile.mkstemp(suffix='.srt')
os.close(fd)
_strip_hi_for_sync(srt_path, stripped_path)
sync_src = stripped_path
else:
stripped_path = None
sync_src = srt_path
print(" Running ffsubsync (WebRTC VAD + FFT) - usually 20-30 seconds...")
result = subprocess.run(
[exe, video_path, '-i', sync_src, '-o', output_path],
capture_output=True, text=True
)
if stripped_path:
try:
os.remove(stripped_path)
except OSError:
pass
combined = result.stdout + result.stderr
if result.returncode != 0 or not os.path.isfile(output_path):
return False, None
# Parse scale factor; if significant, apply it to correct framerate drift.
# A plain offset fixes a constant gap; scaling fixes drift that grows over
# time when the SRT was authored for a different framerate than the video.
scale_m = re.search(r'framerate scale factor[:\s]+([\d.]+)', combined)
if scale_m:
scale = float(scale_m.group(1))
if not 0.98 <= scale <= 1.02:
src_fps = 'NTSC 23.976' if scale < 1.0 else 'PAL 25'
vid_fps = 'PAL 25' if scale < 1.0 else 'NTSC 23.976'
drift = abs(1.0 - scale) * 100
print(f" Framerate mismatch: SRT={src_fps}fps, video={vid_fps}fps "
f"(scale {scale:.4f}, ~{drift:.1f}% drift) — applying correction.")
scaled = parse_srt_full(output_path)
with open(output_path, 'w', encoding='utf-8') as _f:
for _i, (_s, _e, _t) in enumerate(scaled, 1):
_f.write(f"{_i}\n{seconds_to_srt(_s * scale)} --> "
f"{seconds_to_srt(_e * scale)}\n{_t}\n\n")
# If HI, we got a synced version of the stripped file; now shift the
# original (with all descriptions) by the same offset instead.
def first_ts(path):
try:
for line in _read_srt_text(path).splitlines():
m = TS_RE.search(line)
if m:
return srt_to_seconds(m.group(1))
except Exception:
pass
return None
t_orig = first_ts(srt_path)
t_synced = first_ts(output_path)
offset = (t_synced - t_orig) if (t_orig is not None and t_synced is not None) else None
if hi and offset is not None:
# Replace ffsubsync's output (stripped) with shifted original (full HI)
shift_srt(srt_path, output_path, offset)
return True, offset
# ---------- Whisper word alignment -------------------------------------------
def whisper_word_times(video_path, model_name, srt_path=None):
import numpy as np
model, device = load_whisper_model(model_name)
print(f" Extracting audio (first {ANALYZE_S//60} min)...")
raw = subprocess.run([
'ffmpeg', '-hide_banner',
'-i', video_path,
'-t', str(ANALYZE_S),
'-vn', '-ac', '1', '-ar', '16000',
'-f', 'f32le', 'pipe:1'
], stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=300).stdout
n = len(raw) // 4
if n == 0:
raise RuntimeError("ffmpeg returned no audio.")
audio = np.frombuffer(raw, dtype=np.float32).copy()
print(" Transcribing...")
result = model.transcribe(audio,
initial_prompt=build_prompt(video_path, srt_path),
language=WHISPER_LANGUAGE,
word_timestamps=True, verbose=False)
words = []
for seg in result.get('segments', []):
for wd in seg.get('words', []):
w = normalize_word(wd.get('word', ''))
if len(w) >= MIN_WORD_LEN and w not in STOP_WORDS:
words.append((w, wd['start']))
return words
def compute_offset_whisper(srt_path, video_path, model_name):
entries = parse_srt_full(srt_path)
if not entries:
return None, 0, 0, "No entries found in SRT."
window_entries = [(s, e, t) for s, e, t in entries
if s <= ANALYZE_S + MAX_OFFSET_S]
if not window_entries:
print(" Warning: no SRT entries in analysis window - using first 100.")
window_entries = entries[:100]
srt_wt = srt_to_word_times(window_entries)
if not srt_wt:
return None, 0, 0, "No usable words in SRT window."
print(f" Analysis window: 0-{ANALYZE_S}s | {len(window_entries)} SRT cues")
try:
whi_wt = whisper_word_times(video_path, model_name, srt_path)
except Exception as e:
return None, 0, 0, f"Whisper failed: {e}"
if not whi_wt:
return None, 0, 0, "Whisper produced no output."
print(f" SRT: {len(srt_wt)} words | Whisper: {len(whi_wt)} words")
print(" Aligning word sequences...")
matcher = difflib.SequenceMatcher(
None, [w for w, _ in srt_wt], [w for w, _ in whi_wt], autojunk=False
)
raw_offsets = []
for i, j, n in matcher.get_matching_blocks():
for k in range(n):
raw_offsets.append(whi_wt[j+k][1] - srt_wt[i+k][1])
if len(raw_offsets) < 5:
return None, len(raw_offsets), 0, (
f"Only {len(raw_offsets)} word matches. Is this SRT for this video?"
)
rough = median(raw_offsets)
cleaned = [o for o in raw_offsets if abs(o - rough) <= 2.0]
if len(cleaned) < 5:
cleaned = raw_offsets
off = median(cleaned)
spread = max(cleaned) - min(cleaned)
print(f" Matches after outlier filter: {len(cleaned)}/{len(raw_offsets)}")
return off, len(cleaned), spread, None
# ---------- Whisper cross-check of ffsubsync result --------------------------
def whisper_verify(srt_path, video_path, model_name, ffsubsync_offset):
print(" Verifying with Whisper word alignment...")
w_offset, n_matches, spread, err = compute_offset_whisper(
srt_path, video_path, model_name
)
if err:
return None, None, 0, 0, err
agree = abs(w_offset - ffsubsync_offset) <= OFFSET_AGREE_THRESHOLD
return agree, w_offset, n_matches, spread, None
# ---------- Fallback: speech-band energy cross-correlation ------------------
def extract_speech_energy(video_path):
total = int(ANALYZE_S / RESOLUTION_S) + 1
cmd = [
'ffmpeg', '-hide_banner',
'-i', video_path,
'-t', str(ANALYZE_S), '-vn', '-ac', '1',
'-af', f'highpass=f={SPEECH_LO},lowpass=f={SPEECH_HI}',
'-ar', str(RESAMPLE_HZ), '-f', 'f32le', 'pipe:1'
]
try:
r = subprocess.run(cmd, stdout=subprocess.PIPE,
stderr=subprocess.PIPE, timeout=300)
raw = r.stdout
n = len(raw) // 4
if n == 0:
return []
samples = struct.unpack(f'<{n}f', raw)
energy = [0.0] * total
for i in range(0, n, CHUNK_SIZE):
seg = samples[i:i+CHUNK_SIZE]
rms = (sum(x*x for x in seg) / len(seg)) ** 0.5
bi = i // CHUNK_SIZE
if bi < total:
energy[bi] = rms
return energy
except Exception as e:
print(f" Audio extraction error: {e}")
return []
def compute_onsets(energy, lookback=2):
onsets = [0.0] * len(energy)
for i in range(lookback, len(energy)):
d = energy[i] - energy[i - lookback]
if d > 0:
onsets[i] = d
nz = sorted(o for o in onsets if o > 0)
if nz:
thr = nz[len(nz) // 2]
onsets = [o if o >= thr else 0.0 for o in onsets]
return onsets
def crosscorr_offset(entries, energy):
n_bins = len(energy)
max_lag = int(MAX_OFFSET_S / RESOLUTION_S)
onsets = compute_onsets(energy)
seen, starts = set(), []
for s, _e, _t in entries:
si = max(0, int(s / RESOLUTION_S))
if si not in seen:
starts.append(si)
seen.add(si)
if not starts or not any(onsets):
return 0.0, 0.0
scores = [
sum(onsets[i+lag] for i in starts if 0 <= i+lag < n_bins)
for lag in range(-max_lag, max_lag + 1)
]
best = max(range(len(scores)), key=lambda i: scores[i])
mean = sum(scores) / len(scores)
return (best - max_lag) * RESOLUTION_S, scores[best] / max(1e-9, mean)
def compute_offset_fallback(srt_path, video_path):
entries = parse_srt_full(srt_path)
if not entries:
return None, None, "No entries in SRT."
print(" Extracting speech-band audio energy...")
energy = extract_speech_energy(video_path)
if not energy or not any(energy):
return None, None, "Could not extract audio from video."
print(" Running onset cross-correlation...")
offset, conf = crosscorr_offset(entries, energy)
return offset, conf, None
# ---------- Core sync logic (used by Mode 2 and batch) ----------------------
def sync_single(video, src, out, ffsubsync_ok, whisper_ok, interactive=True):
"""
Sync src SRT to video, writing result to out.
ffsubsync runs first; Whisper independently verifies the offset.
interactive=True prompts user on disagreement; False just warns and keeps ffsubsync.
Returns True on success.
"""
synced = False
final_offset = None
# Primary: ffsubsync
if ffsubsync_ok:
ok, fs_offset = sync_with_ffsubsync(video, src, out)
if ok:
if fs_offset is not None:
print(f" ffsubsync offset : {fs_offset:+.3f} s")
# Cross-check with Whisper
if whisper_ok and fs_offset is not None:
agree, w_offset, n_matches, spread, err = whisper_verify(
src, video, WHISPER_MODEL, fs_offset
)
if err:
print(f" Whisper verify skipped: {err}")
else:
quality = ("good" if spread < 2.0 else
"moderate" if spread < 5.0 else "low")
diff = abs(w_offset - fs_offset)
print(f" Whisper offset : {w_offset:+.3f} s "
f"({n_matches} words, spread {spread:.1f}s, {quality})")
if agree:
print(f" Agreement : YES (differ by {diff:.2f}s) "
f"- using ffsubsync result.")
else:
print(f" Agreement : NO (differ by {diff:.2f}s, "
f"threshold {OFFSET_AGREE_THRESHOLD}s)")
if interactive:
print(f" [f] Use ffsubsync ({fs_offset:+.3f}s)")
print(f" [w] Use Whisper ({w_offset:+.3f}s)")
print(f" [e] Enter offset manually")
while True:
choice = input(" Choose [f/w/e]: ").strip().lower()
if choice == 'f':
print(" Using ffsubsync offset.")
break
elif choice == 'w':
print(" Re-applying Whisper offset...")
shift_srt(src, out, w_offset)
final_offset = w_offset
break
elif choice == 'e':
while True:
resp = input(" Enter offset in seconds: ").strip()
manual = parse_offset(resp)
if manual is not None:
shift_srt(src, out, manual)
final_offset = manual
break
print(" Invalid number.")
break
else:
print(f" WARNING: methods disagree by {diff:.2f}s. "
f"Keeping ffsubsync - review manually.")
synced = True
final_offset = final_offset or fs_offset
else:
print(" ffsubsync failed - falling back to Whisper...")
# Fallback 1: Whisper word alignment
if not synced and whisper_ok:
offset, n_matches, spread, err = compute_offset_whisper(src, video, WHISPER_MODEL)
if not err:
quality = ("good" if spread < 2.0 else
"moderate" if spread < 5.0 else "low")
print(f" Whisper offset: {offset:+.3f} s "
f"({n_matches} matches, spread {spread:.1f}s, {quality})")
shift_srt(src, out, offset)
final_offset = offset
synced = True
else:
print(f" Whisper failed: {err}")
print(" Trying audio energy cross-correlation...")
# Fallback 2: energy cross-correlation
if not synced:
offset, conf, err = compute_offset_fallback(src, video)
if not err:
q = "LOW" if conf < 1.5 else "moderate" if conf < 2.5 else "good"
print(f" Energy offset: {offset:+.3f} s (confidence {conf:.2f}x, {q})")
shift_srt(src, out, offset)
final_offset = offset
synced = True
else:
print(f" All methods failed: {err}")
if synced and final_offset is not None:
print(f" Final offset: {final_offset:+.3f} s")
return synced
# ---------- Batch helpers ----------------------------------------------------
def find_srt_for_video(video_path, srt_files):
_, ep = extract_show_info(video_path)
ep_lower = ep.lower() if ep else None
base = os.path.splitext(os.path.basename(video_path))[0]
candidates = [f for f in srt_files
if not f.lower().endswith('-synced.srt')
and not f.lower().endswith('-whisper.srt')]
if ep_lower:
ep_matches = [f for f in candidates if ep_lower in f.lower()]
if ep_matches:
return sorted(ep_matches, key=len)[0]
exact = base + '.srt'
if exact in candidates:
return exact
return None
def batch_sync(ffsubsync_ok, whisper_ok):
video_files = [f for f in sorted(os.listdir('.'))
if f.lower().endswith(VIDEO_EXTS)]
srt_files = [f for f in sorted(os.listdir('.'))
if f.lower().endswith('.srt')]
if not video_files:
print("No video files found.")
return
if not srt_files:
print("No SRT files found.")
return
pairs, unmatched = [], []
for vf in video_files:
sf = find_srt_for_video(vf, srt_files)
if sf:
out = os.path.splitext(sf)[0] + '-synced.srt'
if os.path.isfile(out):
print(f" Skipping {vf} - {os.path.basename(out)} already exists.")
else:
pairs.append((vf, sf, out))
else:
unmatched.append(vf)
if not pairs:
print("No unprocessed pairs found.")
if unmatched:
print("Videos with no matching SRT:")
for v in unmatched:
print(f" {v}")
return
print(f"\nFound {len(pairs)} pair(s) to process:")
for vf, sf, out in pairs:
print(f" {vf} + {sf} -> {os.path.basename(out)}")
if unmatched:
print(f"\n{len(unmatched)} video(s) with no matching SRT (skipped):")
for v in unmatched:
print(f" {v}")
if input("\nProceed? [Y/n]: ").strip().lower() not in ('', 'y'):
print("Cancelled.")
return
ok_count, fail_count, failed = 0, 0, []
for vf, sf, out in pairs:
print(f"\n{'='*60}")
print(f" Video : {vf}")
print(f" SRT : {sf}")
print(f" Output: {os.path.basename(out)}")
if sync_single(vf, sf, out, ffsubsync_ok, whisper_ok, interactive=False):
ok_count += 1
else:
fail_count += 1
failed.append(vf)
print(f"\n{'='*60}")
print(f"Batch complete: {ok_count} synced, {fail_count} failed.")
if failed:
print("Run Mode 2 manually on these:")
for v in failed:
print(f" {v}")
# ---------- TMDB episode lookup + rename -------------------------------------
def tmdb_get(path, params, api_key):
params = dict(params) # don't mutate caller's dict
params['api_key'] = api_key
url = f"https://api.themoviedb.org/3{path}?{urllib.parse.urlencode(params)}"
req = urllib.request.Request(url, headers={'Accept-Encoding': 'gzip, deflate'})
try:
with urllib.request.urlopen(req, timeout=10) as r:
raw = r.read()
if raw[:2] == b'\x1f\x8b':
import gzip
raw = gzip.decompress(raw)
return json.loads(raw.decode('utf-8'))
except Exception as e:
print(f" TMDB error: {e}")
return None
def _get_tmdb_key():
key = TMDB_API_KEY.strip()
if not key:
print(" Get a free key at https://www.themoviedb.org/settings/api")
key = input(" Enter TMDB API key: ").strip()
if not key:
print(" No key - skipping.")
return None
return key
def tmdb_pick_show(show_name, key):
"""Search TMDB for show_name and let the user pick. Returns (show_id, canonical) or (None, None)."""
data = tmdb_get('/search/tv', {'query': show_name, 'page': 1}, key)
if not data or not data.get('results'):
print(" No results found.")
return None, None
results = data['results'][:6]
if len(results) > 1:
print(" Multiple results:")
for i, r in enumerate(results, 1):
year = r.get('first_air_date', '')[:4]
print(f" {i}: {r['name']} ({year})")
choice = input(" Choose [1]: ").strip()
idx = (int(choice)-1) if choice.isdigit() and 1 <= int(choice) <= len(results) else 0
else:
idx = 0
return results[idx]['id'], results[idx]['name']
def tmdb_find_episode_by_title(show_id, ep_title, key):
"""
Scan every season of show_id on TMDB looking for an episode whose title
matches ep_title (case-insensitive). Returns (season, episode_number) or (None, None).
"""
show_data = tmdb_get(f'/tv/{show_id}', {}, key)
if not show_data:
return None, None
n_seasons = show_data.get('number_of_seasons', 0)
target = ep_title.strip().lower()
for s in range(1, n_seasons + 1):
season_data = tmdb_get(f'/tv/{show_id}/season/{s}', {}, key)
if not season_data:
continue
for ep in season_data.get('episodes', []):
if ep.get('name', '').strip().lower() == target:
return s, ep['episode_number']
return None, None
def safe_filename(s):
return re.sub(r'[<>:"/\\|?*]', '', s).strip()
def find_matching_srt(video_path):
base = os.path.splitext(video_path)[0]
dirpath = os.path.dirname(video_path) or '.'
for suffix in ('', '-synced', '-offset', '-whisper'):
c = base + suffix + '.srt'
if os.path.isfile(c):
return c
_, ep_code = extract_show_info(video_path)
if ep_code:
for f in os.listdir(dirpath):
if f.lower().endswith('.srt') and ep_code.lower() in f.lower():
return os.path.join(dirpath, f)
return None
def do_rename(filepath, new_base):
ext = os.path.splitext(filepath)[1]
dirpath = os.path.dirname(filepath) or '.'
new_path = os.path.join(dirpath, new_base + ext)
if os.path.abspath(filepath) == os.path.abspath(new_path):
print(" Already named correctly.")
return new_path
try:
os.rename(filepath, new_path)
print(f" -> {os.path.basename(new_path)}")
return new_path
except Exception as e:
print(f" Rename failed: {e}")
return filepath
def _tmdb_rename(video_path, show_name, episode_code, srt_path):
"""Core TMDB lookup + rename. show_name / episode_code may be empty strings."""
if not episode_code:
print(" No SxxExx found in filename, SRT, or directory path - skipping.")
return
m = re.match(r'S(\d+)E(\d+)', episode_code, re.IGNORECASE)
if not m:
return
season, episode = int(m.group(1)), int(m.group(2))
if not show_name:
show_name = input(" Could not detect show name. Enter show name: ").strip()
if not show_name:
print(" No show name - skipping.")
return
key = _get_tmdb_key()
if not key:
return
print(f" Searching TMDB for '{show_name}'...")
show_id, canonical = tmdb_pick_show(show_name, key)
if not show_id:
return
ep_data = tmdb_get(f'/tv/{show_id}/season/{season}/episode/{episode}', {}, key)
if ep_data and 'name' in ep_data:
new_base = (f"{safe_filename(canonical)} - "
f"S{season:02d}E{episode:02d} - {safe_filename(ep_data['name'])}")
else:
print(" Episode title not found - using show name + SxxExx only.")
new_base = f"{safe_filename(canonical)} - S{season:02d}E{episode:02d}"
_confirm_and_rename(video_path, new_base, srt_path)
def _confirm_and_rename(video_path, new_base, srt_path):
ext = os.path.splitext(video_path)[1]
print(f"\n New name: {new_base}{ext}")
if srt_path:
print(f" SRT : {new_base}.srt")
if input(" Rename? [Y/n]: ").strip().lower() not in ('', 'y'):
print(" Skipped.")
return
do_rename(video_path, new_base)
if srt_path:
do_rename(srt_path, new_base)
def offer_rename(video_path):
if input("\nLook up episode title on TMDB and rename files? [y/N]: ").strip().lower() != 'y':
return
srt_path = find_matching_srt(video_path)
extra = [srt_path] if srt_path else []
show_name, ep_code = extract_show_info(video_path, extra_paths=extra)
print(f" Show : {show_name or '(not detected)'}")
print(f" Episode: {ep_code or '(not detected)'}")
_tmdb_rename(video_path, show_name, ep_code, srt_path)
def rename_mode(video_path, ocr_ok, _method=None):
srt_path = find_matching_srt(video_path)
extra = [srt_path] if srt_path else []
show_name, ep_code = extract_show_info(video_path, extra_paths=extra)
print(f"\n Show : {show_name or '(not detected)'}")
print(f" Episode: {ep_code or '(not detected)'}")
if _method is None:
print("\n How to find the episode title?")
print(" 1: TMDB lookup - search by show name + SxxExx [default]")
if ocr_ok:
print(" 2: Scan video - OCR the first 3 min for a title card")
choice = input(" Choose [1]: ").strip() or '1'
else:
choice = _method
if choice == '2' and ocr_ok:
candidates = scan_title_card(video_path)
if not candidates:
print(" No title candidates found - falling back to TMDB.")
else:
top = candidates[:20]
print(f"\n Candidates (sorted by how many frames they appeared in):")
for i, (text, count, _ts) in enumerate(top, 1):
print(f" {i}: {text} ({count} frame{'s' if count != 1 else ''})")
print("\n Enter a number to select, p to preview that frame, "
"or Enter to fall back to TMDB.")
ep_title = None
while True:
sel = input(" > ").strip()
if not sel:
break
pm = re.match(r'^[pP](\d+)$', sel)
if pm:
pidx = int(pm.group(1))
if 1 <= pidx <= len(top):
_preview_frame(video_path, top[pidx - 1][2])
else:
print(f" Choose 1–{len(top)}.")
continue
if sel.isdigit() and 1 <= int(sel) <= len(top):
ep_title = top[int(sel) - 1][0]
break
print(f" Enter a number (1–{len(top)}), p to preview, or Enter to skip.")
if ep_title:
if not show_name:
show_name = input(" Enter show name: ").strip()
if not show_name:
print(" No show name - skipping.")
return
key = _get_tmdb_key()
if not key:
return
print(f" Searching TMDB for '{show_name}' / episode '{ep_title}'...")
show_id, canonical = tmdb_pick_show(show_name, key)
if show_id:
season, ep_num = tmdb_find_episode_by_title(show_id, ep_title, key)
if season and ep_num:
new_base = (f"{safe_filename(canonical)} - "
f"S{season:02d}E{ep_num:02d} - {safe_filename(ep_title)}")
_confirm_and_rename(video_path, new_base, srt_path)
return
print(" Episode title not found on TMDB.")
if ep_code:
m = re.match(r'S(\d+)E(\d+)', ep_code, re.IGNORECASE)
if m:
season, ep_num = int(m.group(1)), int(m.group(2))
new_base = (f"{safe_filename(canonical or show_name)} - "
f"S{season:02d}E{ep_num:02d} - {safe_filename(ep_title)}")
_confirm_and_rename(video_path, new_base, srt_path)
return
print(" No episode code available either - skipping.")
return
# Default: TMDB lookup
_tmdb_rename(video_path, show_name, ep_code, srt_path)
def _rename_one_batch(video_path, srt_path, show_id, canonical, key, auto):
"""Rename one file within a batch. Returns True if renamed/confirmed, False if skipped."""
extra = [srt_path] if srt_path else []
_, ep_code = extract_show_info(video_path, extra_paths=extra)
if not ep_code:
print(f" {os.path.basename(video_path)}: no SxxExx found - skipping.")
return False
m = re.match(r'S(\d+)E(\d+)', ep_code, re.IGNORECASE)
if not m:
print(f" {os.path.basename(video_path)}: cannot parse {ep_code} - skipping.")
return False
season, ep_num = int(m.group(1)), int(m.group(2))
ep_data = tmdb_get(f'/tv/{show_id}/season/{season}/episode/{ep_num}', {}, key)
if ep_data and 'name' in ep_data:
new_base = (f"{safe_filename(canonical)} - "
f"S{season:02d}E{ep_num:02d} - {safe_filename(ep_data['name'])}")
else:
print(f" {os.path.basename(video_path)}: episode title not found - using SxxExx only.")
new_base = f"{safe_filename(canonical)} - S{season:02d}E{ep_num:02d}"
ext = os.path.splitext(video_path)[1]
print(f" {os.path.basename(video_path)}")
print(f" -> {new_base}{ext}")
if auto:
do_rename(video_path, new_base)
if srt_path:
do_rename(srt_path, new_base)
else:
if input(" Rename? [Y/n]: ").strip().lower() in ('', 'y'):
do_rename(video_path, new_base)
if srt_path:
do_rename(srt_path, new_base)
else:
print(" Skipped.")
return True
def _next_file_prompt(vid_files, idx, allow_auto=False):
"""
After processing vid_files[idx], ask what to do next.
Returns (next_index, go_auto). next_index is None to stop.
Enter = next file, 0 = stop, a = auto rest (if allow_auto), N = jump.
"""
next_idx = idx + 1
if next_idx >= len(vid_files):
print(" No more files.")
return None, False
print(f"\n Next: {os.path.basename(vid_files[next_idx])}")
auto_hint = " [a] auto rest | " if allow_auto else " "
print(f"{auto_hint}[Enter] continue | [0] stop | [1-{len(vid_files)}] jump to file")
ans = input(" > ").strip().lower()
if ans == '0':
return None, False
if ans == 'a' and allow_auto:
return next_idx, True
if ans == '':
return next_idx, False
if ans.isdigit() and 1 <= int(ans) <= len(vid_files):
return int(ans) - 1, False
return next_idx, False
def rename_tmdb_loop():
print("\n Video files in current directory:")
vid_files = list_files(VIDEO_EXTS, "video")
if not vid_files:
print(" No video files found.")
return
video = pick_file(vid_files, " Choose starting file")
if not video or not os.path.isfile(video):
print(" No valid video selected.")
return
srt0 = find_matching_srt(video)
extra0 = [srt0] if srt0 else []
show_name, _ = extract_show_info(video, extra_paths=extra0)
if not show_name:
show_name = input(" Could not detect show name. Enter show name: ").strip()
if not show_name:
return
key = _get_tmdb_key()
if not key:
return
print(f" Searching TMDB for '{show_name}'...")
show_id, canonical = tmdb_pick_show(show_name, key)
if not show_id:
return
print(f" Show: {canonical}\n")
idx = vid_files.index(video) if video in vid_files else 0
auto = False
while True:
vf = vid_files[idx]
srt = find_matching_srt(vf)
_rename_one_batch(vf, srt, show_id, canonical, key, auto=auto)
if auto:
idx += 1
if idx >= len(vid_files):
print(" No more files.")
break
else:
idx, auto = _next_file_prompt(vid_files, idx, allow_auto=True)
if idx is None:
break
def rename_scan_loop(ocr_ok):
print("\n Video files in current directory:")
vid_files = list_files(VIDEO_EXTS, "video")
if not vid_files:
print(" No video files found.")
return
video = pick_file(vid_files, " Choose starting file")
if not video or not os.path.isfile(video):
print(" No valid video selected.")
return
idx = vid_files.index(video) if video in vid_files else 0
while True:
rename_mode(vid_files[idx], ocr_ok, _method='2')
idx, _ = _next_file_prompt(vid_files, idx)
if idx is None:
break
def rename_menu(ocr_ok):
while True:
print("\n RENAME")
print(" 1: TMDB lookup [default]")
if ocr_ok:
print(" 2: Scan video for title card")
print(" 0: Back to main menu")
valid = ('0', '1', '2') if ocr_ok else ('0', '1')
while True:
choice = input(" Choose [1]: ").strip() or '1'
if choice in valid:
break
print(f" Please enter {'0, 1 or 2' if ocr_ok else '0 or 1'}.")
if choice == '0':
break
elif choice == '1':
rename_tmdb_loop()
elif choice == '2':
rename_scan_loop(ocr_ok)
# ---------- Subtitle extraction ----------------------------------------------
def probe_subtitle_streams(video_path):
"""Return list of subtitle stream dicts from ffprobe."""
try:
r = subprocess.run([
'ffprobe', '-v', 'quiet', '-print_format', 'json',
'-show_streams', '-select_streams', 's', video_path
], capture_output=True, text=True, timeout=30)
return json.loads(r.stdout).get('streams', [])
except Exception:
return []
def _sub_out_path(video_path, lang=''):
base = os.path.splitext(video_path)[0]
return f"{base}.{lang}.srt" if lang else f"{base}.srt"
def _extract_text_track(video_path, stream_index, out_path):
r = subprocess.run([
'ffmpeg', '-hide_banner', '-loglevel', 'error',
'-i', video_path, '-map', f'0:{stream_index}',
'-c:s', 'srt', '-y', out_path
], timeout=300)
return r.returncode == 0 and os.path.isfile(out_path)
def _extract_cc(video_path, out_path, cce_cmd):
print(" Running ccextractor...")
r = subprocess.run([cce_cmd, video_path, '-o', out_path], timeout=600)
return r.returncode == 0 and os.path.isfile(out_path)
def _extract_pgs(video_path, stream_index, out_path):
"""Extract Blu-ray PGS subtitle track → SRT via pgsreader + easyocr."""
try:
import easyocr
from pgsreader import PGSReader
import numpy as np
from PIL import Image as _PILImage
except ImportError as e:
print(f" Missing dependency: {e}")
return False
import tempfile
fd, sup_path = tempfile.mkstemp(suffix='.sup')
os.close(fd)
try:
print(" Extracting PGS stream...")
r = subprocess.run([
'ffmpeg', '-hide_banner', '-loglevel', 'error',
'-i', video_path, '-map', f'0:{stream_index}',
'-c:s', 'copy', '-y', sup_path
], timeout=300)
if r.returncode != 0:
print(" ffmpeg extraction failed.")
return False
print(" Reading PGS display sets...")
pgs = PGSReader(sup_path)
reader = easyocr.Reader(['en'], verbose=False)
entries = []
pending = None
for ds in pgs.displaySets:
ts_s = ds.pcs.presentation_timestamp / 90000.0
if ds.has_image:
img = ds.to_image().convert('RGB')
results = reader.readtext(np.array(img), detail=0, paragraph=True)
text = ' '.join(results).strip()
if pending:
entries.append(pending)
pending = [ts_s, None, text] if text else None
else:
if pending:
pending[1] = ts_s
entries.append(pending)
pending = None
if pending:
pending[1] = pending[0] + 3.0
entries.append(pending)
print(f" Writing {len(entries)} subtitle entries...")
with open(out_path, 'w', encoding='utf-8') as f:
for i, (start, end, text) in enumerate(entries, 1):
f.write(f"{i}\n")
f.write(f"{seconds_to_srt(start)} --> {seconds_to_srt(end)}\n")
f.write(f"{text}\n\n")
return True
finally:
try:
os.unlink(sup_path)
except Exception:
pass
def _extract_vobsub(video_path, stream_index, out_path):
"""Extract DVD VOB subtitle track → SRT via vobsub2srt."""
if not ensure_vobsub2srt():
return False
import tempfile, shutil
with tempfile.TemporaryDirectory() as tmpdir:
sub_base = os.path.join(tmpdir, 'subs')
r = subprocess.run([
'ffmpeg', '-hide_banner', '-loglevel', 'error',
'-i', video_path, '-map', f'0:{stream_index}',
'-c:s', 'copy', '-y', sub_base + '.sub'
], timeout=300)
if r.returncode != 0:
print(" ffmpeg extraction failed.")
return False
r2 = subprocess.run(['vobsub2srt', sub_base], timeout=300)
if r2.returncode == 0 and os.path.isfile(sub_base + '.srt'):
shutil.copy(sub_base + '.srt', out_path)
return True
print(" vobsub2srt conversion failed.")
return False
def extract_subs_mode():
print("\n Video files in current directory:")
vid_files = list_files(VIDEO_EXTS, "video")
if vid_files:
video = pick_file(vid_files, " Choose video by number or filename")
else:
video = input(" Enter path to video file (0 to cancel): ").strip()
if video == '0':
return
if not video or not os.path.isfile(video):
print(" No valid video selected.")
return
video = _offer_mp4_remux(video)
streams = probe_subtitle_streams(video)
# Build menu: numbered subtitle tracks + CC option
options = []
if streams:
print("\n Subtitle tracks found:")
for s in streams:
codec = s.get('codec_name', 'unknown')
idx = s.get('index', '?')
lang = s.get('tags', {}).get('language', '')
title = s.get('tags', {}).get('title', '')
label = codec
if lang: label += f" [{lang}]"
if title: label += f" — {title}"
if codec in _TEXT_SUB_CODECS:
label += " (text, instant)"
elif codec in _IMAGE_SUB_CODECS:
label += " (image, needs OCR)"
print(f" {len(options)+1}: {label}")
options.append(('track', s))
else:
print("\n No subtitle tracks found in file.")
print(f" {len(options)+1}: Closed captions from video stream (ccextractor)")
options.append(('cc', None))
print(" 0: Cancel")
while True:
sel = input(" Choose: ").strip()
if sel == '0':
return
if sel.isdigit() and 1 <= int(sel) <= len(options):
break
print(f" Enter 1-{len(options)} or 0.")
kind, stream = options[int(sel) - 1]
base = os.path.splitext(video)[0]
if kind == 'cc':
cce = ensure_ccextractor()
if not cce:
return
out = _sub_out_path(video, 'cc')
if _extract_cc(video, out, cce):
print(f" Done: {os.path.basename(out)}")
else:
print(" ccextractor found no CC in this file.")
return
codec = stream.get('codec_name', '')
stream_idx = stream.get('index')
lang = stream.get('tags', {}).get('language', '')
out = _sub_out_path(video, lang)
if codec in _TEXT_SUB_CODECS:
print(f" Extracting text track {stream_idx} → {os.path.basename(out)} ...")
if _extract_text_track(video, stream_idx, out):
print(f" Done: {os.path.basename(out)}")
else:
print(" Extraction failed.")
elif codec in _IMAGE_SUB_CODECS:
print(f"\n '{codec}' is an image-based subtitle format.")
print(" 1: Native format - extract as .sup / .sub (perfect quality, instant) [default]")
print(" 2: OCR to SRT - read text via OCR (editable, some quality loss)")
fmt = input(" Choose [1]: ").strip() or '1'
if fmt != '2':
# Native extraction — no OCR, perfect quality
if codec in {'hdmv_pgs_subtitle', 'pgssub'}:
native_out = base + (f'.{lang}' if lang else '') + '.sup'
else:
native_out = base + (f'.{lang}' if lang else '') + '.sub'
print(f" Extracting → {os.path.basename(native_out)} ...")
r = subprocess.run([
'ffmpeg', '-hide_banner', '-loglevel', 'error',
'-i', video, '-map', f'0:{stream_idx}',
'-c:s', 'copy', '-y', native_out
], timeout=300)
if r.returncode == 0 and os.path.isfile(native_out):
print(f" Done: {os.path.basename(native_out)}")
else:
print(" Extraction failed.")
elif codec in {'hdmv_pgs_subtitle', 'pgssub'}:
if not ensure_pgsreader() or not ensure_easyocr():
return
print(f" Extracting PGS → {os.path.basename(out)} (OCR, may take a while)...")
if _extract_pgs(video, stream_idx, out):
print(f" Done: {os.path.basename(out)}")
else:
print(" PGS extraction failed.")
else:
print(f" Extracting DVD/DVB subtitle → {os.path.basename(out)} ...")
if not _extract_vobsub(video, stream_idx, out):
print(" Could not extract automatically.")
else:
print(f" Codec '{codec}' not yet supported for direct extraction.")
print(" Try: ffmpeg -i video -map 0:s:N -c:s srt output.srt")
# ---------- MP4 → MKV remux --------------------------------------------------
_LANG_ISO1_TO_639_2 = {
'en': 'eng', 'fr': 'fre', 'de': 'ger', 'es': 'spa', 'it': 'ita',
'pt': 'por', 'nl': 'dut', 'ru': 'rus', 'ja': 'jpn', 'zh': 'chi',
'ko': 'kor', 'ar': 'ara', 'pl': 'pol', 'sv': 'swe', 'no': 'nor',
'da': 'dan', 'fi': 'fin', 'cs': 'cze', 'tr': 'tur', 'hu': 'hun',
}
_SUB_EXTS = ('.srt', '.ass', '.ssa', '.vtt', '.sup', '.sub')
def _do_remux(video_path, out_path):
"""Stream-copy video_path → out_path (MKV). Returns True on success."""
import shutil
use_mkvmerge = bool(shutil.which('mkvmerge'))
if use_mkvmerge:
print(f" mkvmerge: {os.path.basename(video_path)} → {os.path.basename(out_path)}")
cmd = ['mkvmerge', '-o', out_path, video_path]
else:
print(f" ffmpeg stream copy (mkvmerge not found): {os.path.basename(video_path)} → {os.path.basename(out_path)}")
cmd = ['ffmpeg', '-hide_banner', '-loglevel', 'error',
'-i', video_path, '-c', 'copy', '-y', out_path]
try:
r = subprocess.run(cmd, timeout=600)
except subprocess.TimeoutExpired:
print(" Timed out.")
return False
if r.returncode == 0 and os.path.isfile(out_path):
return True
print(" Remux failed.")
if os.path.exists(out_path):
os.remove(out_path)
return False
def _offer_mp4_remux(video_path):
"""If video_path is an MP4, offer (default yes) to remux to MKV first.
Returns the path to use going forward (MKV on success, original otherwise)."""
if not video_path.lower().endswith('.mp4'):
return video_path
base = os.path.splitext(video_path)[0]
mkv_out = base + '.mkv'
print(f"\n '{os.path.basename(video_path)}' is an MP4.")
print(" MKV handles all subtitle types; MP4 only supports mov_text (SRT).")
if os.path.exists(mkv_out):
print(f" MKV already exists: {os.path.basename(mkv_out)}")
resp = input(" Use existing MKV? [Y/n]: ").strip().lower()
if resp != 'n':
return mkv_out
return video_path
resp = input(" Convert to MKV now (lossless)? [Y/n]: ").strip().lower()
if resp == 'n':
return video_path
if _do_remux(video_path, mkv_out):
in_mb = os.path.getsize(video_path) / 1_048_576
out_mb = os.path.getsize(mkv_out) / 1_048_576
print(f" Done: {os.path.basename(mkv_out)} ({in_mb:.0f} MB → {out_mb:.0f} MB)")
resp = input(" Delete original MP4? [y/N]: ").strip().lower()
if resp == 'y':
os.remove(video_path)
print(f" Deleted: {os.path.basename(video_path)}")
return mkv_out
return video_path
def _detect_lang_tag(sub_path):
"""Guess ISO 639-2 language tag from filename stem (e.g. video.en.srt → eng)."""
stem = os.path.splitext(os.path.basename(sub_path))[0]
parts = stem.rsplit('.', 1)
if len(parts) == 2:
code = parts[1].lower()
if code in _LANG_ISO1_TO_639_2:
return _LANG_ISO1_TO_639_2[code]
if len(code) == 3 and code.isalpha():
return code
return ''
def remux_mp4_to_mkv():
"""Mode 6: remux MP4 (or any container) to MKV — stream copy, no re-encode."""
print("\n MP4 → MKV")
all_vid = list_files(VIDEO_EXTS, "video")
mp4_files = [f for f in all_vid if f.lower().endswith('.mp4')]
if mp4_files:
candidates = mp4_files
else:
print(" (no .mp4 found — showing all video files)")
candidates = all_vid
if candidates:
video = pick_file(candidates, " Choose file to remux (0 to cancel)")
else:
video = input(" Enter path to video file (0 to cancel): ").strip()
if video == '0':
return
if not video:
return
if not os.path.isfile(video):
print(" File not found.")
return
base = os.path.splitext(video)[0]
out = base + '.mkv'
if os.path.exists(out):
print(f" Output already exists: {os.path.basename(out)}")
resp = input(" Overwrite? [y/N]: ").strip().lower()
if resp != 'y':
print(" Cancelled.")
return
if _do_remux(video, out):
in_mb = os.path.getsize(video) / 1_048_576
out_mb = os.path.getsize(out) / 1_048_576
print(f" Done: {os.path.basename(out)} ({in_mb:.0f} MB → {out_mb:.0f} MB)")
resp = input(" Delete original? [y/N]: ").strip().lower()
if resp == 'y':
os.remove(video)
print(f" Deleted: {os.path.basename(video)}")
def embed_subs_mode():
"""Mode 7: soft-mux a subtitle file into a video using mkvmerge."""
if not ensure_mkvtoolnix():
return
# --- pick video ---
print("\n EMBED: Video files in current directory:")
vid_files = list_files(VIDEO_EXTS, "video")
if vid_files:
video = pick_file(vid_files, " Choose video (0 to cancel)")
else:
video = input(" Enter path to video file (0 to cancel): ").strip()
if video == '0':
return
if not video or not os.path.isfile(video):
print(" No valid video selected.")
return
# offer MP4 → MKV before anything else
video = _offer_mp4_remux(video)
# --- pick subtitle file ---
sub_files = sorted(
f for f in os.listdir('.')
if f.lower().endswith(_SUB_EXTS) and not f.endswith('.idx')
)
if sub_files:
print("\n Subtitle files in current directory:")
for i, f in enumerate(sub_files, 1):
print(f" {i}: {f}")
sub = pick_file(sub_files, " Choose subtitle file (0 to cancel)")
else:
sub = input(" Enter path to subtitle file (0 to cancel): ").strip()
if sub == '0':
return
if not sub or not os.path.isfile(sub):
print(" No valid subtitle file selected.")
return
# --- language tag ---
detected = _detect_lang_tag(sub)
if detected:
print(f" Detected language tag: {detected}")
resp = input(f" Use '{detected}'? [Y/n]: ").strip().lower()
lang = detected if resp != 'n' else ''
else:
lang = ''
if not lang:
lang = input(" Enter ISO 639-2 language tag (e.g. eng, fre) or Enter to skip: ").strip().lower()
# --- build mkvmerge command ---
base = os.path.splitext(video)[0]
tmp_out = base + '._embed_tmp.mkv'
cmd = ['mkvmerge', '-o', tmp_out, video]
if lang:
cmd += ['--language', f'0:{lang}']
cmd.append(sub)
print(f"\n Embedding {os.path.basename(sub)} → {os.path.basename(video)} ...")
try:
r = subprocess.run(cmd, timeout=600)
except subprocess.TimeoutExpired:
print(" Timed out.")
return
if r.returncode not in (0, 1) or not os.path.isfile(tmp_out):
# mkvmerge returns 1 for warnings (still produces output)
print(" mkvmerge failed.")
if os.path.exists(tmp_out):
os.remove(tmp_out)
return
# replace original with muxed file
os.replace(tmp_out, video)
print(f" Done: subtitle embedded into {os.path.basename(video)}")
resp = input(" Delete separate subtitle file? [y/N]: ").strip().lower()
if resp == 'y':
os.remove(sub)
# also remove .idx if present alongside .sub
idx = os.path.splitext(sub)[0] + '.idx'
if os.path.exists(idx):
os.remove(idx)
print(f" Deleted: {os.path.basename(sub)}")
def _extract_all_noninteractive(video_path):
"""--extract-all: dump every subtitle track + CC without prompting."""
if not os.path.isfile(video_path):
print(f"File not found: {video_path}", file=sys.stderr)
sys.exit(1)
print(f"Extracting all subtitles from: {video_path}")
streams = probe_subtitle_streams(video_path)
extracted = 0
for s in streams:
codec = s.get('codec_name', '')
stream_idx = s.get('index')
lang = s.get('tags', {}).get('language', '')
out = _sub_out_path(video_path, lang or str(stream_idx))
if codec in _TEXT_SUB_CODECS:
if _extract_text_track(video_path, stream_idx, out):
print(f" Extracted text track {stream_idx} → {os.path.basename(out)}")
extracted += 1
elif codec in {'hdmv_pgs_subtitle', 'pgssub'}:
native_out = os.path.splitext(video_path)[0] + (f'.{lang}' if lang else f'.{stream_idx}') + '.sup'
r = subprocess.run([
'ffmpeg', '-hide_banner', '-loglevel', 'error',
'-i', video_path, '-map', f'0:{stream_idx}', '-c:s', 'copy', '-y', native_out
], timeout=300)
if r.returncode == 0 and os.path.isfile(native_out):
print(f" Extracted PGS track {stream_idx} → {os.path.basename(native_out)}")
extracted += 1
elif codec in _IMAGE_SUB_CODECS:
native_out = os.path.splitext(video_path)[0] + (f'.{lang}' if lang else f'.{stream_idx}') + '.sub'
r = subprocess.run([
'ffmpeg', '-hide_banner', '-loglevel', 'error',
'-i', video_path, '-map', f'0:{stream_idx}', '-c:s', 'copy', '-y', native_out
], timeout=300)
if r.returncode == 0 and os.path.isfile(native_out):
print(f" Extracted VOB SUB track {stream_idx} → {os.path.basename(native_out)}")
extracted += 1
# try ccextractor for broadcast CC
import shutil as _sh
cce = _sh.which('ccextractor') or _sh.which('ccextractorwin')
if cce:
cc_out = _sub_out_path(video_path, 'cc')
r = subprocess.run([cce, video_path, '-o', cc_out], timeout=600,
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
if r.returncode == 0 and os.path.isfile(cc_out):
print(f" Extracted CC → {os.path.basename(cc_out)}")
extracted += 1
print(f"Done. {extracted} track(s) extracted.")
sys.exit(0)
# ---------- Mode 8: Burnt-in subtitle OCR and removal -----------------------
def _probe_video_size(video_path):
"""Return (width, height) of the first video stream."""
try:
r = subprocess.run([
'ffprobe', '-v', 'quiet', '-print_format', 'json',
'-show_streams', '-select_streams', 'v:0', video_path
], capture_output=True, text=True, timeout=15)
s = json.loads(r.stdout)['streams'][0]
return int(s['width']), int(s['height'])
except Exception:
return 1920, 1080
def _video_duration(video_path):
try:
r = subprocess.run([
'ffprobe', '-v', 'quiet', '-show_entries', 'format=duration',
'-print_format', 'json', video_path
], capture_output=True, text=True, timeout=15)
return float(json.loads(r.stdout)['format']['duration'])
except Exception:
return 0.0
def scan_burnt_in_subs(video_path, fps=1, crop_fraction=0.28):
"""
OCR burnt-in subtitles from the bottom crop_fraction of each frame at fps.
Returns (entries, region):
entries = [(start_sec, end_sec, text), ...]
region = (x, y, w, h) estimated black-box in full-frame pixels, or None
"""
if not ensure_easyocr():
return [], None
import easyocr
width, height = _probe_video_size(video_path)
crop_y = int(height * (1.0 - crop_fraction))
crop_h = height - crop_y
duration = _video_duration(video_path)
est = int(duration * fps) if duration else '?'
print(f" Extracting frames at {fps}fps (~{est} frames, bottom {int(crop_fraction*100)}%)...")
import tempfile
with tempfile.TemporaryDirectory() as tmpdir:
frame_pat = os.path.join(tmpdir, 'f_%06d.png')
r = subprocess.run([
'ffmpeg', '-hide_banner', '-loglevel', 'error', '-i', video_path,
'-vf', f'crop={width}:{crop_h}:0:{crop_y},fps={fps}',
frame_pat
], timeout=7200)
if r.returncode != 0:
print(" Frame extraction failed.")
return [], None
frames = sorted(glob.glob(os.path.join(tmpdir, 'f_*.png')))
if not frames:
print(" No frames extracted.")
return [], None
print(f" OCR on {len(frames)} frames (first run downloads ~170 MB model)...")
reader = easyocr.Reader(['en'], verbose=False)
entries = []
current_text = None
start_time = None
all_bboxes = [] # (x1,y1,x2,y2) in full-frame pixels
for i, fp in enumerate(frames):
ts = i / fps
try:
results = reader.readtext(fp, detail=1, paragraph=False)
except Exception:
results = []
texts = []
for (bbox, text, conf) in results:
if conf < 0.35 or not text.strip():
continue
texts.append(text.strip())
bx1 = int(min(p[0] for p in bbox))
by1 = int(min(p[1] for p in bbox)) + crop_y
bx2 = int(max(p[0] for p in bbox))
by2 = int(max(p[1] for p in bbox)) + crop_y
all_bboxes.append((bx1, by1, bx2, by2))
line = postprocess_text(' '.join(texts)) if texts else ''
if line:
if line != current_text:
if current_text is not None:
entries.append((start_time, ts, current_text))
current_text = line
start_time = ts
else:
if current_text is not None:
entries.append((start_time, ts, current_text))
current_text = None
if current_text is not None and start_time is not None:
entries.append((start_time, len(frames) / fps, current_text))
entries = [(s, e, t) for s, e, t in entries if e - s >= 0.4]
region = None
if all_bboxes:
x1 = max(0, min(b[0] for b in all_bboxes) - 20)
y1 = max(0, min(b[1] for b in all_bboxes) - 15)
x2 = min(width, max(b[2] for b in all_bboxes) + 20)
y2 = min(height,max(b[3] for b in all_bboxes) + 15)
region = (x1, y1, x2 - x1, y2 - y1)
return entries, region
def remove_burnt_in_region(video_path, x, y, w, h, output_path):
"""
Remove a rectangular region using ffmpeg delogo filter.
Re-encodes video; audio and subtitle tracks are stream-copied.
Limitation: pixels under the box are gone — delogo blends from
surrounding pixels. Simple/static backgrounds look good; busy action
scenes will show visible blending artifacts.
"""
print(f" Applying delogo: x={x} y={y} w={w} h={h}")
print(" Re-encoding video (libx264 CRF 18) — this will take a while...")
cmd = [
'ffmpeg', '-hide_banner', '-loglevel', 'error',
'-i', video_path,
'-vf', f'delogo=x={x}:y={y}:w={w}:h={h}:show=0',
'-c:v', 'libx264', '-crf', '18', '-preset', 'medium',
'-c:a', 'copy', '-c:s', 'copy',
'-y', output_path
]
try:
r = subprocess.run(cmd, timeout=7200)
return r.returncode == 0 and os.path.isfile(output_path)
except subprocess.TimeoutExpired:
print(" Timed out.")
return False
def _burnt_in_two_file_sync():
"""
Two-file workflow: OCR burnt-in subs from a CC copy, then sync the
resulting SRT against a clean (no burnt-in subs) copy of the same video.
Useful when you have both the CC broadcast version and a clean retail copy.
"""
print("\n TWO-FILE SYNC")
print(" Step 1 of 2 — pick the video WITH burnt-in subtitles (the CC copy):")
vid_files = list_files(VIDEO_EXTS, "video")
if vid_files:
cc_video = pick_file(vid_files, " Choose CC video (0 to cancel)")
else:
cc_video = input(" Path to CC video (0 to cancel): ").strip()
if cc_video == '0':
return
if not cc_video or not os.path.isfile(cc_video):
print(" No valid file selected.")
return
print("\n Step 2 of 2 — pick the CLEAN video (no burnt-in subs):")
if vid_files:
remaining = [f for f in vid_files if f != cc_video]
if remaining:
for i, f in enumerate(remaining, 1):
print(f" {i}: {f}")
clean_video = pick_file(remaining, " Choose clean video (0 to cancel)")
else:
clean_video = input(" Path to clean video (0 to cancel): ").strip()
if clean_video == '0':
return
else:
clean_video = input(" Path to clean video (0 to cancel): ").strip()
if clean_video == '0':
return
if not clean_video or not os.path.isfile(clean_video):
print(" No valid file selected.")
return
print("\n Scan rate (affects timing accuracy and speed):")
print(" 1: 1 fps - ±1s accuracy, fast [default]")
print(" 2: 2 fps - ±0.5s accuracy, slower")
fps = 2 if (input(" Choose [1]: ").strip() == '2') else 1
# ── Step A: OCR the CC video ──────────────────────────────────────────────
print(f"\n Scanning '{os.path.basename(cc_video)}' for burnt-in subtitles...")
entries, _region = scan_burnt_in_subs(cc_video, fps=fps)
if not entries:
print(" No subtitles detected in the CC video. Aborting.")
return
print(f" Detected {len(entries)} subtitle entries.")
import tempfile
fd, raw_srt = tempfile.mkstemp(suffix='-burntocr-raw.srt')
os.close(fd)
with open(raw_srt, 'w', encoding='utf-8') as f:
for i, (s, e, t) in enumerate(entries, 1):
f.write(f"{i}\n{seconds_to_srt(s)} --> {seconds_to_srt(e)}\n{t}\n\n")
# ── Step B: sync the raw SRT against the clean video ─────────────────────
base = os.path.splitext(clean_video)[0]
out_srt = f"{base}-burntocr-synced.srt"
print(f"\n Syncing OCR'd SRT against '{os.path.basename(clean_video)}'...")
if not ensure_ffsubsync():
# No ffsubsync — just write the raw SRT alongside the clean video
import shutil
shutil.copy(raw_srt, out_srt)
os.remove(raw_srt)
print(f" ffsubsync not available — wrote unsynced SRT: {os.path.basename(out_srt)}")
print(" You can sync it later with Mode 1 (SYNC).")
return
ok, _offset = sync_with_ffsubsync(clean_video, raw_srt, out_srt)
os.remove(raw_srt)
if ok and os.path.isfile(out_srt):
kb = os.path.getsize(out_srt) / 1024
print(f"\n Done: {os.path.basename(out_srt)} ({kb:.0f} KB, {len(entries)} entries)")
print(" This SRT is timed to the clean video and ready to use.")
# Offer manual fine-tune: OCR timing is at best ±0.5s so a nudge may help
print("\n Fine-tune timing?")
print(" Subtitle text appears BEFORE you hear it → use a POSITIVE number (+0.5)")
print(" You hear the sound BEFORE the text appears → use a NEGATIVE number (-0.5)")
print(" Enter to skip.")
while True:
resp = input(" Offset seconds [Enter to skip]: ").strip()
if resp == '' or resp == '0':
break
offset = parse_offset(resp)
if offset is None:
print(" Invalid — enter a number like 0.5 or -1.2.")
continue
import shutil
tmp = out_srt + '.bak'
shutil.copy(out_srt, tmp)
shift_srt(tmp, out_srt, offset)
os.remove(tmp)
print(f" Applied {offset:+.3f}s offset to {os.path.basename(out_srt)}")
again = input(" Try another offset? [y/N]: ").strip().lower()
if again != 'y':
break
# Load from the current (already-shifted) file each time — offsets stack
else:
print(" Sync failed. The raw OCR SRT has been discarded.")
print(" Tip: re-run with '1: Transcribe only' on the CC video and sync manually.")
def burnt_in_subs_mode():
"""Mode 8: OCR burnt-in subtitles → SRT and/or remove them from video."""
print("\n BURNSUBS — what would you like to do?")
print(" 1: Transcribe only - OCR burnt-in subs → SRT")
print(" 2: Remove only - erase subtitle band from video (re-encodes)")
print(" 3: Both - transcribe then remove [default]")
print(" 4: Two-file sync - OCR subs from CC copy, sync SRT to clean copy")
print(" 0: Cancel")
while True:
ch = input(" Choose [3]: ").strip() or '3'
if ch in ('0', '1', '2', '3', '4'):
break
print(" Enter 0-4.")
if ch == '0':
return
# ── Option 4: two-file workflow ──────────────────────────────────────────
if ch == '4':
_burnt_in_two_file_sync()
return
# ── Options 1-3: single-file workflow ────────────────────────────────────
print("\n BURNSUBS: Video files in current directory:")
vid_files = list_files(VIDEO_EXTS, "video")
if vid_files:
video = pick_file(vid_files, " Choose video (0 to cancel)")
else:
video = input(" Enter path to video file (0 to cancel): ").strip()
if video == '0':
return
if not video or not os.path.isfile(video):
print(" No valid video selected.")
return
video = _offer_mp4_remux(video)
do_ocr = ch in ('1', '3')
do_remove = ch in ('2', '3')
print("\n Scan rate (affects timing accuracy and speed):")
print(" 1: 1 fps - ±1s accuracy, fast [default]")
print(" 2: 2 fps - ±0.5s accuracy, slower")
fps = 2 if (input(" Choose [1]: ").strip() == '2') else 1
region = None
srt_path = None
if do_ocr:
print(f"\n Scanning for burnt-in subtitles...")
entries, region = scan_burnt_in_subs(video, fps=fps)
if not entries:
print(" No subtitles detected.")
if do_remove and region is None:
print(" Cannot auto-detect removal region. Run transcribe pass first, or enter region manually.")
do_remove = True # fall through to manual entry below
else:
print(f" Detected {len(entries)} subtitle entries.")
base = os.path.splitext(video)[0]
srt_path = f"{base}-burntocr.srt"
with open(srt_path, 'w', encoding='utf-8') as f:
for i, (s, e, t) in enumerate(entries, 1):
f.write(f"{i}\n{seconds_to_srt(s)} --> {seconds_to_srt(e)}\n{t}\n\n")
print(f" SRT: {os.path.basename(srt_path)}")
if do_remove:
if region:
x, y, w, h = region
print(f"\n Auto-detected subtitle region: x={x} y={y} w={w} h={h}")
print(" Note: pixels under the black box cannot be recovered.")
print(" delogo blends from surrounding pixels — looks good on")
print(" simple backgrounds, may show artifacts on busy scenes.")
if input(" Adjust region? [y/N]: ").strip().lower() == 'y':
region = None
if region is None:
vw, vh = _probe_video_size(video)
print(f"\n Enter subtitle region (video is {vw}x{vh}).")
print(" Format: x y width height — e.g. for full-width bottom band: 0 920 1920 100")
while True:
raw = input(" Region (0 to cancel): ").strip()
if raw == '0':
return
try:
x, y, w, h = map(int, raw.split())
region = (x, y, w, h)
break
except ValueError:
print(" Enter four integers.")
x, y, w, h = region
base = os.path.splitext(video)[0]
ext = os.path.splitext(video)[1]
out = f"{base}-clean{ext}"
if input(f"\n Write to {os.path.basename(out)} — proceed? [Y/n]: ").strip().lower() == 'n':
return
if remove_burnt_in_region(video, x, y, w, h, out):
mb = os.path.getsize(out) / 1_048_576
print(f" Done: {os.path.basename(out)} ({mb:.0f} MB)")
if input(" Delete original? [y/N]: ").strip().lower() == 'y':
os.remove(video)
print(f" Deleted: {os.path.basename(video)}")
else:
print(" Removal failed.")
# ---------- Mode 1: Sync (with language detection + transcribe/translate) ----
_LANG_NAMES = {
'id': 'Indonesian', 'ms': 'Malay', 'fr': 'French', 'es': 'Spanish',
'de': 'German', 'it': 'Italian', 'pt': 'Portuguese', 'nl': 'Dutch',
'ru': 'Russian', 'zh-cn': 'Chinese', 'zh-tw': 'Chinese (Traditional)',
'ja': 'Japanese', 'ko': 'Korean', 'ar': 'Arabic', 'th': 'Thai',
'vi': 'Vietnamese', 'pl': 'Polish', 'sv': 'Swedish', 'no': 'Norwegian',
'da': 'Danish', 'fi': 'Finnish', 'tr': 'Turkish', 'cs': 'Czech',
'hu': 'Hungarian', 'ro': 'Romanian', 'uk': 'Ukrainian', 'tl': 'Filipino',
}
def ensure_langdetect():
try:
import langdetect # noqa: F401
return True
except ImportError:
pass
print("\nlangdetect not installed (used for subtitle language detection).")
if input(" Install it now? [Y/n]: ").strip().lower() == 'n':
return False
if not _pip_install('langdetect'):
return False
import importlib
importlib.invalidate_caches()
try:
import langdetect # noqa: F401
return True
except ImportError:
return False
def _srt_detect_language(srt_path):
"""Detect the language of an SRT file.
Returns (lang_code, lang_name) or (None, None) if detection fails.
Uses langdetect for Latin-script languages (Indonesian, Malay, French, etc.)
and falls back to Unicode character analysis for non-Latin scripts.
"""
entries = parse_srt_full(srt_path, limit=60)
if not entries:
return None, None
all_text = ' '.join(t for _, _, t in entries)
letters = [c for c in all_text if c.isalpha()]
if not letters:
return None, None
# Fast path: non-Latin scripts (CJK, Arabic, Cyrillic, etc.)
non_ascii = sum(1 for c in letters if ord(c) > 127)
if (non_ascii / len(letters)) > 0.15:
# Try langdetect for the name, fall back to 'unknown'
try:
if ensure_langdetect():
from langdetect import detect
code = detect(all_text[:2000])
return code, _LANG_NAMES.get(code, code.upper())
except Exception:
pass
return 'xx', 'non-Latin script'
# Latin-script: needs langdetect to distinguish Indonesian/Malay/English/etc.
if not ensure_langdetect():
return None, None
try:
from langdetect import detect, DetectorFactory
DetectorFactory.seed = 0 # make results deterministic
code = detect(all_text[:2000])
if code == 'en':
return 'en', 'English'
return code, _LANG_NAMES.get(code, code.upper())
except Exception:
return None, None
def split_sync_intro_show(video):
"""
Two-pass sync for series episodes with a recurring intro.
Pass 1: sync intro.srt against the video audio → correct timing for the
intro; the synced intro entries are used directly in the output.
Pass 2: extract show audio from where the intro ends, sync the show SRT
(which is treated as show-only content, starting near 00:00:00)
against that clip → offset_B, then shift timestamps to absolute
video time by adding intro_end_video.
The episode SRT should cover only the show content; it does not need
intro subtitles — those come from intro.srt.
"""
if not ensure_ffsubsync():
print(" ffsubsync is required for split sync.")
return
# --- Locate intro.srt ---
intro_srt = 'intro.srt'
if not os.path.isfile(intro_srt):
vid_dir = os.path.dirname(os.path.abspath(video))
intro_srt = os.path.join(vid_dir, 'intro.srt')
if os.path.isfile(intro_srt):
ans = input(f" Found {os.path.basename(intro_srt)} — use it as intro reference? [y/N]: ").strip().lower()
if ans != 'y':
intro_srt = ''
if not intro_srt or not os.path.isfile(intro_srt):
print(" SRT files in current directory:")
srt_candidates = list_files('.srt', 'SRT')
if not srt_candidates:
print(" No SRT files found — cannot run split sync.")
return
intro_srt = pick_file(srt_candidates, " Choose intro SRT (0 to cancel)")
if not intro_srt:
return
print(f" Intro reference: {os.path.basename(intro_srt)}")
# --- Pick show SRT (show content only, need not contain intro lines) ---
print("\n Show SRT files (show content only — intro comes from intro.srt):")
srt_files = [f for f in list_files('.srt', 'SRT') if f != os.path.basename(intro_srt)]
if srt_files:
episode_srt = pick_file(srt_files, " Choose show SRT (0 to cancel)")
else:
episode_srt = input(" Path to show SRT (0 to cancel): ").strip()
if episode_srt == '0':
return
if not episode_srt or not os.path.isfile(episode_srt):
print(" No valid SRT selected.")
return
import tempfile
# ── Pass 1: sync intro against the full video ─────────────────────────────
print(f"\n Pass 1 of 2 — syncing {os.path.basename(intro_srt)} against {os.path.basename(video)}...")
fd, intro_synced_tmp = tempfile.mkstemp(suffix='.srt')
os.close(fd)
ok1, offset_A = sync_with_ffsubsync(video, intro_srt, intro_synced_tmp)
if not ok1 or offset_A is None:
print(" Intro sync failed — cannot determine split point.")
try: os.remove(intro_synced_tmp)
except OSError: pass
return
print(f" Intro offset: {offset_A:+.3f}s")
# The synced intro entries already have correct absolute timestamps.
intro_synced_entries = parse_srt_full(intro_synced_tmp)
try: os.remove(intro_synced_tmp)
except OSError: pass
if not intro_synced_entries:
print(" Could not read synced intro SRT — aborting.")
return
intro_end_video = max(e for _, e, _ in intro_synced_entries)
print(f" Intro ends at {seconds_to_srt(intro_end_video)} in video")
print(f" Intro: {len(intro_synced_entries)} entries ready")
# Write a preview file so the user can open it and check before deciding
intro_preview = os.path.splitext(intro_srt)[0] + '-synced-preview.srt'
with open(intro_preview, 'w', encoding='utf-8') as f:
for i, (s, e, t) in enumerate(intro_synced_entries, 1):
f.write(f"{i}\n{seconds_to_srt(s)} --> {seconds_to_srt(e)}\n{t}\n\n")
print(f" Preview written: {os.path.basename(intro_preview)}")
print(" Framerate correction (if needed) was applied automatically.")
print(" Open the preview in a text editor or subtitle viewer to check timing.")
input(" Press Enter when ready to continue...")
# Optional manual nudge on the intro before combining
print("\n Intro timing fine-tune (or Enter to skip):")
print(" Subtitle text appears BEFORE you hear it → positive number (+3.0)")
print(" You hear the sound BEFORE the text appears → negative number (-3.0)")
while True:
resp = input(" Intro offset seconds [Enter to skip]: ").strip()
if resp == '' or resp == '0':
break
extra = parse_offset(resp)
if extra is None:
print(" Invalid — enter a number like 3.0 or -1.5.")
continue
intro_synced_entries = [
(max(0.0, s + extra), max(0.0, e + extra), t)
for s, e, t in intro_synced_entries
]
intro_end_video = max(e for _, e, _ in intro_synced_entries)
with open(intro_preview, 'w', encoding='utf-8') as f:
for i, (s, e, t) in enumerate(intro_synced_entries, 1):
f.write(f"{i}\n{seconds_to_srt(s)} --> {seconds_to_srt(e)}\n{t}\n\n")
print(f" Applied {extra:+.3f}s — intro now ends at {seconds_to_srt(intro_end_video)}")
print(f" Preview updated: {os.path.basename(intro_preview)}")
again = input(" Try another offset? [y/N]: ").strip().lower()
if again != 'y':
break
# ── Extract show audio from intro_end onwards ─────────────────────────────
print(f"\n Extracting show audio from {seconds_to_srt(intro_end_video)}...")
fd2, show_wav = tempfile.mkstemp(suffix='.wav')
os.close(fd2)
r = subprocess.run([
'ffmpeg', '-hide_banner', '-loglevel', 'error',
'-i', video, '-ss', str(intro_end_video),
'-vn', '-ac', '1', '-ar', '16000', '-y', show_wav
], timeout=600)
if r.returncode != 0:
print(" Failed to extract show audio — aborting.")
try: os.remove(show_wav)
except OSError: pass
return
# ── Pass 2: sync show SRT against the show audio clip ────────────────────
# ffsubsync finds the best alignment regardless of what offset the show SRT
# currently has; output timestamps are relative to the clip start (i.e.
# relative to intro_end_video).
show_ep = parse_srt_full(episode_srt)
print(f" Pass 2 of 2 — syncing {len(show_ep)} show entries against show audio...")
fd3, show_synced_tmp = tempfile.mkstemp(suffix='.srt')
os.close(fd3)
ok2, offset_B = sync_with_ffsubsync(show_wav, episode_srt, show_synced_tmp)
try: os.remove(show_wav)
except OSError: pass
if ok2 and offset_B is not None:
print(f" Show offset: {offset_B:+.3f}s (relative to intro end)")
show_synced_entries = parse_srt_full(show_synced_tmp)
else:
print(" Show sync failed — writing show entries unsynced as fallback.")
show_synced_entries = show_ep
try: os.remove(show_synced_tmp)
except OSError: pass
# ── Merge: intro (absolute) + show (relative → absolute) ─────────────────
base = os.path.splitext(episode_srt)[0]
out = f"{base}-splitsync.srt"
with open(out, 'w', encoding='utf-8') as f:
idx = 1
# Intro: timestamps already correct from pass 1
for s, e, t in intro_synced_entries:
f.write(f"{idx}\n{seconds_to_srt(s)} --> {seconds_to_srt(e)}\n{t}\n\n")
idx += 1
# Show: add intro_end_video to convert clip-relative → absolute video time
for s, e, t in show_synced_entries:
ws = s + intro_end_video
we = max(ws + 0.1, e + intro_end_video)
f.write(f"{idx}\n{seconds_to_srt(ws)} --> {seconds_to_srt(we)}\n{t}\n\n")
idx += 1
kb = os.path.getsize(out) / 1024
print(f"\n Done: {os.path.basename(out)} ({kb:.0f} KB, {idx-1} entries)")
print(f" Intro: {len(intro_synced_entries)} entries (offset {offset_A:+.3f}s)")
if ok2 and offset_B is not None:
print(f" Show: {len(show_synced_entries)} entries (offset {offset_B:+.3f}s from intro end)")
def sync_mode():
global WHISPER_MODEL, WHISPER_TASK, WHISPER_LANGUAGE
# --- Pick video ---
print("\n SYNC: Video files in current directory:")
vid_files = list_files(VIDEO_EXTS, "video")
if vid_files:
video = pick_file(vid_files, " Choose video by number or filename")
else:
video = input(" Enter path to video file (0 to cancel): ").strip()
if video == '0':
return
if not video or not os.path.isfile(video):
print(" No valid video selected.")
return
ffsubsync_ok = ensure_ffsubsync()
whisper_ok = WHISPER_AVAILABLE # don't install just to show the menu
# --- Sync method ---
print("\n Sync method:")
print(" f: ffsubsync only (fast, recommended) [default]")
print(" w: Whisper only (speech recognition)")
print(" b: Both - ffsubsync + Whisper cross-check")
print(" m: Manual offset (enter seconds yourself)")
print(" s: Split sync (intro + show have different offsets, uses intro.srt)")
print(" 0: Cancel")
while True:
ch = input(" Choose or Enter for default: ").strip().lower()
if ch == '':
ch = 'f'
break
if ch in ('f', 'w', 'b', 'm', 's', '0'):
break
print(" Enter f, w, b, m, s or 0.")
if ch == '0':
return
if ch == 's':
split_sync_intro_show(video)
return
if ch in ('w', 'b'):
whisper_ok = ensure_whisper()
if not whisper_ok:
print(" Whisper required for this method.")
return
print("\n Whisper model:")
for k, (name, desc) in WHISPER_MODELS.items():
marker = " <-- default" if name == WHISPER_MODEL else ""
print(f" {k}: {name:20s} {desc}{marker}")
choice = input(" Choose model [Enter for default]: ").strip()
if choice in WHISPER_MODELS:
WHISPER_MODEL = WHISPER_MODELS[choice][0]
print(f" Using: {WHISPER_MODEL}\n")
# --- Pick SRT ---
print("\n SRT files in current directory:")
srt_files = list_files('.srt', "SRT")
if srt_files:
src = pick_file(srt_files, " Choose SRT by number or filename")
else:
src = input(" Enter path to .srt file (0 to cancel): ").strip()
if src == '0':
return
if not src or not os.path.isfile(src):
print(" No valid SRT selected.")
return
# Detect SRT language from the actual subtitle text — offer English if non-English
also_english = False
lang_code, lang_name = _srt_detect_language(src)
if lang_code and lang_code != 'en':
print(f"\n Detected language: {lang_name}.")
if ensure_whisper():
whisper_ok = True
also_english = input(
f" Also generate an English SRT via Whisper translate after sync? [Y/n]: "
).strip().lower() != 'n'
# --- Sync ---
base = os.path.splitext(src)[0]
out = f"{base}-synced.srt"
print(f"\n Syncing -> {os.path.basename(out)}")
synced_ok = False
if ch == 'f':
ok, offset = sync_with_ffsubsync(video, src, out)
if ok:
if offset is not None:
print(f" Offset applied: {offset:+.3f} s")
print(f" Done: {os.path.basename(out)}")
synced_ok = True
else:
print(" ffsubsync failed.")
while True:
resp = input(" Enter offset manually (seconds, e.g. -3.5) or Enter to skip: ").strip()
if resp == '':
break
offset = parse_offset(resp)
if offset is None:
print(" Invalid.")
else:
shift_srt(src, out, offset)
print(f" Written -> {os.path.basename(out)}")
synced_ok = True
break
elif ch == 'w':
offset, n_matches, spread, err = compute_offset_whisper(src, video, WHISPER_MODEL)
if not err:
quality = "good" if spread < 2.0 else "moderate" if spread < 5.0 else "low"
print(f" Whisper offset: {offset:+.3f} s ({n_matches} matches, spread {spread:.1f}s, {quality})")
shift_srt(src, out, offset)
print(f" Done: {os.path.basename(out)}")
synced_ok = True
else:
print(f" Whisper failed: {err}")
elif ch == 'm':
print(" Subtitle text appears BEFORE you hear it → use a POSITIVE number (+0.52)")
print(" You hear the sound BEFORE the text appears → use a NEGATIVE number (-0.52)")
print(" Enter 0 or blank to cancel.")
last_offset = 0.0
while True:
hint = f" Offset seconds [last: {last_offset:+.3f}]: "
resp = input(hint).strip()
if resp in ('0', ''):
break
offset = parse_offset(resp)
if offset is None:
print(" Invalid — enter a number like 1.5 or -0.52.")
continue
last_offset = offset
shift_srt(src, out, offset)
print(f" Written -> {os.path.basename(out)}")
synced_ok = True
again = input(" Try another offset? [y/N]: ").strip().lower()
if again != 'y':
break
# re-apply to original each time so offsets don't stack
print(" (applying to original each time — offsets do not stack)")
else: # b
if sync_single(video, src, out, ffsubsync_ok, whisper_ok, interactive=True):
print(f" Done: {os.path.basename(out)}")
synced_ok = True
else:
while True:
resp = input("\n All methods failed. Enter offset manually or Enter to skip: ").strip()
if resp == '':
break
offset = parse_offset(resp)
if offset is None:
print(" Invalid.")
else:
shift_srt(src, out, offset)
print(f" Written -> {os.path.basename(out)}")
synced_ok = True
break
# --- Also generate English SRT? ---
if also_english and whisper_ok:
print("\n Generating English SRT via Whisper translate...")
WHISPER_TASK = 'translate'
WHISPER_LANGUAGE = None
final = generate_and_sync(video, WHISPER_MODEL, ffsubsync_ok=ffsubsync_ok)
if final:
print(f" English SRT: {os.path.basename(final)}")
if synced_ok:
offer_rename(video)
# ---------- Main -------------------------------------------------------------
def main():
global WHISPER_MODEL, WHISPER_LANGUAGE, WHISPER_TASK
if '--translate' in sys.argv:
WHISPER_TASK = 'translate'
WHISPER_LANGUAGE = None # auto-detect source; --lang overrides below
print("Translate mode: Whisper will output English regardless of source language.")
if '--extract-all' in sys.argv:
idx = sys.argv.index('--extract-all')
if idx + 1 < len(sys.argv):
_extract_all_noninteractive(sys.argv[idx + 1])
else:
print("--extract-all requires a file path.", file=sys.stderr)
sys.exit(1)
if '--lang' in sys.argv:
idx = sys.argv.index('--lang')
if idx + 1 < len(sys.argv):
WHISPER_LANGUAGE = sys.argv[idx + 1]
print(f"Language override: {WHISPER_LANGUAGE}")
else:
print("--lang requires a language code (e.g. --lang fr). Using default.")
elif '--lang-auto' in sys.argv:
WHISPER_LANGUAGE = None
print("Language: auto-detect")
while True:
print("\nWhat would you like to do?")
print(" 1: SYNC - sync an existing SRT to the video")
gen_label = "translate foreign audio → English SRT" if WHISPER_TASK == 'translate' \
else "create a new SRT by transcribing with Whisper"
print(f" 2: GENERATE - {gen_label}")
print(" 3: BATCH - sync all video+SRT pairs in this directory")
print(" 4: RENAME - rename video + SRT to Plex format")
print(" 5: EXTRACT - extract embedded subtitles / CC to SRT")
print(" 6: REMUX - convert MP4 → MKV (stream copy, no re-encode)")
print(" 7: EMBED - soft-mux subtitle file into video (mkvmerge)")
print(" 8: BURNSUBS - OCR burnt-in subs → SRT and/or erase from video")
print(" 0: Exit")
while True:
mode = input("Choose: ").strip()
if mode in ('0', '1', '2', '3', '4', '5', '6', '7', '8'):
break
print("Please enter 0-8.")
if mode == '0':
print("Goodbye.")
break
# ---- Mode 4: Rename (has its own sub-menu loop) ---------------------
if mode == '4':
ocr_ok = ensure_easyocr()
rename_menu(ocr_ok)
continue
# ---- Mode 5: Extract subtitles --------------------------------------
if mode == '5':
extract_subs_mode()
continue
# ---- Mode 6: Remux MP4 → MKV ----------------------------------------
if mode == '6':
remux_mp4_to_mkv()
continue
# ---- Mode 7: Embed subtitle into video ------------------------------
if mode == '7':
embed_subs_mode()
continue
# ---- Mode 8: Burnt-in subtitle OCR / removal ------------------------
if mode == '8':
burnt_in_subs_mode()
continue
# ---- Mode 1: Sync / transcribe / translate --------------------------
if mode == '1':
sync_mode()
continue
# ---- Mode 2: Generate SRT -------------------------------------------
whisper_ok = ensure_whisper()
ffsubsync_ok = ensure_ffsubsync()
if not whisper_ok:
print("Whisper is required to generate an SRT.")
continue
if whisper_ok:
print("\nWhisper model (larger = more accurate, more RAM, slower first load):")
for k, (name, desc) in WHISPER_MODELS.items():
marker = " <-- default" if name == WHISPER_MODEL else ""
print(f" {k}: {name:20s} {desc}{marker}")
choice = input("Choose model [Enter for default]: ").strip()
if choice in WHISPER_MODELS:
WHISPER_MODEL = WHISPER_MODELS[choice][0]
print(f" Using: {WHISPER_MODEL}\n")
# ---- Mode 3: Batch sync ---------------------------------------------
if mode == '3':
batch_sync(ffsubsync_ok, whisper_ok)
continue
print("\nVideo files in current directory:")
vid_files = list_files(VIDEO_EXTS, "video")
if vid_files:
video = pick_file(vid_files, "Choose video by number or filename")
else:
video = input("Enter path to video file (0 to cancel): ").strip()
if video == '0':
continue
if not video or not os.path.isfile(video):
print("No valid video selected.")
continue
final = generate_and_sync(video, WHISPER_MODEL, ffsubsync_ok=ffsubsync_ok)
if final:
print(f"\nDone - final SRT: {final}")
offer_rename(video)
if __name__ == '__main__':
main()