Files
k3nnyandClaude Opus 5.5 8145e5e985 Initial version of arte-dl
Wrapper around yt-dlp to download whole arte.tv series into
Series (year)/Season XX/Series - SxxEyy - Title.mkv:

- resolve series / season / episode URLs through the Arte API
- pick video, audio (VO, VF, AD...) and subtitle tracks from a TOML config
- download with yt-dlp, convert WebVTT to SRT (ffmpeg 4.4 yields empty
  subtitles from Arte's CRLF files), remux with track languages, titles,
  default/forced flags and episode tags
- optional TMDB matching: series name and year, numbering (including
  multi-episode files when Arte merges two episodes), localized titles
  and synopses following a language priority list

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
2026-09-30 20:32:17 +02:00

55 lines
1.9 KiB
Python
Raw Permalink Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Minimal WebVTT -> SRT conversion.
Arte serves VTT files with CRLF line endings and STYLE blocks, which older
ffmpeg releases (e.g. 4.4, also used by yt-dlp --convert-subs) silently turn
into empty subtitles. Converting ourselves avoids depending on the ffmpeg version.
"""
from __future__ import annotations
import html
import re
_TIMING_RE = re.compile(
r'^\s*(?P<start>(?:\d+:)?\d{2}:\d{2}[.,]\d{3})\s+-->\s+(?P<end>(?:\d+:)?\d{2}:\d{2}[.,]\d{3})')
_KEEP_TAGS_RE = re.compile(r'</?(?:i|b|u)>')
_TAG_RE = re.compile(r'<[^>]*>')
def _timestamp(ts: str) -> str:
ts = ts.replace(',', '.')
parts = ts.split(':')
if len(parts) == 2:
parts.insert(0, '0')
h, m, s = parts
sec, ms = s.split('.')
return f'{int(h):02d}:{int(m):02d}:{int(sec):02d},{ms}'
def _clean(line: str) -> str:
# Keep <i>, <b>, <u>; drop <c.class>, <v Speaker>, <lang>, inline timestamps...
kept = []
def stash(m):
kept.append(m[0])
return f'\x00{len(kept) - 1}\x00'
line = _KEEP_TAGS_RE.sub(stash, line)
line = html.unescape(_TAG_RE.sub('', line)).replace(' ', ' ')
return re.sub(r'\x00(\d+)\x00', lambda m: kept[int(m[1])], line)
def vtt_to_srt(vtt: str) -> str:
text = vtt.lstrip('').replace('\r\n', '\n').replace('\r', '\n')
cues = []
for block in re.split(r'\n{2,}', text):
lines = block.strip('\n').split('\n')
timing_idx = next((i for i, l in enumerate(lines) if _TIMING_RE.match(l)), None)
if timing_idx is None: # header, STYLE, NOTE, REGION blocks
continue
m = _TIMING_RE.match(lines[timing_idx])
payload = [c for c in (_clean(l).strip() for l in lines[timing_idx + 1:]) if c]
if payload:
cues.append((_timestamp(m['start']), _timestamp(m['end']), payload))
return ''.join(f'{n}\n{start} --> {end}\n' + '\n'.join(payload) + '\n\n'
for n, (start, end, payload) in enumerate(cues, 1))