From 8145e5e985f61b641b58cc08587c7ee542066edd Mon Sep 17 00:00:00 2001 From: k3nny Date: Wed, 30 Sep 2026 20:32:17 +0200 Subject: [PATCH] Initial version of arte-dl Wrapper around yt-dlp to download whole arte.tv series into Series (year)/Season XX/Series - SxxEyy - Title.mkv: - resolve series / season / episode URLs through the Arte API - pick video, audio (VO, VF, AD...) and subtitle tracks from a TOML config - download with yt-dlp, convert WebVTT to SRT (ffmpeg 4.4 yields empty subtitles from Arte's CRLF files), remux with track languages, titles, default/forced flags and episode tags - optional TMDB matching: series name and year, numbering (including multi-episode files when Arte merges two episodes), localized titles and synopses following a language priority list Co-Authored-By: Claude Opus 5.5 (1M context) --- .gitignore | 6 + README.md | 133 ++++++++++++++++ arte_dl/__init__.py | 3 + arte_dl/__main__.py | 3 + arte_dl/arte_api.py | 242 +++++++++++++++++++++++++++++ arte_dl/cli.py | 154 +++++++++++++++++++ arte_dl/config.py | 131 ++++++++++++++++ arte_dl/download.py | 177 +++++++++++++++++++++ arte_dl/metadata.py | 330 ++++++++++++++++++++++++++++++++++++++++ arte_dl/selection.py | 195 ++++++++++++++++++++++++ arte_dl/subtitles.py | 54 +++++++ config.example.toml | 49 ++++++ pyproject.toml | 23 +++ tests/test_metadata.py | 189 +++++++++++++++++++++++ tests/test_misc.py | 78 ++++++++++ tests/test_selection.py | 96 ++++++++++++ 16 files changed, 1863 insertions(+) create mode 100644 .gitignore create mode 100644 README.md create mode 100644 arte_dl/__init__.py create mode 100644 arte_dl/__main__.py create mode 100644 arte_dl/arte_api.py create mode 100644 arte_dl/cli.py create mode 100644 arte_dl/config.py create mode 100644 arte_dl/download.py create mode 100644 arte_dl/metadata.py create mode 100644 arte_dl/selection.py create mode 100644 arte_dl/subtitles.py create mode 100644 config.example.toml create mode 100644 pyproject.toml create mode 100644 tests/test_metadata.py create mode 100644 tests/test_misc.py create mode 100644 tests/test_selection.py diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..3ce59c4 --- /dev/null +++ b/.gitignore @@ -0,0 +1,6 @@ +.venv/ +__pycache__/ +*.egg-info/ +.pytest_cache/ +dist/ +.arte-dl-tmp/ diff --git a/README.md b/README.md new file mode 100644 index 0000000..51871d0 --- /dev/null +++ b/README.md @@ -0,0 +1,133 @@ +# arte-dl + +Wrapper autour de [yt-dlp](https://github.com/yt-dlp/yt-dlp) pour télécharger une série +arte.tv complète (toutes ses saisons) en MKV, selon des préférences de qualité, de pistes +audio et de sous-titres, et la ranger dans une arborescence de vidéothèque (Plex / Jellyfin / Kodi) : + +``` +Meurtres à Sandhamn (2010)/ +├── Season 01/ +│ ├── Meurtres à Sandhamn - S01E01 - Enquête 1 - La reine de la Baltique.mkv +│ └── … +└── Season 06/ + └── Meurtres à Sandhamn - S06E01-E02 - Le prix à payer.mkv +``` + +Avec une clé [TMDB](https://www.themoviedb.org/), le nom de la série, l'année, la numérotation +et les titres d'épisodes sont alignés sur TMDB (la référence de Jellyfin, et une source de Plex). + +## Installation + +Nécessite Python ≥ 3.10 et `ffmpeg` dans le `PATH`. + +```sh +uv tool install . # ou : pipx install . +arte-dl --version +``` + +Pour mettre à jour yt-dlp (Arte change régulièrement son API) : `uv tool upgrade arte-dl`. + +## Utilisation + +```sh +# Toute la série +arte-dl https://www.arte.tv/fr/videos/RC-027900/the-hack-sur-ecoute/ + +# Voir l'arborescence prévue sans rien télécharger +arte-dl --list https://www.arte.tv/fr/videos/RC-022391/meurtres-a-sandhamn/ + +# Voir aussi les pistes retenues pour chaque épisode +arte-dl --dry-run -s 4 https://www.arte.tv/fr/videos/RC-022391/meurtres-a-sandhamn/ + +# Certaines saisons / certains épisodes, dans un autre dossier +arte-dl -s 1,3-5 -e 1-2 -o ~/Vidéos/Séries https://www.arte.tv/fr/videos/RC-022391/meurtres-a-sandhamn/ + +# Corriger l'identification TMDB (mémorisée pour les fois suivantes) +arte-dl --list --tmdb-id 55270 https://www.arte.tv/fr/videos/RC-022391/meurtres-a-sandhamn/ +``` + +`-s` / `-e` portent sur la numérotation finale (celle de TMDB quand elle est utilisée), +celle qu'affiche `--list`. + +Types d'URL acceptés : + +| URL | Téléchargé | +|---------------------------------------|-------------------------------------| +| série `…/videos/RC-xxxxxx/…` | toutes les saisons disponibles | +| saison `…/videos/RC-xxxxxx/…` | cette saison | +| épisode `…/videos/059534-001-A/…` | cet épisode, bien numéroté/rangé | + +Les fichiers déjà présents sont ignorés (`--force` pour les retélécharger) : relancer la +commande reprend simplement là où elle s'était arrêtée, ou récupère les nouveaux épisodes. + +## Configuration + +Fichier TOML : `~/.config/arte-dl/config.toml` par défaut, ou `-c fichier.toml`. +Voir [`config.example.toml`](config.example.toml) pour toutes les options. Exemple : + +```toml +[output] +directory = "~/Vidéos/Séries" + +[video] +max_height = 1080 +codecs = ["hevc", "avc"] + +[audio] +tracks = ["original", "fr"] # VO par défaut + VF + +[subtitles] +tracks = ["fr", "fr-forced"] +default = "auto" + +[metadata] +api_key = "…" # ou variable d'environnement TMDB_API_KEY +languages = ["fr-FR", "arte", "en-US"] # priorité pour les noms, titres et résumés +``` + +## Métadonnées TMDB + +Clé gratuite : créer un compte sur themoviedb.org, puis *Paramètres → API*. La clé API (v3) +comme le jeton d'accès en lecture (v4) conviennent. + +- **Identification de la série** : recherche par titre original et titre Arte, départagée + par l'année de production et la langue originale fournies par Arte. En cas de doute + (score faible ou deux candidats proches), rien n'est deviné : les données Arte sont gardées + et `--tmdb-id` permet de trancher. L'association est mémorisée dans + `~/.local/share/arte-dl/tmdb-ids.json`. +- **Numérotation** : saison par saison, les épisodes Arte sont associés aux épisodes TMDB : + - même nombre d'épisodes : correspondance directe ; + - Arte fusionne des épisodes (ex. Meurtres à Sandhamn saison 6 : 4 × 88 min sur Arte, + 8 × 45 min sur TMDB) : le fichier est nommé `S06E01-E02`, format multi-épisode reconnu + par Plex et Jellyfin. Les durées sont vérifiées ; + - sinon, alignement par durée si toute la saison est disponible ; + - à défaut, ou si la saison n'existe pas sur TMDB, la numérotation Arte est conservée + (et signalée). +- **Titres et résumés** : pris dans l'ordre de `languages`. Une langue qui n'a pas le titre + (ou seulement « Épisode 3 ») passe à la suivante. Pour deux épisodes fusionnés, + « X (part 1) » + « X (part 2) » donnent « X ». + +## Fonctionnement + +1. **Structure** : la série, ses saisons et épisodes sont lus via l'API Arte (celle + qu'utilise yt-dlp). Le numéro de saison vient du titre (« Saison 4 »), le numéro d'épisode + du « (1/3) », le titre de l'épisode du sous-titre Arte (à défaut « Épisode N »), puis + tout cela est corrigé par TMDB si une clé est configurée. +2. **Sélection** : pour chaque épisode, yt-dlp extrait les formats ; arte-dl choisit la vidéo + et les pistes audio d'après la configuration. Les identifiants de format Arte + (`VF-STF-audio_0-suédois__VO_`…) varient d'un épisode à l'autre, d'où la sélection + par langue et par type plutôt que par identifiant. +3. **Téléchargement** : yt-dlp télécharge et fusionne vidéo + audios, et récupère les + sous-titres WebVTT. +4. **Remux final** (ffmpeg) : sous-titres convertis en SRT, langue et titre de chaque piste, + pistes par défaut / forcées / SDH, tags série / saison / épisode. Le fichier est écrit en + `.part.mkv` puis renommé, donc un fichier `.mkv` présent est toujours complet. + +La conversion VTT → SRT est faite en Python : les VTT d'Arte (fins de ligne CRLF) donnent +des sous-titres **vides** avec ffmpeg 4.4 (celui d'Ubuntu 22.04), y compris via `yt-dlp --convert-subs srt`. + +## Tests + +```sh +uv venv && uv pip install -e '.[dev]' && .venv/bin/pytest +``` diff --git a/arte_dl/__init__.py b/arte_dl/__init__.py new file mode 100644 index 0000000..236567c --- /dev/null +++ b/arte_dl/__init__.py @@ -0,0 +1,3 @@ +"""arte-dl: download whole arte.tv series with yt-dlp.""" + +__version__ = '0.1.0' diff --git a/arte_dl/__main__.py b/arte_dl/__main__.py new file mode 100644 index 0000000..eb53e2f --- /dev/null +++ b/arte_dl/__main__.py @@ -0,0 +1,3 @@ +from .cli import main + +raise SystemExit(main()) diff --git a/arte_dl/arte_api.py b/arte_dl/arte_api.py new file mode 100644 index 0000000..7cf8ec0 --- /dev/null +++ b/arte_dl/arte_api.py @@ -0,0 +1,242 @@ +"""Resolve arte.tv URLs into a Series -> Season -> Episode structure. + +yt-dlp can flatten an RC-xxxxxx collection, but it loses the season structure +and episode numbering, so we query the same Arte APIs it uses ourselves. +""" +from __future__ import annotations + +import json +import re +import time +import urllib.error +import urllib.request +from dataclasses import dataclass, field + +API_PLAYER = 'https://api.arte.tv/api/player/v2' +API_OPA = 'https://api.arte.tv/api/opa/v3' + +try: # keep the public token in sync with yt-dlp when it rotates + from yt_dlp.extractor.arte import ArteTVPlaylistIE + _OPA_TOKEN = ArteTVPlaylistIE._API_TOKEN +except (ImportError, AttributeError): + _OPA_TOKEN = 'Nzc1Yjc1ZjJkYjk1NWFhN2I2MWEwMmRlMzAzNjI5NmU3NWU3ODg4ODJjOWMxNTMxYzEzZGRjYjg2ZGE4MmIwOA' + +URL_RE = re.compile( + r'arte\.tv/(?Pfr|de|en|es|it|pl)/videos/(?PRC-\d{6}|\d{6}-\d{3}-[AF])') +EPISODE_ID_RE = re.compile(r'^\d{6}-\d{3}-[AF]$') +COLLECTION_ID_RE = re.compile(r'RC-\d{6}') +SEASON_RE = re.compile(r'\b(?:saison|staffel|season|temporada|stagione|sezon)\s*(\d+)', re.I) +EPISODE_NUMBER_RE = re.compile(r'\((\d+)\s*/\s*(\d+)\)\s*$') + + +class ArteError(Exception): + pass + + +@dataclass +class Episode: + id: str + url: str + series: str + season: int + number: int + title: str + description: str | None = None + duration: int | None = None # seconds + total: int | None = None # episodes in the season, from Arte's "(n/m)" + last_number: int | None = None # set when the file holds several episodes (S06E01-E02) + + def __post_init__(self): + # Arte's own numbering, kept when metadata matching renumbers the episode + self.arte_season, self.arte_number, self.arte_title = self.season, self.number, self.title + + @property + def numbers(self) -> list[int]: + return list(range(self.number, (self.last_number or self.number) + 1)) + + @property + def label(self) -> str: + return f'S{self.season:02d}' + '-'.join(f'E{n:02d}' for n in ( + [self.number, self.last_number] if self.last_number else [self.number])) + + +@dataclass +class Season: + id: str + number: int + title: str + episodes: list[Episode] = field(default_factory=list) + + +@dataclass +class Series: + id: str + title: str + lang: str + seasons: list[Season] = field(default_factory=list) + unavailable: list[str] = field(default_factory=list) # season ids with nothing online + original_title: str | None = None + original_language: str | None = None # ISO 639-1 + year: int | None = None + tmdb_id: int | None = None + tvdb_id: int | None = None + + @property + def episodes(self) -> list[Episode]: + return [e for s in self.seasons for e in s.episodes] + + +def parse_url(url: str) -> tuple[str, str]: + m = URL_RE.search(url) + if not m: + raise ArteError(f'Not an arte.tv series or episode URL: {url}') + return m['lang'], m['id'] + + +class ArteClient: + def __init__(self, lang: str, retries: int = 6): + self.lang = lang + self.retries = retries + + def _get(self, url: str, headers: dict[str, str]) -> dict | None: + """GET a JSON document; None on 404. Retries on rate limiting / server errors.""" + req = urllib.request.Request(url, headers={'User-Agent': 'Mozilla/5.0', **headers}) + for attempt in range(self.retries + 1): + try: + with urllib.request.urlopen(req, timeout=30) as resp: + return json.load(resp) + except urllib.error.HTTPError as e: + if e.code == 404: + return None + if (e.code == 429 or e.code >= 500) and attempt < self.retries: + delay = int(e.headers.get('Retry-After') or 0) or 2 ** attempt + time.sleep(min(delay, 60)) + continue + raise ArteError(f'HTTP {e.code} for {url}') from e + except urllib.error.URLError as e: + if attempt < self.retries: + time.sleep(2 ** attempt) + continue + raise ArteError(f'{e.reason} for {url}') from e + raise AssertionError('unreachable') + + def program(self, program_id: str) -> dict: + data = self._get(f'{API_OPA}/programs/{self.lang}/{program_id}', + {'Authorization': f'Bearer {_OPA_TOKEN}'}) + programs = (data or {}).get('programs') or [] + if not programs: + raise ArteError(f'Unknown Arte program {program_id}') + return programs[0] + + def playlist(self, collection_id: str) -> dict | None: + data = self._get(f'{API_PLAYER}/playlist/{self.lang}/{collection_id}', + {'x-validated-age': '18'}) + return ((data or {}).get('data') or {}).get('attributes') + + # --- resolution ------------------------------------------------------- + + def resolve(self, url: str) -> Series: + """Series URL -> every season; season URL -> that season; episode URL -> that episode.""" + _, pid = parse_url(url) + if EPISODE_ID_RE.match(pid): + return self._resolve_episode(pid) + + prog = self.program(pid) + if prog.get('catalogType') == 'SEASON': + series_id = next((p for p in prog.get('parents') or [] + if COLLECTION_ID_RE.fullmatch(p) and p != pid), None) + if series_id: + return self._series(series_id, only_season=pid) + return self._series(pid) # orphan season: treat as its own series + return self._series(pid, prog=prog) + + def _resolve_episode(self, episode_id: str) -> Series: + prog = self.program(episode_id) + collections = prog.get('collections') or [] + season = next((c for c in collections if c.get('catalogType') == 'SEASON'), None) + coll = season or (collections[0] if collections else None) + series_id = None + if coll: + m = COLLECTION_ID_RE.search(coll.get('url') or '') + series_id = m[0] if m else coll.get('collectionId') + if series_id: + series = self._series(series_id, only_season=season and season.get('collectionId'), + only_episode=episode_id) + if any(s.episodes for s in series.seasons): + return series + # Standalone program, or not found in its collection: single episode, season 1. + title = prog.get('title') or episode_id + ep = Episode(id=episode_id, url=f'https://www.arte.tv/{self.lang}/videos/{episode_id}/', + series=title, season=1, number=1, title=prog.get('subtitle') or title, + description=prog.get('shortDescription')) + return Series(id=episode_id, title=title, lang=self.lang, + seasons=[Season(id=episode_id, number=1, title=title, episodes=[ep])], + original_title=(prog.get('originalTitle') or '').strip(' ()') or None, + original_language=(prog.get('originalLanguage') or {}).get('iso6391Code'), + year=prog.get('productionYear')) + + def _series(self, series_id: str, *, prog: dict | None = None, + only_season: str | None = None, only_episode: str | None = None) -> Series: + prog = prog or self.program(series_id) + series = Series(id=series_id, title=(prog.get('title') or series_id).strip(), lang=self.lang, + original_title=(prog.get('originalTitle') or '').strip(' ()') or None, + original_language=(prog.get('originalLanguage') or {}).get('iso6391Code'), + year=prog.get('productionYear')) + + refs = [c for c in prog.get('children') or [] if c.get('catalogType') == 'SEASON'] + refs.sort(key=lambda c: c.get('order') or 0) + if not refs: # mini-series: the collection itself holds the episodes + refs = [{'programId': series_id}] + + for idx, ref in enumerate(refs, 1): + sid = ref['programId'] + if only_season and sid != only_season: + continue + attrs = self.playlist(sid) + items = (attrs or {}).get('items') or [] + if not items and sid == series_id: + items = self._fallback_items(prog) + if not items: + series.unavailable.append(sid) + continue + stitle = ((attrs or {}).get('metadata') or {}).get('title') or series.title + m = SEASON_RE.search(stitle) + number = int(m[1]) if m else (ref.get('order') or idx) + season = Season(id=sid, number=number, title=stitle) + season.episodes = self._episodes(items, series.title, number, only_episode) + if season.episodes: + series.seasons.append(season) + series.seasons.sort(key=lambda s: s.number) + return series + + def _fallback_items(self, prog: dict) -> list[dict]: + """Build playlist-like items from the OPA 'videos' list.""" + return [{'providerId': v.get('programId'), 'title': v.get('title'), + 'subtitle': v.get('subtitle'), 'link': {'url': v.get('url')}, + 'description': v.get('shortDescription')} + for v in prog.get('videos') or [] if v.get('kind') == 'SHOW'] + + def _episodes(self, items: list[dict], series_title: str, season: int, + only_episode: str | None) -> list[Episode]: + episodes = [] + position = 0 + for it in items: + pid = it.get('providerId') or '' + if not EPISODE_ID_RE.match(pid): + continue + position += 1 + raw_title = (it.get('title') or '').strip() + m = EPISODE_NUMBER_RE.search(raw_title) + number = int(m[1]) if m else position + total = int(m[2]) if m else None + title = (it.get('subtitle') or '').strip() or (f'Épisode {number}' if m else raw_title) + if only_episode and pid != only_episode: + continue + episodes.append(Episode( + id=pid, + url=((it.get('link') or {}).get('url') + or f'https://www.arte.tv/{self.lang}/videos/{pid}/'), + series=series_title, season=season, number=number, title=title, + description=it.get('description'), + duration=(it.get('duration') or {}).get('seconds'), total=total)) + return episodes diff --git a/arte_dl/cli.py b/arte_dl/cli.py new file mode 100644 index 0000000..56aef10 --- /dev/null +++ b/arte_dl/cli.py @@ -0,0 +1,154 @@ +"""Command-line entry point.""" +from __future__ import annotations + +import argparse +import shutil +import subprocess +import sys +from pathlib import Path + +from . import __version__ +from .arte_api import ArteClient, ArteError, Series, parse_url +from .config import ConfigError, default_config_path, load_config +from .download import DownloadError, destination, download, plan +from .metadata import TMDBError +from .metadata import apply as apply_metadata +from .selection import SelectionError + + +def parse_ranges(spec: str) -> set[int]: + """"1,3-5" -> {1, 3, 4, 5}""" + numbers = set() + for part in spec.split(','): + part = part.strip() + if not part: + continue + lo, sep, hi = part.partition('-') + try: + numbers.update(range(int(lo), int(hi) + 1) if sep else {int(lo)}) + except ValueError: + raise argparse.ArgumentTypeError(f'invalid range "{part}"') from None + return numbers + + +def build_parser() -> argparse.ArgumentParser: + p = argparse.ArgumentParser( + prog='arte-dl', + description='Download arte.tv series (every season) into Series/Season XX/*.mkv') + p.add_argument('urls', nargs='+', metavar='URL', + help='arte.tv series (RC-xxxxxx), season or episode URL') + p.add_argument('-c', '--config', type=Path, + help=f'TOML config file (default: {default_config_path()})') + p.add_argument('-o', '--output', help='output root directory (overrides [output] directory)') + p.add_argument('-s', '--seasons', type=parse_ranges, help='only these seasons, e.g. "1,3-5"') + p.add_argument('-e', '--episodes', type=parse_ranges, help='only these episode numbers') + p.add_argument('-l', '--list', action='store_true', + help='list seasons / episodes and destination paths, then exit') + p.add_argument('-n', '--dry-run', action='store_true', + help='also show the selected tracks for each episode, without downloading') + p.add_argument('-f', '--force', action='store_true', help='re-download existing files') + p.add_argument('--tmdb-id', type=int, metavar='ID', + help='TMDB show id to use instead of searching, with a single URL (remembered for next runs)') + p.add_argument('--no-metadata', action='store_true', help="don't query TMDB, use Arte data only") + p.add_argument('-V', '--version', action='version', version=f'%(prog)s {__version__}') + return p + + +def _filter(series: Series, args) -> None: + for season in series.seasons: + season.episodes = [e for e in season.episodes + if not args.episodes or args.episodes & set(e.numbers)] + series.seasons = [s for s in series.seasons + if s.episodes and (not args.seasons or s.number in args.seasons)] + + +def process(url: str, cfg, args) -> tuple[int, int]: + """Returns (ok, failed) episode counts.""" + lang, _ = parse_url(url) + series = ArteClient(lang).resolve(url) + total = len(series.episodes) + print(f'== {series.title} ({series.id}) — {len(series.seasons)} season(s), {total} episode(s)') + if series.unavailable: + print(f' not available online: {", ".join(series.unavailable)}') + + meta = cfg.metadata + if meta.provider == 'tmdb' and not args.no_metadata: + if meta.key: + try: + apply_metadata(series, meta, forced_id=args.tmdb_id) + except TMDBError as e: + print(f' TMDB: {e} — keeping Arte metadata', file=sys.stderr) + else: + print(' TMDB: no API key ([metadata] api_key or TMDB_API_KEY) — using Arte metadata') + _filter(series, args) + + ok = failed = 0 + for season in series.seasons: + print(f'-- Season {season.number}: {season.title}') + for ep in season.episodes: + dest = destination(series, ep, cfg) + exists = dest.exists() + arte_label = f'S{ep.arte_season:02d}E{ep.arte_number:02d}' + renumbered = f' (Arte {arte_label})' if arte_label != ep.label else '' + print(f' {ep.label} [{ep.id}] {ep.title}{renumbered}') + print(f' -> {dest}{" (exists)" if exists else ""}') + if args.list or (exists and not args.force): + ok += 1 + continue + try: + info, sel = plan(ep, cfg) + h = sel.video + print(f' video : {h["format_id"]} {h.get("height")}p {h.get("vcodec")}') + print(f' audio : {", ".join(a.title for a in sel.audio) or "(muxed)"}') + print(' subs : ' + (', '.join( + s.title + (' *' if i == sel.default_subtitle else '') + for i, s in enumerate(sel.subtitles)) or '-')) + for w in sel.warnings: + print(f' warning: {w}') + if not args.dry_run: + download(series, ep, info, sel, dest, cfg) + print(' done') + ok += 1 + except KeyboardInterrupt: + raise + except (DownloadError, SelectionError, subprocess.CalledProcessError) as e: + failed += 1 + print(f' FAILED: {e}', file=sys.stderr) + except Exception as e: # keep going with the next episode + failed += 1 + print(f' FAILED: {type(e).__name__}: {e}', file=sys.stderr) + return ok, failed + + +def main(argv: list[str] | None = None) -> int: + parser = build_parser() + args = parser.parse_args(argv) + if args.tmdb_id and len(args.urls) > 1: + parser.error('--tmdb-id only makes sense with a single URL') + sys.stdout.reconfigure(line_buffering=True) # keep our lines in order with yt-dlp's stderr + try: + cfg = load_config(args.config) + except ConfigError as e: + print(f'config error: {e}', file=sys.stderr) + return 2 + if args.output: + cfg.output.directory = args.output + if not (args.list or args.dry_run) and not shutil.which('ffmpeg'): + print('ffmpeg not found in PATH', file=sys.stderr) + return 2 + + ok = failed = 0 + for url in args.urls: + try: + o, f = process(url, cfg, args) + except ArteError as e: + print(f'error: {e}', file=sys.stderr) + failed += 1 + continue + except KeyboardInterrupt: + print('\ninterrupted', file=sys.stderr) + return 130 + ok, failed = ok + o, failed + f + if failed: + print(f'\n{failed} failure(s), {ok} ok', file=sys.stderr) + return 1 if failed else 0 diff --git a/arte_dl/config.py b/arte_dl/config.py new file mode 100644 index 0000000..bd42e11 --- /dev/null +++ b/arte_dl/config.py @@ -0,0 +1,131 @@ +"""Configuration: TOML file merged over built-in defaults.""" +from __future__ import annotations + +import os +import re +import sys +from dataclasses import dataclass, field, fields +from pathlib import Path + +if sys.version_info >= (3, 11): + import tomllib +else: + import tomli as tomllib + + +class ConfigError(Exception): + pass + + +@dataclass +class OutputConfig: + directory: str = '.' + series_dir: str = '{series} ({year})' + season_dir: str = 'Season {season:02d}' + filename: str = '{series} - S{season:02d}E{episode:02d} - {title}' + + +@dataclass +class VideoConfig: + max_height: int = 1080 + # Preference order when several codecs exist at the same height: "hevc" (H.265), "avc" (H.264) + codecs: list[str] = field(default_factory=lambda: ['hevc', 'avc']) + + +@dataclass +class AudioConfig: + # Tracks to include, in order; the first one found becomes the default track. + # "original" -> original version (VO), whatever its language + # "fr", "de" -> that language (excluding audio description / "confort audio") + # "fr-ad" -> audio description, "fr-comfort" -> "confort audio" (clearer dialogue) + tracks: list[str] = field(default_factory=lambda: ['original', 'fr']) + + +@dataclass +class SubtitlesConfig: + # "fr" -> full subtitles, "fr-forced" -> forced (foreign lines only), "fr-acc" -> SDH + tracks: list[str] = field(default_factory=lambda: ['fr', 'fr-forced']) + # Default subtitle track: a key from `tracks`, "none", or "auto" + # ("auto": forced subs when the default audio is in the subtitle language, full subs otherwise) + default: str = 'auto' + + +@dataclass +class MetadataConfig: + # "tmdb": series name / year / numbering / titles from themoviedb.org; "none": Arte data only + provider: str = 'tmdb' + # TMDB API key (v3) or read access token (v4); the TMDB_API_KEY environment variable also works + api_key: str = '' + # Priority order for series name, episode titles and synopses. TMDB languages ("fr-FR", + # "en-US", "fr"...), "arte" (Arte's own titles) and "original" (the show's original language) + languages: list[str] = field(default_factory=lambda: ['fr-FR', 'arte', 'en-US']) + + @property + def key(self) -> str: + return self.api_key or os.environ.get('TMDB_API_KEY', '') + + +@dataclass +class Config: + output: OutputConfig = field(default_factory=OutputConfig) + video: VideoConfig = field(default_factory=VideoConfig) + audio: AudioConfig = field(default_factory=AudioConfig) + subtitles: SubtitlesConfig = field(default_factory=SubtitlesConfig) + metadata: MetadataConfig = field(default_factory=MetadataConfig) + + @property + def output_dir(self) -> Path: + return Path(os.path.expandvars(self.output.directory)).expanduser() + + +def default_config_path() -> Path: + base = os.environ.get('XDG_CONFIG_HOME') or Path.home() / '.config' + return Path(base) / 'arte-dl' / 'config.toml' + + +def _apply(section_obj, data: dict, section: str) -> None: + known = {f.name: f for f in fields(section_obj)} + for key, value in data.items(): + if key not in known: + raise ConfigError(f'Unknown option [{section}] {key}') + default = getattr(section_obj, key) + if isinstance(default, list) and not ( + isinstance(value, list) and all(isinstance(v, str) for v in value)): + raise ConfigError(f'[{section}] {key} must be a list of strings') + if isinstance(default, (str, int)) and type(value) is not type(default): + raise ConfigError(f'[{section}] {key} must be a {type(default).__name__}') + setattr(section_obj, key, value) + + +def load_config(path: Path | None) -> Config: + """Load `path`, or the default location if it exists, or built-in defaults.""" + cfg = Config() + if path is None: + path = default_config_path() + if not path.exists(): + return cfg + try: + with open(path, 'rb') as f: + data = tomllib.load(f) + except FileNotFoundError: + raise ConfigError(f'Config file not found: {path}') from None + except tomllib.TOMLDecodeError as e: + raise ConfigError(f'{path}: {e}') from None + + for section, values in data.items(): + if not hasattr(cfg, section) or not isinstance(values, dict): + raise ConfigError(f'Unknown section [{section}]') + _apply(getattr(cfg, section), values, section) + _validate(cfg) + return cfg + + +LANGUAGE_RE = re.compile(r'^(?:[a-z]{2}(?:-[A-Z]{2})?|arte|original)$') + + +def _validate(cfg: Config) -> None: + if cfg.metadata.provider not in ('tmdb', 'none'): + raise ConfigError('[metadata] provider must be "tmdb" or "none"') + bad = [lang for lang in cfg.metadata.languages if not LANGUAGE_RE.match(lang)] + if bad: + raise ConfigError(f'[metadata] languages: invalid {bad} (expected "fr-FR", "fr", "arte", "original")') diff --git a/arte_dl/download.py b/arte_dl/download.py new file mode 100644 index 0000000..32219ab --- /dev/null +++ b/arte_dl/download.py @@ -0,0 +1,177 @@ +"""Download one episode with yt-dlp, then remux it into the final MKV with ffmpeg. + +yt-dlp downloads and merges the selected video + audio formats and the VTT +subtitles into a temporary directory; a final ffmpeg pass adds the subtitles, +track languages / titles / default & forced flags and episode tags. +""" +from __future__ import annotations + +import re +import shutil +import subprocess +from pathlib import Path + +from yt_dlp import YoutubeDL +from yt_dlp.utils import ISO639Utils + +from .arte_api import Episode, Series +from .config import Config +from .selection import Selection, select +from .subtitles import vtt_to_srt + +_FORBIDDEN = re.compile(r'[\x00-\x1f"*<>?|\\]') + + +class DownloadError(Exception): + pass + + +def sanitize(name: str) -> str: + name = name.replace('/', '-').replace(':', ' -') + name = _FORBIDDEN.sub('', name) + return re.sub(r'\s+', ' ', name).strip().rstrip('.') + + +class EpisodeNumbers: + """{episode:02d} -> "01", or "01-E02" for a file holding several episodes.""" + + def __init__(self, numbers: list[int]): + self.numbers = numbers + + def __format__(self, spec: str) -> str: + first, last = self.numbers[0], self.numbers[-1] + return format(first, spec) + (f'-E{format(last, spec)}' if last != first else '') + + +# Groups left empty by a missing field: "Series ()", "[tmdbid-]", "{tmdb-}" +_EMPTY_GROUP_RE = re.compile(r'\s*[(\[{]\s*(?:[a-z]+-)?\s*[)\]}]') + + +def destination(series: Series, ep: Episode, cfg: Config) -> Path: + fields = {'series': sanitize(ep.series), 'year': series.year or '', + 'tmdb_id': series.tmdb_id or '', 'tvdb_id': series.tvdb_id or '', + 'season': ep.season, 'episode': EpisodeNumbers(ep.numbers), + 'title': sanitize(ep.title), 'id': ep.id} + try: + parts = [_EMPTY_GROUP_RE.sub('', tpl.format(**fields)).strip() for tpl in ( + cfg.output.series_dir, cfg.output.season_dir, cfg.output.filename)] + except (KeyError, ValueError, AttributeError) as e: + raise DownloadError(f'Invalid output template: {e!r}') from None + series_dir, season_dir, filename = parts + return cfg.output_dir / series_dir / season_dir / f'{filename}.mkv' + + +def _ydl_params(**extra) -> dict: + return { + 'quiet': True, + 'no_warnings': False, + 'noprogress': False, + 'retries': 10, + 'fragment_retries': 10, + 'extractor_retries': 5, + **extra, + } + + +def extract(ep: Episode) -> dict: + """Extract and process formats (yt-dlp rewrites format ids while processing: + "suédois (VO)" -> "suédois__VO_"), without downloading.""" + with YoutubeDL(_ydl_params()) as ydl: + info = ydl.extract_info(ep.url, download=False) + if not info or not info.get('formats'): + raise DownloadError('No formats found (not available in this country / expired?)') + # Same cleanup as --load-info-json, so the info dict can be processed again + return YoutubeDL.sanitize_info(info, remove_private_keys=True) + + +def plan(ep: Episode, cfg: Config) -> tuple[dict, Selection]: + info = extract(ep) + return info, select(info, cfg) + + +def _lang3(lang: str | None) -> str: + return (lang and ISO639Utils.short2long(lang)) or 'und' + + +def download(series: Series, ep: Episode, info: dict, sel: Selection, dest: Path, cfg: Config) -> None: + work = cfg.output_dir / '.arte-dl-tmp' / ep.id + shutil.rmtree(work, ignore_errors=True) + work.mkdir(parents=True) + + params = _ydl_params( + format=sel.format_spec, + allow_multiple_audio_streams=len(sel.audio) > 1, + merge_output_format='mkv', + outtmpl={'default': str(work / 'media.%(ext)s'), + 'subtitle': str(work / 'sub.%(ext)s')}, + writesubtitles=bool(sel.subtitles), + subtitleslangs=[re.escape(s.key) for s in sel.subtitles], + subtitlesformat='vtt', + ) + with YoutubeDL(params) as ydl: + ydl.process_ie_result(info, download=True) + + media = next((p for p in work.glob('media.*') + if p.suffix not in ('.part', '.ytdl') and '.f' not in p.stem), None) + if media is None: + raise DownloadError(f'yt-dlp produced no media file in {work}') + sub_files = [] + for s in sel.subtitles: + vtt = work / f'sub.{s.key}.vtt' + if not vtt.exists(): + raise DownloadError(f'subtitle file missing: {vtt.name}') + srt_text = vtt_to_srt(vtt.read_text(encoding='utf-8-sig', errors='replace')) + if not srt_text: + raise DownloadError(f'subtitle file {vtt.name} has no cues') + srt = vtt.with_suffix('.srt') + srt.write_text(srt_text, encoding='utf-8') + sub_files.append(srt) + + dest.parent.mkdir(parents=True, exist_ok=True) + part = dest.with_name(dest.stem + '.part.mkv') + subprocess.run(_mux_command(media, sub_files, series, ep, sel, part), check=True) + part.replace(dest) + shutil.rmtree(work, ignore_errors=True) + try: + work.parent.rmdir() # .arte-dl-tmp, if no other download is in progress + except OSError: + pass + + +def _mux_command(media: Path, subs: list[Path], series: Series, ep: Episode, sel: Selection, + out: Path) -> list[str]: + cmd = ['ffmpeg', '-hide_banner', '-loglevel', 'error', '-nostdin', '-y', '-i', str(media)] + for path in subs: + cmd += ['-i', str(path)] + cmd += ['-map', '0:v:0', '-map', '0:a?'] + for i in range(len(subs)): + cmd += ['-map', f'{i + 1}:0'] + cmd += ['-c', 'copy', '-c:s', 'srt', '-map_metadata', '-1'] + + tags = { + 'title': ep.title, + 'show': ep.series, + 'season_number': str(ep.season), + 'episode_sort': str(ep.number), + 'episode_id': ep.label, + 'description': ep.description or '', + 'comment': ep.url, + 'tmdb': f'tv/{series.tmdb_id}' if series.tmdb_id else '', + 'date': str(series.year or ''), + } + for k, v in tags.items(): + if v: + cmd += ['-metadata', f'{k}={v}'] + + for i, a in enumerate(sel.audio): + cmd += [f'-metadata:s:a:{i}', f'language={_lang3(a.lang)}', + f'-metadata:s:a:{i}', f'title={a.title}', + f'-disposition:a:{i}', 'default' if i == 0 else '0'] + for i, s in enumerate(sel.subtitles): + flags = [f for f, on in (('default', i == sel.default_subtitle), + ('forced', s.kind == 'forced'), + ('hearing_impaired', s.kind == 'sdh')) if on] + cmd += [f'-metadata:s:s:{i}', f'language={_lang3(s.lang)}', + f'-metadata:s:s:{i}', f'title={s.title}', + f'-disposition:s:{i}', '+'.join(flags) or '0'] + return cmd + [str(out)] diff --git a/arte_dl/metadata.py b/arte_dl/metadata.py new file mode 100644 index 0000000..cfef7a6 --- /dev/null +++ b/arte_dl/metadata.py @@ -0,0 +1,330 @@ +"""TMDB matching: reliable series name / year, episode numbering and titles. + +Arte's numbering doesn't always follow the reference one: e.g. from season 6, +"Meurtres à Sandhamn" episodes are 88 min on Arte but two 45 min episodes on +TMDB, so Arte's S06E01 becomes S06E01-E02 (multi-episode file, understood by +Plex and Jellyfin). +""" +from __future__ import annotations + +import json +import os +import re +import time +import unicodedata +import urllib.error +import urllib.parse +import urllib.request +from dataclasses import dataclass +from difflib import SequenceMatcher +from functools import lru_cache +from pathlib import Path + +from .arte_api import Episode, Series +from .config import MetadataConfig + +API = 'https://api.themoviedb.org/3' + +GENERIC_TITLE_RE = re.compile( + r'^\s*(?:episode|épisode|episodio|folge|odcinek|avsnitt|afsnit|jakso|aflevering|épisodio)' + r'\s*\d+\s*$', re.I) +_PART_WORD = r'(?:part(?:ie)?|teil|del|deel|parte|osa|część)' +# "X part 1", "X (part1)", "X - Partie 2", "X (2)" +PART_SUFFIX_RE = re.compile( + rf'\s*[-,:]?\s*(?:[(\[]\s*(?:{_PART_WORD}\s*)?\d+\s*[)\]]|{_PART_WORD}\s*\d+)\s*$', re.I) + +# Arte duration vs TMDB runtime(s): accepted ratio range +DURATION_TOLERANCE = (0.75, 1.3) + + +class TMDBError(Exception): + pass + + +class TMDBClient: + def __init__(self, key: str, retries: int = 5): + self.key = key + self.retries = retries + + def _get(self, path: str, **params) -> dict | None: + headers = {'Accept': 'application/json', 'User-Agent': 'arte-dl'} + if self.key.startswith('eyJ'): # v4 read access token (JWT) + headers['Authorization'] = f'Bearer {self.key}' + else: + params['api_key'] = self.key + url = f'{API}{path}?{urllib.parse.urlencode(params)}' + req = urllib.request.Request(url, headers=headers) + for attempt in range(self.retries + 1): + try: + with urllib.request.urlopen(req, timeout=30) as resp: + return json.load(resp) + except urllib.error.HTTPError as e: + if e.code == 404: + return None + if e.code == 401: + raise TMDBError('TMDB rejected the API key (HTTP 401)') from e + if (e.code == 429 or e.code >= 500) and attempt < self.retries: + time.sleep(min(int(e.headers.get('Retry-After') or 0) or 2 ** attempt, 30)) + continue + raise TMDBError(f'HTTP {e.code} for {path}') from e + except urllib.error.URLError as e: + if attempt < self.retries: + time.sleep(2 ** attempt) + continue + raise TMDBError(f'{e.reason} for {path}') from e + raise AssertionError('unreachable') + + @lru_cache(maxsize=None) + def search(self, query: str, language: str) -> tuple[dict, ...]: + data = self._get('/search/tv', query=query, language=language, include_adult='false') + return tuple((data or {}).get('results') or ()) + + @lru_cache(maxsize=None) + def show(self, show_id: int) -> dict: + data = self._get(f'/tv/{show_id}', append_to_response='external_ids,translations') + if not data: + raise TMDBError(f'Unknown TMDB show {show_id}') + return data + + @lru_cache(maxsize=None) + def season(self, show_id: int, number: int, language: str) -> dict | None: + return self._get(f'/tv/{show_id}/season/{number}', language=language) + + +# --- persisted Arte collection -> TMDB id mapping ----------------------------- + +def _ids_path() -> Path: + base = os.environ.get('XDG_DATA_HOME') or Path.home() / '.local' / 'share' + return Path(base) / 'arte-dl' / 'tmdb-ids.json' + + +def load_ids() -> dict[str, int]: + try: + return json.loads(_ids_path().read_text()) + except (FileNotFoundError, ValueError): + return {} + + +def save_id(arte_id: str, tmdb_id: int) -> None: + ids = load_ids() + if ids.get(arte_id) == tmdb_id: + return + ids[arte_id] = tmdb_id + path = _ids_path() + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(ids, indent=2, sort_keys=True) + '\n') + + +# --- series matching ------------------------------------------------------------ + +def normalize(text: str) -> str: + text = unicodedata.normalize('NFKD', text or '') + text = ''.join(c for c in text if not unicodedata.combining(c)).lower() + return ' '.join(re.sub(r'[^\w]+', ' ', text).split()) + + +def _similarity(a: str, b: str) -> float: + return SequenceMatcher(None, normalize(a), normalize(b)).ratio() + + +def score_candidate(series: Series, queries: list[str], result: dict) -> tuple[float, float]: + """Returns (score, name similarity).""" + sim = max((_similarity(q, name) for q in queries + for name in (result.get('name'), result.get('original_name')) if name), default=0) + score = sim + year = (result.get('first_air_date') or '')[:4] + if series.year and year.isdigit(): + diff = abs(int(year) - series.year) + score += 0.2 if diff == 0 else 0.1 if diff == 1 else -0.5 if diff > 2 else 0 + if series.original_language and result.get('original_language') == series.original_language: + score += 0.1 + return score, sim + + +def find_show(client: TMDBClient, series: Series, language: str) -> tuple[int | None, str]: + """Search TMDB for the series. Returns (id or None, explanation).""" + queries = list(dict.fromkeys(q for q in (series.original_title, series.title) if q)) + # "The Hack : sur écoute" -> also try "The Hack" + queries += [q.split(' : ')[0] for q in queries if ' : ' in q and q.split(' : ')[0] not in queries] + candidates = {} + for q in queries: + for r in client.search(q, language)[:10]: + candidates[r['id']] = r + scored = sorted(((score_candidate(series, queries, r), r) for r in candidates.values()), + key=lambda x: x[0][0], reverse=True) + if not scored: + return None, f'no TMDB result for {queries}' + + def describe(r): + return f'{r.get("name")} ({(r.get("first_air_date") or "?")[:4]}) id={r["id"]}' + + (score, sim), best = scored[0] + if sim < 0.8 or score < 0.95: + return None, f'no confident match (best: {describe(best)}, score {score:.2f})' + if len(scored) > 1 and scored[1][0][0] > score - 0.05: + return None, f'ambiguous: {describe(best)} / {describe(scored[1][1])}' + return best['id'], f'matched {describe(best)}' + + +# --- episode alignment ------------------------------------------------------------- + +def _duration_ok(arte_seconds: int | None, runtimes: list[int | None]) -> bool | None: + """None when it can't be checked.""" + if not arte_seconds or not runtimes or not all(runtimes): + return None + ratio = arte_seconds / (60 * sum(runtimes)) + return DURATION_TOLERANCE[0] <= ratio <= DURATION_TOLERANCE[1] + + +def align(arte: list[Episode], tmdb: list[dict]) -> tuple[dict[str, list[int]] | None, str]: + """Map each Arte episode of a season to one or several consecutive TMDB episode numbers.""" + tmdb = sorted(tmdb, key=lambda e: e['episode_number']) + numbers = [e['episode_number'] for e in tmdb] + runtime = {e['episode_number']: e.get('runtime') for e in tmdb} + arte = sorted(arte, key=lambda e: e.arte_number) + c = len(numbers) + m = max([e.total or 0 for e in arte] + [e.arte_number for e in arte]) + if not c: + return None, 'TMDB season has no episodes' + + def check(mapping): + results = [_duration_ok(e.duration, [runtime[n] for n in mapping[e.id]]) for e in arte] + return False not in results + + if c % m == 0: + k = c // m + mapping = {e.id: numbers[(e.arte_number - 1) * k: e.arte_number * k] for e in arte} + if check(mapping): + return mapping, '1:1' if k == 1 else f'1 Arte episode = {k} TMDB episodes' + if k == 1: # same count but durations disagree: still the most likely mapping + return mapping, '1:1 (durations differ)' + + # Uneven split: walk the whole season by duration (needs every Arte episode and runtime). + if len(arte) == m and all(e.duration for e in arte) and all(runtime.values()): + mapping, i = {}, 0 + for e in arte: + group, acc = [], 0 + while i < c and (not group or abs(acc + 60 * runtime[numbers[i]] - e.duration) + < abs(acc - e.duration)): + acc += 60 * runtime[numbers[i]] + group.append(numbers[i]) + i += 1 + if not group: + break + mapping[e.id] = group + if i == c and len(mapping) == len(arte) and check(mapping): + return mapping, 'aligned by duration' + return None, f'Arte has {m} episode(s), TMDB {c}: no reliable mapping' + + +# --- localized titles --------------------------------------------------------------- + +def _is_generic(title: str | None) -> bool: + return not title or not title.strip() or bool(GENERIC_TITLE_RE.match(title)) + + +def merge_titles(titles: list[str]) -> str: + """Titles of the episodes held in one file: "X (part 1)", "X (part 2)" -> "X".""" + if len(titles) == 1: + return titles[0] + stripped = list(dict.fromkeys(PART_SUFFIX_RE.sub('', t).strip() for t in titles)) + return stripped[0] if len(stripped) == 1 else ' / '.join(stripped) + + +@dataclass +class Metadata: + client: TMDBClient + show: dict + languages: list[str] + + @property + def show_id(self) -> int: + return self.show['id'] + + def tmdb_language(self, lang: str) -> str: + return (self.show.get('original_language') or 'en') if lang == 'original' else lang + + def series_name(self, arte_title: str) -> str: + translations = (self.show.get('translations') or {}).get('translations') or [] + for lang in self.languages: + if lang == 'arte': + return arte_title + if lang == 'original': + name = self.show.get('original_name') + else: + iso, _, region = lang.partition('-') + matches = [t for t in translations if t.get('iso_639_1') == iso + and (not region or t.get('iso_3166_1') == region)] + # Untranslated: TMDB (and Plex / Jellyfin) show the original name in that language + name = next((t['data'].get('name') for t in matches if t['data'].get('name')), + self.show.get('original_name')) + if name: + return name + return self.show.get('name') or arte_title + + def episodes(self, season: int, lang: str) -> dict[int, dict]: + data = self.client.season(self.show_id, season, self.tmdb_language(lang)) or {} + return {e['episode_number']: e for e in data.get('episodes') or []} + + def localize(self, ep: Episode) -> None: + """Pick the episode title and synopsis following the language priority.""" + title = overview = None + for lang in self.languages: + if lang == 'arte': + t, o = (None if _is_generic(ep.arte_title) else ep.arte_title), ep.description + else: + eps = self.episodes(ep.season, lang) + found = [eps.get(n) or {} for n in ep.numbers] + names = [f.get('name') for f in found] + t = None if any(_is_generic(n) for n in names) else merge_titles(names) + o = '\n\n'.join(f['overview'] for f in found if f.get('overview')) or None + title = title or t + overview = overview or o + if title and overview: + break + ep.title = title or ep.arte_title + ep.description = overview or ep.description + + +def apply(series: Series, cfg: MetadataConfig, forced_id: int | None = None, + log=print) -> None: + """Rename / renumber the series' episodes from TMDB. Keeps Arte data when unsure.""" + client = TMDBClient(cfg.key) + first_lang = next((l for l in cfg.languages if l not in ('arte', 'original')), 'en-US') + + show_id = forced_id or load_ids().get(series.id) + if show_id: + how = 'forced' if forced_id else 'remembered' + else: + show_id, how = find_show(client, series, first_lang) + if not show_id: + log(f' TMDB: {how} — keeping Arte metadata (use --tmdb-id to set it)') + return + show = client.show(show_id) + save_id(series.id, show_id) + + meta = Metadata(client, show, cfg.languages) + series.tmdb_id = show_id + series.tvdb_id = (show.get('external_ids') or {}).get('tvdb_id') + series.year = int(show['first_air_date'][:4]) if show.get('first_air_date') else series.year + name = meta.series_name(series.title) + log(f' TMDB: {how} — "{name}" ({series.year}) https://www.themoviedb.org/tv/{show_id}') + + tmdb_seasons = {s['season_number'] for s in show.get('seasons') or [] if s.get('episode_count')} + for season in series.seasons: + for ep in season.episodes: + ep.series = name + if season.number not in tmdb_seasons: + log(f' TMDB: season {season.number} not found — keeping Arte numbering') + continue + tmdb_eps = list(meta.episodes(season.number, first_lang).values()) + mapping, how = align(season.episodes, tmdb_eps) + if mapping is None: + log(f' TMDB: season {season.number}: {how} — keeping Arte numbering') + continue + if how != '1:1': + log(f' TMDB: season {season.number}: {how}') + for ep in season.episodes: + nums = mapping[ep.id] + ep.number, ep.last_number = nums[0], (nums[-1] if len(nums) > 1 else None) + meta.localize(ep) diff --git a/arte_dl/selection.py b/arte_dl/selection.py new file mode 100644 index 0000000..37f697b --- /dev/null +++ b/arte_dl/selection.py @@ -0,0 +1,195 @@ +"""Pick video / audio / subtitle tracks from a yt-dlp info dict according to the config. + +Arte HLS masters expose one video ladder plus several audio renditions whose +format ids look like "VF-STF-audio_0-suédois__VO_", "VF-STF-audio_0-français", +"VF-STF-audio_0-français__audiodescription_", "...__confort_audio_". +Subtitles are keyed "fr" (full), "fr-forced", "fr-acc" (SDH), "de-forced", ... +""" +from __future__ import annotations + +import re +from dataclasses import dataclass, field + +from .config import Config + +AUDIO_ID_RE = re.compile(r'-audio_\d+-(?P.+)$') + +CODEC_ALIASES = { + 'avc': ('avc1', 'avc3', 'h264'), + 'h264': ('avc1', 'avc3', 'h264'), + 'hevc': ('hev1', 'hvc1', 'h265', 'hevc'), + 'h265': ('hev1', 'hvc1', 'h265', 'hevc'), + 'av1': ('av01',), + 'vp9': ('vp9', 'vp09'), +} + +LANGUAGE_NAMES = { + 'fr': 'Français', 'de': 'Deutsch', 'en': 'English', 'es': 'Español', + 'it': 'Italiano', 'pl': 'Polski', 'sv': 'Svenska', 'da': 'Dansk', + 'no': 'Norsk', 'fi': 'Suomi', 'nl': 'Nederlands', 'pt': 'Português', +} +SUB_KINDS = {'': 'full', 'forced': 'forced', 'acc': 'sdh'} +SUB_KIND_LABELS = {'full': '', 'forced': ' (forcés)', 'sdh': ' (SDH)'} + + +class SelectionError(Exception): + pass + + +@dataclass +class AudioTrack: + format_id: str + lang: str | None + title: str + original: bool = False + kind: str = 'main' # main | ad | comfort + + +@dataclass +class SubtitleTrack: + key: str + lang: str + kind: str # full | forced | sdh + title: str + + +@dataclass +class Selection: + video: dict + audio: list[AudioTrack] = field(default_factory=list) + subtitles: list[SubtitleTrack] = field(default_factory=list) + default_subtitle: int | None = None + warnings: list[str] = field(default_factory=list) + + @property + def format_spec(self) -> str: + return '+'.join([self.video['format_id'], *(a.format_id for a in self.audio)]) + + +def normalize_lang(lang: str | None) -> str | None: + if not lang: + return None + lang = lang.lower().split('-')[0] + if len(lang) == 3: + from yt_dlp.utils import ISO639Utils + return ISO639Utils.long2short(lang) or lang + return lang + + +def parse_audio(fmt: dict) -> AudioTrack: + m = AUDIO_ID_RE.search(fmt['format_id']) + name = m['name'] if m else (fmt.get('format_note') or fmt.get('language') or 'audio') + # "suédois__VO_" (processed) or "suédois (VO)" (raw) + parts = [p.strip('_ ') for p in re.sub(r'[\s()]', '_', name).split('__')] + base = parts[0].replace('_', ' ').strip() + flags = ' '.join(parts[1:]).lower().replace('_', ' ') + original = bool(re.search(r'\bvo\b', flags)) + kind = 'ad' if 'audiodescription' in flags else 'comfort' if 'confort' in flags else 'main' + + lang = normalize_lang(fmt.get('language')) + label = base[:1].upper() + base[1:] if base else LANGUAGE_NAMES.get(lang or '', lang or 'Audio') + suffix = [s for s, on in (('VO', original), ('audiodescription', kind == 'ad'), + ('confort audio', kind == 'comfort')) if on] + title = f'{label} ({", ".join(suffix)})' if suffix else label + return AudioTrack(format_id=fmt['format_id'], lang=lang, title=title, + original=original, kind=kind) + + +def audio_matches(spec: str, track: AudioTrack) -> bool: + spec = spec.lower() + if spec in ('original', 'vo'): + return track.original and track.kind == 'main' + lang, _, kind = spec.partition('-') + kind = {'': 'main', 'ad': 'ad', 'comfort': 'comfort'}.get(kind) + if kind is None: + raise SelectionError(f'Invalid audio track "{spec}" (use original, fr, fr-ad, fr-comfort...)') + return track.lang == lang and track.kind == kind + + +def _codec_rank(vcodec: str, preferences: list[str]) -> int: + vcodec = (vcodec or '').lower() + for i, pref in enumerate(preferences): + if vcodec.startswith(CODEC_ALIASES.get(pref.lower(), (pref.lower(),))): + return len(preferences) - i + return 0 + + +def is_video(f: dict) -> bool: + return f.get('vcodec') not in (None, 'none') and bool(f.get('height')) + + +def pick_video(formats: list[dict], cfg: Config) -> dict: + videos = [f for f in formats if is_video(f)] + if not videos: + raise SelectionError('No video format found') + capped = [f for f in videos if not cfg.video.max_height or f['height'] <= cfg.video.max_height] + return max(capped or videos, key=lambda f: ( + f['height'], _codec_rank(f.get('vcodec'), cfg.video.codecs), f.get('tbr') or 0)) + + +def select(info: dict, cfg: Config) -> Selection: + formats = info.get('formats') or [] + video = pick_video(formats, cfg) + sel = Selection(video=video) + + # Audio renditions (Arte reports acodec=None for them, so test vcodec only). Keep + # the first occurrence of each format id: several "versions" can repeat them. + audio_tracks, seen = [], set() + for f in formats: + if f.get('vcodec') == 'none' and f['format_id'] not in seen: + seen.add(f['format_id']) + audio_tracks.append(parse_audio(f)) + + chosen = [] + for spec in cfg.audio.tracks: + track = next((t for t in audio_tracks if audio_matches(spec, t)), None) + if track is None: + sel.warnings.append(f'audio "{spec}" not available') + elif all(t.format_id != track.format_id for t in chosen): + chosen.append(track) + if not chosen and audio_tracks: + best = max((f for f in formats if f['format_id'] in seen), + key=lambda f: (f.get('language_preference') or 0, f.get('abr') or 0)) + chosen.append(parse_audio(best)) + sel.warnings.append(f'no configured audio track found, falling back to "{chosen[0].title}"') + elif not audio_tracks and video.get('acodec') in (None, 'none'): + sel.warnings.append('no separate audio track found') + sel.audio = chosen + + # Subtitles: the same file can be listed under several keys (e.g. "sv" and "fr"). + subs = info.get('subtitles') or {} + seen_urls = set() + for key in cfg.subtitles.tracks: + entries = subs.get(key) + if not entries: + sel.warnings.append(f'subtitles "{key}" not available') + continue + url = entries[0].get('url') + if url in seen_urls: + continue + seen_urls.add(url) + lang, _, suffix = key.partition('-') + kind = SUB_KINDS.get(suffix, 'full') + name = LANGUAGE_NAMES.get(lang, lang) + sel.subtitles.append(SubtitleTrack(key=key, lang=lang, kind=kind, + title=name + SUB_KIND_LABELS[kind])) + sel.default_subtitle = _default_subtitle(sel, cfg.subtitles.default) + return sel + + +def _default_subtitle(sel: Selection, mode: str) -> int | None: + subs = sel.subtitles + if not subs or mode == 'none': + return None + if mode != 'auto': + return next((i for i, s in enumerate(subs) if s.key == mode), None) + audio_lang = sel.audio[0].lang if sel.audio else None + sub_lang = subs[0].lang + + def find(kind): + return next((i for i, s in enumerate(subs) if s.lang == sub_lang and s.kind == kind), None) + + if audio_lang == sub_lang: + return find('forced') + full = find('full') + return full if full is not None else find('sdh') diff --git a/arte_dl/subtitles.py b/arte_dl/subtitles.py new file mode 100644 index 0000000..7ca98ce --- /dev/null +++ b/arte_dl/subtitles.py @@ -0,0 +1,54 @@ +"""Minimal WebVTT -> SRT conversion. + +Arte serves VTT files with CRLF line endings and STYLE blocks, which older +ffmpeg releases (e.g. 4.4, also used by yt-dlp --convert-subs) silently turn +into empty subtitles. Converting ourselves avoids depending on the ffmpeg version. +""" +from __future__ import annotations + +import html +import re + +_TIMING_RE = re.compile( + r'^\s*(?P(?:\d+:)?\d{2}:\d{2}[.,]\d{3})\s+-->\s+(?P(?:\d+:)?\d{2}:\d{2}[.,]\d{3})') +_KEEP_TAGS_RE = re.compile(r'') +_TAG_RE = re.compile(r'<[^>]*>') + + +def _timestamp(ts: str) -> str: + ts = ts.replace(',', '.') + parts = ts.split(':') + if len(parts) == 2: + parts.insert(0, '0') + h, m, s = parts + sec, ms = s.split('.') + return f'{int(h):02d}:{int(m):02d}:{int(sec):02d},{ms}' + + +def _clean(line: str) -> str: + # Keep , , ; drop , , , inline timestamps... + kept = [] + + def stash(m): + kept.append(m[0]) + return f'\x00{len(kept) - 1}\x00' + + line = _KEEP_TAGS_RE.sub(stash, line) + line = html.unescape(_TAG_RE.sub('', line)).replace(' ', ' ') + return re.sub(r'\x00(\d+)\x00', lambda m: kept[int(m[1])], line) + + +def vtt_to_srt(vtt: str) -> str: + text = vtt.lstrip('').replace('\r\n', '\n').replace('\r', '\n') + cues = [] + for block in re.split(r'\n{2,}', text): + lines = block.strip('\n').split('\n') + timing_idx = next((i for i, l in enumerate(lines) if _TIMING_RE.match(l)), None) + if timing_idx is None: # header, STYLE, NOTE, REGION blocks + continue + m = _TIMING_RE.match(lines[timing_idx]) + payload = [c for c in (_clean(l).strip() for l in lines[timing_idx + 1:]) if c] + if payload: + cues.append((_timestamp(m['start']), _timestamp(m['end']), payload)) + return ''.join(f'{n}\n{start} --> {end}\n' + '\n'.join(payload) + '\n\n' + for n, (start, end, payload) in enumerate(cues, 1)) diff --git a/config.example.toml b/config.example.toml new file mode 100644 index 0000000..d8f363c --- /dev/null +++ b/config.example.toml @@ -0,0 +1,49 @@ +# arte-dl — copier dans ~/.config/arte-dl/config.toml (ou passer avec -c) +# Toutes les options sont facultatives ; les valeurs ci-dessous sont les valeurs par défaut. + +[output] +# Racine de la vidéothèque (~ et $VARIABLES acceptés) +directory = "." +# Champs disponibles : {series} {year} {tmdb_id} {tvdb_id} {season} {episode} {title} {id} +# Un groupe vide (année ou id inconnus) est retiré : "Série ()" -> "Série". +# Pour forcer l'identification par Jellyfin : "{series} ({year}) [tmdbid-{tmdb_id}]" +series_dir = "{series} ({year})" +season_dir = "Season {season:02d}" +# {episode:02d} donne "01", ou "01-E02" pour un fichier qui contient deux épisodes TMDB +filename = "{series} - S{season:02d}E{episode:02d} - {title}" + +[video] +# Hauteur maximale (216, 360, 432, 720, 1080) +max_height = 1080 +# Codec préféré à hauteur égale : "hevc" (H.265, meilleur débit chez Arte) ou "avc" (H.264, plus compatible) +codecs = ["hevc", "avc"] + +[audio] +# Pistes à inclure, dans l'ordre ; la première trouvée est la piste par défaut. +# "original" -> version originale (VO), quelle que soit sa langue +# "fr", "de"… -> piste dans cette langue (hors audiodescription / confort audio) +# "fr-ad" -> audiodescription +# "fr-comfort" -> « confort audio » (dialogues renforcés) +# Si la VO est en français, "original" et "fr" désignent la même piste : elle n'est incluse qu'une fois. +tracks = ["original", "fr"] + +[subtitles] +# "fr" -> sous-titres complets, "fr-forced" -> forcés (passages en langue étrangère), "fr-acc" -> sourds et malentendants +tracks = ["fr", "fr-forced"] +# Piste de sous-titres par défaut : une clé de `tracks`, "none" ou "auto" +# ("auto" : forcés si la piste audio par défaut est dans la langue des sous-titres, complets sinon) +default = "auto" + +[metadata] +# "tmdb" : nom de série, année, numérotation et titres d'épisodes depuis themoviedb.org +# "none" : uniquement les données Arte +provider = "tmdb" +# Clé API TMDB (v3) ou jeton d'accès en lecture (v4). La variable d'environnement +# TMDB_API_KEY fonctionne aussi. Sans clé : données Arte uniquement. +api_key = "" +# Ordre de priorité pour le nom de la série, les titres et les résumés d'épisodes. +# "fr-FR", "en-US", "de"… -> langue TMDB +# "arte" -> titres / résumés Arte +# "original" -> langue originale de la série +# Une langue sans titre (ou avec un titre générique « Épisode 3 ») passe à la suivante. +languages = ["fr-FR", "arte", "en-US"] diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..6d11514 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,23 @@ +[project] +name = "arte-dl" +version = "0.1.0" +description = "Download whole arte.tv series with yt-dlp into a tidy Series/Season XX/ tree of MKV files" +readme = "README.md" +requires-python = ">=3.10" +dependencies = [ + "yt-dlp>=2025.1.1", + "tomli>=2.0; python_version < '3.11'", +] + +[project.optional-dependencies] +dev = ["pytest>=8"] + +[project.scripts] +arte-dl = "arte_dl.cli:main" + +[build-system] +requires = ["hatchling"] +build-backend = "hatchling.build" + +[tool.hatch.build.targets.wheel] +packages = ["arte_dl"] diff --git a/tests/test_metadata.py b/tests/test_metadata.py new file mode 100644 index 0000000..f0aeaf4 --- /dev/null +++ b/tests/test_metadata.py @@ -0,0 +1,189 @@ +import json + +import pytest + +from arte_dl import metadata +from arte_dl.arte_api import Episode, Season, Series +from arte_dl.config import MetadataConfig +from arte_dl.metadata import Metadata, align, find_show, merge_titles + + +def arte_ep(season, n, total, minutes, title, pid=None): + return Episode(pid or f'{season:03d}{n:03d}-000-A', 'u', 'Meurtres à Sandhamn', season, n, title, + description=f'arte {season}/{n}', duration=minutes * 60, total=total) + + +def tmdb_ep(n, name, runtime=45, overview=''): + return {'episode_number': n, 'name': name, 'runtime': runtime, 'overview': overview} + + +# As seen on TMDB (show 55270) and Arte (RC-022391) +S01_TMDB = [tmdb_ep(i, n) for i, n in enumerate( + ['La Reine de la Baltique (1)', 'La Reine de la Baltique (2)', 'La Reine de la Baltique (3)'], 1)] +S06_TMDB = [tmdb_ep(i, f'{name} part {p}') for i, (name, p) in enumerate( + [(t, p) for t in ['Le prix à payer', 'Au nom de la vérité', 'À la vie, à la mort', 'Un goût amer'] + for p in (1, 2)], 1)] + + +def s06_arte(): + return [arte_ep(6, i, 4, 88, f'Enquête {5 + i}') for i in range(1, 5)] + + +def test_align_one_to_one(): + eps = [arte_ep(1, i, 3, 43, 'Enquête 1 : La reine de la Baltique') for i in range(1, 4)] + mapping, how = align(eps, S01_TMDB) + assert how == '1:1' and [mapping[e.id] for e in eps] == [[1], [2], [3]] + + +def test_align_arte_merges_two_parts(): + eps = s06_arte() + mapping, how = align(eps, S06_TMDB) + assert how == '1 Arte episode = 2 TMDB episodes' + assert [mapping[e.id] for e in eps] == [[1, 2], [3, 4], [5, 6], [7, 8]] + + +def test_align_partial_season_uses_arte_total(): + eps = s06_arte()[2:3] # only Arte episode 3/4 still online + mapping, _ = align(eps, S06_TMDB) + assert mapping[eps[0].id] == [5, 6] + + +def test_align_uneven_by_duration(): + eps = [arte_ep(2, 1, 2, 88, 'a'), arte_ep(2, 2, 2, 45, 'b')] + tmdb = [tmdb_ep(1, 'x'), tmdb_ep(2, 'y'), tmdb_ep(3, 'z')] + mapping, how = align(eps, tmdb) + assert how == 'aligned by duration' and [mapping[e.id] for e in eps] == [[1, 2], [3]] + + +def test_align_refuses_inconsistent(): + eps = [arte_ep(2, i, 3, 45, 'a') for i in range(1, 4)] + mapping, how = align(eps, [tmdb_ep(i, 'x') for i in range(1, 6)]) + assert mapping is None and 'no reliable mapping' in how + + +def test_merge_titles(): + assert merge_titles(['Le prix à payer part 1', 'Le prix à payer part 2']) == 'Le prix à payer' + assert merge_titles(['Au nom de la vérité (part1)', 'Au nom de la vérité (part 2)']) == 'Au nom de la vérité' + assert merge_titles(['A', 'B']) == 'A / B' + assert merge_titles(['Mensonges bleus (1)', 'Mensonges bleus (2)']) == 'Mensonges bleus' + assert merge_titles(['Madeleine', 'Madeleine (2)']) == 'Madeleine' + assert merge_titles(['Chapitre 12', 'Chapitre 13']) == 'Chapitre 12 / Chapitre 13' + assert merge_titles(['Solo (part 1)']) == 'Solo (part 1)' # a single episode keeps its title + + +class FakeClient: + def __init__(self, search=(), shows=None, seasons=None): + self._search, self._shows, self._seasons = search, shows or {}, seasons or {} + self.calls = [] + + def search(self, query, language): + self.calls.append(('search', query)) + return tuple(r for r in self._search if query.lower() in json.dumps(r, ensure_ascii=False).lower()) + + def show(self, show_id): + return self._shows[show_id] + + def season(self, show_id, number, language): + self.calls.append(('season', number, language)) + eps = self._seasons.get((number, language)) + return {'episodes': eps} if eps is not None else None + + +def test_find_show_prefers_year_and_language(): + series = Series('RC-027900', 'The Hack : sur écoute', 'fr', original_title='The Hack', + original_language='en', year=2025) + client = FakeClient(search=[ + {'id': 1, 'name': 'The Hack', 'original_name': 'The Hack', 'first_air_date': '2012-01-01', + 'original_language': 'en'}, + {'id': 2, 'name': 'The Hack', 'original_name': 'The Hack', 'first_air_date': '2025-09-24', + 'original_language': 'en'}, + {'id': 3, 'name': 'Hacks', 'original_name': 'Hacks', 'first_air_date': '2021-05-13', + 'original_language': 'en'}, + ]) + assert find_show(client, series, 'fr-FR')[0] == 2 + + +def test_find_show_rejects_weak_or_ambiguous(): + series = Series('RC-1', 'Inconnue', 'fr', year=2020) + client = FakeClient(search=[{'id': 9, 'name': 'Inconnue au bataillon', 'first_air_date': '2020-01-01'}]) + assert find_show(client, series, 'fr-FR')[0] is None + series = Series('RC-2', 'Twin', 'fr') # no year: two identical names + client = FakeClient(search=[{'id': 1, 'name': 'Twin'}, {'id': 2, 'name': 'Twin'}]) + show_id, why = find_show(client, series, 'fr-FR') + assert show_id is None and why.startswith('ambiguous') + + +SHOW = { + 'id': 55270, 'name': 'Meurtres à Sandhamn', 'original_name': 'Morden i Sandhamn', + 'original_language': 'sv', 'first_air_date': '2010-01-09', 'external_ids': {'tvdb_id': 158851}, + 'seasons': [{'season_number': 0, 'episode_count': 2}, {'season_number': 1, 'episode_count': 3}, + {'season_number': 6, 'episode_count': 8}], + 'translations': {'translations': [ + {'iso_639_1': 'fr', 'iso_3166_1': 'FR', 'data': {'name': 'Meurtres à Sandhamn'}}, + {'iso_639_1': 'en', 'iso_3166_1': 'US', 'data': {'name': 'The Sandhamn Murders'}}, + {'iso_639_1': 'sv', 'iso_3166_1': 'SE', 'data': {'name': ''}}, + ]}, +} + + +def test_series_name_language_priority(): + meta = Metadata(FakeClient(), SHOW, ['en-US', 'fr-FR']) + assert meta.series_name('Arte') == 'The Sandhamn Murders' + assert Metadata(FakeClient(), SHOW, ['sv', 'fr-FR']).series_name('Arte') == 'Morden i Sandhamn' + assert Metadata(FakeClient(), SHOW, ['original']).series_name('Arte') == 'Morden i Sandhamn' + assert Metadata(FakeClient(), SHOW, ['arte', 'fr-FR']).series_name('Arte') == 'Arte' + # No French translation (The Hack): TMDB shows the original name, not Arte's + untranslated = {**SHOW, 'original_name': 'The Hack', 'translations': {'translations': [ + {'iso_639_1': 'fr', 'iso_3166_1': 'FR', 'data': {'name': ''}}]}} + assert Metadata(FakeClient(), untranslated, ['fr-FR', 'arte']).series_name('Arte') == 'The Hack' + + +def test_localize_falls_back_across_languages(): + client = FakeClient(seasons={ + (1, 'fr-FR'): [tmdb_ep(1, 'Épisode 1'), tmdb_ep(2, 'Le retour', overview='fr')], + (1, 'en-US'): [tmdb_ep(1, 'Pilot', overview='en'), tmdb_ep(2, 'Return')], + }) + meta = Metadata(client, {**SHOW, 'id': 1}, ['fr-FR', 'arte', 'en-US']) + generic = Episode('x-1', 'u', 's', 1, 1, 'Épisode 1', description='arte') + meta.localize(generic) + assert (generic.title, generic.description) == ('Pilot', 'arte') # fr title is a placeholder + titled = Episode('x-2', 'u', 's', 1, 2, 'Titre Arte') + meta.localize(titled) + assert (titled.title, titled.description) == ('Le retour', 'fr') + + +@pytest.fixture +def data_home(tmp_path, monkeypatch): + monkeypatch.setenv('XDG_DATA_HOME', str(tmp_path)) + return tmp_path + + +def test_apply_end_to_end(data_home, monkeypatch): + seasons = {(6, 'fr-FR'): S06_TMDB, (1, 'fr-FR'): S01_TMDB} + client = FakeClient(search=[{'id': 55270, 'name': 'Meurtres à Sandhamn', + 'original_name': 'Morden i Sandhamn', 'first_air_date': '2010-01-09', + 'original_language': 'sv'}], + shows={55270: SHOW}, seasons=seasons) + monkeypatch.setattr(metadata, 'TMDBClient', lambda key: client) + + series = Series('RC-022391', 'Meurtres à Sandhamn', 'fr', original_title='Morden I Sandhamn', + original_language='sv', year=2010) + series.seasons = [Season('RC-022393', 6, 'Saison 6', s06_arte()), + Season('RC-X', 7, 'Saison 7', [arte_ep(7, 1, 4, 88, 'Enquête 10')])] + logs = [] + metadata.apply(series, MetadataConfig(api_key='k'), log=logs.append) + + s6 = series.seasons[0].episodes + assert [(e.label, e.title) for e in s6[:2]] == [('S06E01-E02', 'Le prix à payer'), + ('S06E03-E04', 'Au nom de la vérité')] + assert s6[0].arte_number == 1 and s6[0].description == 'arte 6/1' # no TMDB overview + s7 = series.seasons[1].episodes[0] + assert (s7.label, s7.title) == ('S07E01', 'Enquête 10') # unknown on TMDB: Arte kept + assert (series.tmdb_id, series.tvdb_id, series.year) == (55270, 158851, 2010) + assert any('season 7 not found' in line for line in logs) + assert metadata.load_ids() == {'RC-022391': 55270} + + # Second run: id remembered, no search + client.calls.clear() + metadata.apply(series, MetadataConfig(api_key='k'), log=logs.append) + assert not [c for c in client.calls if c[0] == 'search'] diff --git a/tests/test_misc.py b/tests/test_misc.py new file mode 100644 index 0000000..b3553e6 --- /dev/null +++ b/tests/test_misc.py @@ -0,0 +1,78 @@ +from pathlib import Path + +import pytest + +from arte_dl.arte_api import ArteClient, parse_url +from arte_dl.cli import parse_ranges +from arte_dl.config import Config, ConfigError, load_config +from arte_dl.download import destination, sanitize +from arte_dl.subtitles import vtt_to_srt + +VTT = ( + 'WEBVTT\r\n\r\nSTYLE\r\n::cue(.red) {\r\n color: red;\r\n}\r\n\r\n' + 'NOTE a comment\r\n\r\n' + 'cue-1\r\n00:00:15.320 --> 00:00:16.880 line:91% align:center\r\n' + 'Je la tiens & vite !\r\n\r\n' + '01:02.000 --> 01:03.500\r\nBonjour,\r\nMaria.\r\n\r\n' + '00:01:04.000 --> 00:01:05.000\r\n\r\n' +) + + +def test_vtt_to_srt(): + assert vtt_to_srt(VTT) == ( + '1\n00:00:15,320 --> 00:00:16,880\nJe la tiens & vite !\n\n' + '2\n00:01:02,000 --> 00:01:03,500\nBonjour,\nMaria.\n\n' + ) + + +def test_parse_url(): + assert parse_url('https://www.arte.tv/fr/videos/RC-027900/the-hack-sur-ecoute/') == ('fr', 'RC-027900') + assert parse_url('https://www.arte.tv/de/videos/059534-001-A/x/') == ('de', '059534-001-A') + + +def test_parse_ranges(): + assert parse_ranges('1,3-5, 8') == {1, 3, 4, 5, 8} + + +def test_episode_numbering_and_titles(): + items = [ + {'providerId': '044639-009-A', 'title': 'Twin Peaks - Saison 2 (1/22)', 'subtitle': 'Le géant'}, + {'providerId': 'RC-000000', 'title': 'bonus collection'}, + {'providerId': '125066-002-A', 'title': 'The Hack (2/7)', 'subtitle': None}, + {'providerId': '125066-009-A', 'title': 'Making-of'}, + ] + eps = ArteClient('fr')._episodes(items, 'Series', 2, None) + assert [(e.number, e.title) for e in eps] == [(1, 'Le géant'), (2, 'Épisode 2'), (3, 'Making-of')] + assert eps[0].url == 'https://www.arte.tv/fr/videos/044639-009-A/' + + +def test_destination(): + from arte_dl.arte_api import Episode, Series + cfg = Config() + cfg.output.directory = '/media/series' + series = Series('RC-027900', 'The Hack : sur écoute', 'fr') + ep = Episode('125066-001-A', 'u', 'The Hack : sur écoute', 1, 1, 'Qui ? Quoi / où') + # no year known: the empty "()" disappears + assert destination(series, ep, cfg) == Path( + '/media/series/The Hack - sur écoute/Season 01/The Hack - sur écoute - S01E01 - Qui Quoi - où.mkv') + + series.year, series.tmdb_id = 2025, 12345 + ep.series, ep.season, ep.number, ep.last_number = 'Meurtres à Sandhamn', 6, 1, 2 + cfg.output.series_dir = '{series} ({year}) [tmdbid-{tmdb_id}] {{tvdb-{tvdb_id}}}' + assert destination(series, ep, cfg) == Path( + '/media/series/Meurtres à Sandhamn (2025) [tmdbid-12345]/Season 06/' + 'Meurtres à Sandhamn - S06E01-E02 - Qui Quoi - où.mkv') + assert sanitize('a: b.') == 'a - b' + + +def test_config(tmp_path): + path = tmp_path / 'c.toml' + path.write_text('[video]\nmax_height = 720\n[audio]\ntracks = ["fr"]\n') + cfg = load_config(path) + assert (cfg.video.max_height, cfg.audio.tracks, cfg.video.codecs) == (720, ['fr'], ['hevc', 'avc']) + path.write_text('[video]\nmax_heigth = 720\n') + with pytest.raises(ConfigError, match='max_heigth'): + load_config(path) + path.write_text('[audio]\ntracks = "fr"\n') + with pytest.raises(ConfigError, match='list'): + load_config(path) diff --git a/tests/test_selection.py b/tests/test_selection.py new file mode 100644 index 0000000..779e638 --- /dev/null +++ b/tests/test_selection.py @@ -0,0 +1,96 @@ +import pytest + +from arte_dl.config import Config +from arte_dl.selection import SelectionError, parse_audio, select + + +def audio(name, lang): + return {'format_id': f'VF-STF-audio_0-{name}', 'vcodec': 'none', 'acodec': None, 'language': lang} + + +def video(fid, height, vcodec, tbr): + return {'format_id': fid, 'vcodec': vcodec, 'acodec': 'none', 'height': height, 'tbr': tbr} + + +# Formats as seen on https://www.arte.tv/fr/videos/059534-001-A/ after yt-dlp processing +INFO = { + 'formats': [ + audio('allemand', 'de'), + audio('allemand__audiodescription_', 'de'), + audio('français__audiodescription_', 'fr'), + audio('français__confort_audio_', 'fr'), + audio('suédois__VO_', 'sv'), + audio('français', 'fr'), + video('VF-STF-427', 216, 'avc1.42e00d', 427), + video('VF-STF-2314', 720, 'avc1.4d401f', 2314), + video('VF-STF-2312', 1080, 'avc1.4d0028', 2312), + video('VF-STF-3117', 1080, 'hev1.2.4.L123.B0', 3117), + ], + 'subtitles': { + 'fr-forced': [{'url': 'https://x/st_VF-FRA.m3u8'}], + 'fr': [{'url': 'https://x/st_VO-FRA.m3u8'}], + 'fr-acc': [{'url': 'https://x/st_VF-MAL.m3u8'}], + 'sv': [{'url': 'https://x/st_VO-FRA.m3u8'}], # same file as "fr" + }, +} + + +def test_parse_audio_flags(): + vo = parse_audio(audio('suédois__VO_', 'sv')) + assert (vo.original, vo.kind, vo.title) == (True, 'main', 'Suédois (VO)') + ad = parse_audio(audio('français__audiodescription_', 'fr')) + assert (ad.original, ad.kind, ad.title) == (False, 'ad', 'Français (audiodescription)') + raw = parse_audio(audio('français (confort audio)', 'fr')) # unprocessed id + assert (raw.kind, raw.title) == ('comfort', 'Français (confort audio)') + + +def test_default_selection(): + sel = select(INFO, Config()) + assert sel.video['format_id'] == 'VF-STF-3117' + assert [a.title for a in sel.audio] == ['Suédois (VO)', 'Français'] + assert sel.format_spec == 'VF-STF-3117+VF-STF-audio_0-suédois__VO_+VF-STF-audio_0-français' + assert [s.key for s in sel.subtitles] == ['fr', 'fr-forced'] + assert sel.default_subtitle == 0 # VO audio -> full subs + assert not sel.warnings + + +def test_codec_and_height_preferences(): + cfg = Config() + cfg.video.codecs = ['avc', 'hevc'] + assert select(INFO, cfg).video['format_id'] == 'VF-STF-2312' + cfg.video.max_height = 720 + assert select(INFO, cfg).video['format_id'] == 'VF-STF-2314' + + +def test_french_first_gets_forced_subs(): + cfg = Config() + cfg.audio.tracks = ['fr', 'original'] + sel = select(INFO, cfg) + assert [a.lang for a in sel.audio] == ['fr', 'sv'] + assert sel.subtitles[sel.default_subtitle].key == 'fr-forced' + + +def test_missing_and_duplicate_tracks(): + cfg = Config() + cfg.audio.tracks = ['original', 'en', 'fr-ad'] + cfg.subtitles.tracks = ['fr', 'sv', 'de-forced'] + cfg.subtitles.default = 'none' + sel = select(INFO, cfg) + assert [a.title for a in sel.audio] == ['Suédois (VO)', 'Français (audiodescription)'] + assert [s.key for s in sel.subtitles] == ['fr'] # "sv" is the same file + assert sel.default_subtitle is None + assert sel.warnings == ['audio "en" not available', 'subtitles "de-forced" not available'] + + +def test_fallback_audio_when_nothing_matches(): + cfg = Config() + cfg.audio.tracks = ['it'] + sel = select(INFO, cfg) + assert len(sel.audio) == 1 and 'falling back' in sel.warnings[-1] + + +def test_invalid_audio_spec(): + cfg = Config() + cfg.audio.tracks = ['fr-xyz'] + with pytest.raises(SelectionError): + select(INFO, cfg)