diff --git a/README.md b/README.md index 51871d0..ee18e47 100644 --- a/README.md +++ b/README.md @@ -1,8 +1,9 @@ # arte-dl Wrapper autour de [yt-dlp](https://github.com/yt-dlp/yt-dlp) pour télécharger une série -arte.tv complète (toutes ses saisons) en MKV, selon des préférences de qualité, de pistes -audio et de sous-titres, et la ranger dans une arborescence de vidéothèque (Plex / Jellyfin / Kodi) : +arte.tv complète (toutes ses saisons), un film ou un documentaire en MKV, selon des préférences +de qualité, de pistes audio et de sous-titres, et les ranger dans une arborescence de +vidéothèque (Plex / Jellyfin / Kodi) : ``` Meurtres à Sandhamn (2010)/ @@ -11,6 +12,15 @@ Meurtres à Sandhamn (2010)/ │ └── … └── Season 06/ └── Meurtres à Sandhamn - S06E01-E02 - Le prix à payer.mkv + +Films/ +└── Le Parrain (1972)/ + └── Le Parrain (1972).mkv + +Documentaires/ +└── L'empire LVMH (2026)/ + ├── L'empire LVMH (2026) - part1.mkv + └── L'empire LVMH (2026) - part2.mkv ``` Avec une clé [TMDB](https://www.themoviedb.org/), le nom de la série, l'année, la numérotation @@ -44,10 +54,18 @@ arte-dl -s 1,3-5 -e 1-2 -o ~/Vidéos/Séries https://www.arte.tv/fr/videos/RC-02 # Corriger l'identification TMDB (mémorisée pour les fois suivantes) arte-dl --list --tmdb-id 55270 https://www.arte.tv/fr/videos/RC-022391/meurtres-a-sandhamn/ + +# Un film, une trilogie, un documentaire en deux parties +arte-dl https://www.arte.tv/fr/videos/051404-000-A/les-vieux-espions-vous-saluent-bien/ +arte-dl https://www.arte.tv/fr/videos/RC-028368/le-parrain-la-trilogie/ +arte-dl https://www.arte.tv/fr/videos/RC-028069/l-empire-lvmh/ + +# Le même documentaire rangé comme une mini-série (Season 01/…S01E01…) +arte-dl --as-series https://www.arte.tv/fr/videos/RC-028069/l-empire-lvmh/ ``` `-s` / `-e` portent sur la numérotation finale (celle de TMDB quand elle est utilisée), -celle qu'affiche `--list`. +celle qu'affiche `--list`. Pour un documentaire en plusieurs parties, `-e` choisit les parties. Types d'URL acceptés : @@ -56,6 +74,10 @@ Types d'URL acceptés : | série `…/videos/RC-xxxxxx/…` | toutes les saisons disponibles | | saison `…/videos/RC-xxxxxx/…` | cette saison | | épisode `…/videos/059534-001-A/…` | cet épisode, bien numéroté/rangé | +| film `…/videos/051404-000-A/…` | ce film | +| documentaire `…/videos/RC-xxxxxx/…` | toutes ses parties (1/2, 2/2…) | +| partie `…/videos/122704-002-A/…` | cette partie du documentaire | +| collection `…/videos/RC-xxxxxx/…` (trilogie, cycle) | chacun de ses films / documentaires | Les fichiers déjà présents sont ignorés (`--force` pour les retélécharger) : relancer la commande reprend simplement là où elle s'était arrêtée, ou récupère les nouveaux épisodes. @@ -68,6 +90,8 @@ Voir [`config.example.toml`](config.example.toml) pour toutes les options. Exemp ```toml [output] directory = "~/Vidéos/Séries" +movies_directory = "~/Vidéos/Films" +documentaries_directory = "~/Vidéos/Documentaires" [video] max_height = 1080 @@ -107,10 +131,31 @@ comme le jeton d'accès en lecture (v4) conviennent. (ou seulement « Épisode 3 ») passe à la suivante. Pour deux épisodes fusionnés, « X (part 1) » + « X (part 2) » donnent « X ». +- **Films et documentaires** : recherche parmi les films TMDB (et aussi les séries pour un + documentaire, que TMDB range souvent en mini-série) par titre original, titre Arte, année et + langue, avec les mêmes garde-fous. Le titre et le résumé suivent `languages` ; `--tmdb-id` + accepte `238` (film) ou `tv/12345` (documentaire rangé en série sur TMDB). + +## Films et documentaires + +Arte indique le genre de chaque programme : + +- **film** (genre « Cinéma », y compris les téléfilms) : `Titre (année)/Titre (année).mkv` + dans `movies_directory`. Les collections thématiques auxquelles il appartient (« Comédie », + « Cinéma sous haute tension »…) sont ignorées ; l'URL d'une telle collection (ex. *Le parrain - + La trilogie*) télécharge chacun de ses films ; +- **documentaire unitaire** : même rangement, dans `documentaries_directory` ; +- **documentaire en plusieurs parties** (mini-série documentaire Arte, « (1/2) », « (2/2) ») : + un film en parties, `Titre (année) - part1.mkv`, `- part2.mkv`, que Plex et Jellyfin + enchaînent. Le titre propre à chaque partie est gardé dans les tags du fichier. + `--as-series` le range plutôt comme une série (`Season 01`, `S01E01`). + +Les autres programmes sans série (concert, spectacle…) sont rangés comme des films. + ## Fonctionnement -1. **Structure** : la série, ses saisons et épisodes sont lus via l'API Arte (celle - qu'utilise yt-dlp). Le numéro de saison vient du titre (« Saison 4 »), le numéro d'épisode +1. **Structure** : la série, ses saisons et épisodes (ou le film, les parties du documentaire) + sont lus via l'API Arte (celle qu'utilise yt-dlp). Le numéro de saison vient du titre (« Saison 4 »), le numéro d'épisode du « (1/3) », le titre de l'épisode du sous-titre Arte (à défaut « Épisode N »), puis tout cela est corrigé par TMDB si une clé est configurée. 2. **Sélection** : pour chaque épisode, yt-dlp extrait les formats ; arte-dl choisit la vidéo @@ -120,7 +165,7 @@ comme le jeton d'accès en lecture (v4) conviennent. 3. **Téléchargement** : yt-dlp télécharge et fusionne vidéo + audios, et récupère les sous-titres WebVTT. 4. **Remux final** (ffmpeg) : sous-titres convertis en SRT, langue et titre de chaque piste, - pistes par défaut / forcées / SDH, tags série / saison / épisode. Le fichier est écrit en + pistes par défaut / forcées / SDH, tags série / saison / épisode (ou titre, partie, TMDB / IMDb). Le fichier est écrit en `.part.mkv` puis renommé, donc un fichier `.mkv` présent est toujours complet. La conversion VTT → SRT est faite en Python : les VTT d'Arte (fins de ligne CRLF) donnent diff --git a/arte_dl/arte_api.py b/arte_dl/arte_api.py index 7cf8ec0..01e55ac 100644 --- a/arte_dl/arte_api.py +++ b/arte_dl/arte_api.py @@ -27,6 +27,8 @@ EPISODE_ID_RE = re.compile(r'^\d{6}-\d{3}-[AF]$') COLLECTION_ID_RE = re.compile(r'RC-\d{6}') SEASON_RE = re.compile(r'\b(?:saison|staffel|season|temporada|stagione|sezon)\s*(\d+)', re.I) EPISODE_NUMBER_RE = re.compile(r'\((\d+)\s*/\s*(\d+)\)\s*$') +# Arte genre codes (labels are localized) +GENRE_DOCUMENTARY, GENRE_CINEMA = 1, 2 class ArteError(Exception): @@ -86,6 +88,70 @@ class Series: return [e for s in self.seasons for e in s.episodes] +@dataclass +class Movie: + """A film or a documentary, possibly in several parts (Arte's "(1/2)", "(2/2)").""" + id: str + title: str + lang: str + kind: str = 'film' # "film" or "documentary" + parts: list[Episode] = field(default_factory=list) + total_parts: int = 1 + description: str | None = None + original_title: str | None = None + original_language: str | None = None + year: int | None = None + tmdb_id: int | None = None + tmdb_type: str = 'movie' # a documentary may only exist as a TV mini-series on TMDB + imdb_id: str | None = None + + @property + def episodes(self) -> list[Episode]: + return self.parts + + @property + def multipart(self) -> bool: + return self.total_parts > 1 or len(self.parts) > 1 + + +def kind_of(prog: dict) -> str | None: + """"film", "documentary" or None (series, magazine...) from an OPA program.""" + code = (prog.get('genre') or {}).get('code') + if prog.get('catalogType') == 'MOVIE' or code == GENRE_CINEMA: + return 'film' + return 'documentary' if code == GENRE_DOCUMENTARY else None + + +def _original(prog: dict) -> dict: + return {'original_title': (prog.get('originalTitle') or '').strip(' ()') or None, + 'original_language': (prog.get('originalLanguage') or {}).get('iso6391Code'), + 'year': prog.get('productionYear') or None} + + +def merge(items: list) -> list: + """Merge the Series / Movies resolved separately from one collection's videos.""" + out: dict[str, Series | Movie] = {} + for it in items: + prev = out.setdefault(it.id, it) + if prev is it: + continue + if isinstance(prev, Movie) and isinstance(it, Movie): + known = {p.id for p in prev.parts} + prev.parts += [p for p in it.parts if p.id not in known] + prev.parts.sort(key=lambda p: p.number) + elif isinstance(prev, Series) and isinstance(it, Series): + seasons = {s.id: s for s in prev.seasons} + for season in it.seasons: + if season.id in seasons: + known = {e.id for e in seasons[season.id].episodes} + seasons[season.id].episodes += [e for e in season.episodes if e.id not in known] + seasons[season.id].episodes.sort(key=lambda e: e.number) + else: + prev.seasons.append(season) + prev.seasons.sort(key=lambda s: s.number) + return list(out.values()) + + def parse_url(url: str) -> tuple[str, str]: m = URL_RE.search(url) if not m: @@ -135,53 +201,92 @@ class ArteClient: # --- resolution ------------------------------------------------------- - def resolve(self, url: str) -> Series: - """Series URL -> every season; season URL -> that season; episode URL -> that episode.""" + def resolve(self, url: str, as_series: bool = False) -> list[Series | Movie]: + """Series URL -> every season; season URL -> that season; episode URL -> that episode; + film / documentary -> a Movie (every part of a multi-part documentary); + thematic collection (e.g. a film trilogy) -> each of its videos. + `as_series` keeps multi-part documentaries as mini-series.""" _, pid = parse_url(url) if EPISODE_ID_RE.match(pid): - return self._resolve_episode(pid) + return [self._resolve_video(pid, as_series)] prog = self.program(pid) - if prog.get('catalogType') == 'SEASON': + catalog = prog.get('catalogType') + if catalog == 'TOPIC': + return self._topic(pid, prog, as_series) + if catalog == 'SEASON': series_id = next((p for p in prog.get('parents') or [] if COLLECTION_ID_RE.fullmatch(p) and p != pid), None) if series_id: - return self._series(series_id, only_season=pid) - return self._series(pid) # orphan season: treat as its own series - return self._series(pid, prog=prog) + return [self._series(series_id, only_season=pid)] + return [self._series(pid)] # orphan season: treat as its own series + if catalog == 'MINI_SERIES' and not as_series and self._collection_kind(prog) == 'documentary': + return [self._multipart(pid, prog)] + return [self._series(pid, prog=prog)] - def _resolve_episode(self, episode_id: str) -> Series: - prog = self.program(episode_id) - collections = prog.get('collections') or [] + def _collection_kind(self, prog: dict) -> str | None: + """Some collections have no genre: use their first video's.""" + kind = kind_of(prog) + if kind or prog.get('genre'): + return kind + first = next((v.get('programId') for v in prog.get('videos') or [] + if v.get('kind') == 'SHOW' and EPISODE_ID_RE.match(v.get('programId') or '')), None) + return kind_of(self.program(first)) if first else None + + def _topic(self, pid: str, prog: dict, as_series: bool) -> list[Series | Movie]: + """Thematic collection (trilogy, cycle...): each video on its own.""" + attrs = self.playlist(pid) + items = (attrs or {}).get('items') or self._fallback_items(prog) + ids = list(dict.fromkeys(it.get('providerId') for it in items + if EPISODE_ID_RE.match(it.get('providerId') or ''))) + return merge([self._resolve_video(vid, as_series) for vid in ids]) + + def _resolve_video(self, video_id: str, as_series: bool) -> Series | Movie: + prog = self.program(video_id) + kind = kind_of(prog) + # Thematic collections ("Comédie", "Le parrain - La trilogie") aren't series + collections = [c for c in prog.get('collections') or [] + if c.get('catalogType') not in (None, 'TOPIC')] season = next((c for c in collections if c.get('catalogType') == 'SEASON'), None) - coll = season or (collections[0] if collections else None) + coll = season or (collections[0] if collections and kind != 'film' else None) series_id = None if coll: m = COLLECTION_ID_RE.search(coll.get('url') or '') series_id = m[0] if m else coll.get('collectionId') + if series_id and not season and coll.get('catalogType') == 'MINI_SERIES' and not as_series: + cprog = self.program(series_id) + if self._collection_kind(cprog) == 'documentary': + movie = self._multipart(series_id, cprog, only_part=video_id) + if movie.parts: + return movie + series_id = None if series_id: series = self._series(series_id, only_season=season and season.get('collectionId'), - only_episode=episode_id) + only_episode=video_id) if any(s.episodes for s in series.seasons): return series - # Standalone program, or not found in its collection: single episode, season 1. - title = prog.get('title') or episode_id - ep = Episode(id=episode_id, url=f'https://www.arte.tv/{self.lang}/videos/{episode_id}/', - series=title, season=1, number=1, title=prog.get('subtitle') or title, - description=prog.get('shortDescription')) - return Series(id=episode_id, title=title, lang=self.lang, - seasons=[Season(id=episode_id, number=1, title=title, episodes=[ep])], - original_title=(prog.get('originalTitle') or '').strip(' ()') or None, - original_language=(prog.get('originalLanguage') or {}).get('iso6391Code'), - year=prog.get('productionYear')) + # Film, standalone documentary or program: a single-file movie. + title = (prog.get('title') or video_id).strip() + part = Episode(id=video_id, url=f'https://www.arte.tv/{self.lang}/videos/{video_id}/', + series=title, season=1, number=1, title=title, + description=prog.get('shortDescription'), duration=prog.get('durationSeconds')) + return Movie(id=video_id, title=title, lang=self.lang, kind=kind or 'film', parts=[part], + description=prog.get('shortDescription'), **_original(prog)) + + def _multipart(self, pid: str, prog: dict, only_part: str | None = None) -> Movie: + title = (prog.get('title') or pid).strip() + items = ((self.playlist(pid) or {}).get('items')) or self._fallback_items(prog) + parts = self._episodes(items, title, 1, None) + total = max([p.total or 0 for p in parts] + [len(parts)]) + return Movie(id=pid, title=title, lang=self.lang, kind='documentary', + parts=[p for p in parts if not only_part or p.id == only_part], + total_parts=total, description=prog.get('shortDescription'), **_original(prog)) def _series(self, series_id: str, *, prog: dict | None = None, only_season: str | None = None, only_episode: str | None = None) -> Series: prog = prog or self.program(series_id) series = Series(id=series_id, title=(prog.get('title') or series_id).strip(), lang=self.lang, - original_title=(prog.get('originalTitle') or '').strip(' ()') or None, - original_language=(prog.get('originalLanguage') or {}).get('iso6391Code'), - year=prog.get('productionYear')) + **_original(prog)) refs = [c for c in prog.get('children') or [] if c.get('catalogType') == 'SEASON'] refs.sort(key=lambda c: c.get('order') or 0) diff --git a/arte_dl/cli.py b/arte_dl/cli.py index 56aef10..778d436 100644 --- a/arte_dl/cli.py +++ b/arte_dl/cli.py @@ -2,17 +2,19 @@ from __future__ import annotations import argparse +import re import shutil import subprocess import sys from pathlib import Path from . import __version__ -from .arte_api import ArteClient, ArteError, Series, parse_url +from .arte_api import ArteClient, ArteError, Movie, Series, parse_url from .config import ConfigError, default_config_path, load_config from .download import DownloadError, destination, download, plan from .metadata import TMDBError from .metadata import apply as apply_metadata +from .metadata import apply_movie, parse_ref from .selection import SelectionError @@ -31,92 +33,140 @@ def parse_ranges(spec: str) -> set[int]: return numbers +TMDB_REF_RE = re.compile(r'^(?:(?:movie|tv)/)?\d+$') + + +def tmdb_ref(value: str) -> str: + if not TMDB_REF_RE.match(value): + raise argparse.ArgumentTypeError(f'expected 123, movie/123 or tv/123, not "{value}"') + return value + + def build_parser() -> argparse.ArgumentParser: p = argparse.ArgumentParser( prog='arte-dl', - description='Download arte.tv series (every season) into Series/Season XX/*.mkv') + description='Download arte.tv series (every season) into Series/Season XX/*.mkv, ' + 'films and documentaries into Title (year)/*.mkv') p.add_argument('urls', nargs='+', metavar='URL', - help='arte.tv series (RC-xxxxxx), season or episode URL') + help='arte.tv series (RC-xxxxxx), season, episode, film, documentary or collection URL') p.add_argument('-c', '--config', type=Path, help=f'TOML config file (default: {default_config_path()})') p.add_argument('-o', '--output', help='output root directory (overrides [output] directory)') p.add_argument('-s', '--seasons', type=parse_ranges, help='only these seasons, e.g. "1,3-5"') - p.add_argument('-e', '--episodes', type=parse_ranges, help='only these episode numbers') + p.add_argument('-e', '--episodes', type=parse_ranges, + help='only these episode numbers (documentaries: part numbers)') p.add_argument('-l', '--list', action='store_true', help='list seasons / episodes and destination paths, then exit') p.add_argument('-n', '--dry-run', action='store_true', help='also show the selected tracks for each episode, without downloading') p.add_argument('-f', '--force', action='store_true', help='re-download existing files') - p.add_argument('--tmdb-id', type=int, metavar='ID', - help='TMDB show id to use instead of searching, with a single URL (remembered for next runs)') + p.add_argument('--as-series', action='store_true', + help='file a documentary in several parts as a mini-series (Season 01/…E01) ' + 'instead of a movie in parts') + p.add_argument('--tmdb-id', type=tmdb_ref, metavar='ID', + help='TMDB id to use instead of searching, with a single URL (remembered for next runs): ' + 'show id for a series, movie id for a film, "tv/ID" for a documentary filed as a TV show') p.add_argument('--no-metadata', action='store_true', help="don't query TMDB, use Arte data only") p.add_argument('-V', '--version', action='version', version=f'%(prog)s {__version__}') return p -def _filter(series: Series, args) -> None: - for season in series.seasons: +def _filter(item: Series | Movie, args) -> None: + if isinstance(item, Movie): + item.parts = [p for p in item.parts if not args.episodes or p.number in args.episodes] + return + for season in item.seasons: season.episodes = [e for e in season.episodes if not args.episodes or args.episodes & set(e.numbers)] - series.seasons = [s for s in series.seasons - if s.episodes and (not args.seasons or s.number in args.seasons)] + item.seasons = [s for s in item.seasons + if s.episodes and (not args.seasons or s.number in args.seasons)] + + +def _apply_metadata(item: Series | Movie, cfg, args) -> None: + meta = cfg.metadata + if meta.provider != 'tmdb' or args.no_metadata: + return + if not meta.key: + print(' TMDB: no API key ([metadata] api_key or TMDB_API_KEY) — using Arte metadata') + return + try: + if isinstance(item, Movie): + apply_movie(item, meta, forced=args.tmdb_id) + else: + kind, forced = parse_ref(args.tmdb_id, 'tv') or ('tv', None) + if kind != 'tv': + print(f' TMDB: --tmdb-id {args.tmdb_id} is not a TV show — ignored', file=sys.stderr) + forced = None + apply_metadata(item, meta, forced_id=forced) + except TMDBError as e: + print(f' TMDB: {e} — keeping Arte metadata', file=sys.stderr) + + +def _fetch(item: Series | Movie, ep, label: str, cfg, args, note: str = '') -> bool: + """Show one episode / part and download it. False on failure.""" + dest = destination(item, ep, cfg) + exists = dest.exists() + print(f' {label} [{ep.id}] {ep.title}{note}') + print(f' -> {dest}{" (exists)" if exists else ""}') + if args.list or (exists and not args.force): + return True + try: + info, sel = plan(ep, cfg) + h = sel.video + print(f' video : {h["format_id"]} {h.get("height")}p {h.get("vcodec")}') + print(f' audio : {", ".join(a.title for a in sel.audio) or "(muxed)"}') + print(' subs : ' + (', '.join( + s.title + (' *' if i == sel.default_subtitle else '') + for i, s in enumerate(sel.subtitles)) or '-')) + for w in sel.warnings: + print(f' warning: {w}') + if not args.dry_run: + download(item, ep, info, sel, dest, cfg) + print(' done') + return True + except KeyboardInterrupt: + raise + except (DownloadError, SelectionError, subprocess.CalledProcessError) as e: + print(f' FAILED: {e}', file=sys.stderr) + except Exception as e: # keep going with the next episode + print(f' FAILED: {type(e).__name__}: {e}', file=sys.stderr) + return False def process(url: str, cfg, args) -> tuple[int, int]: """Returns (ok, failed) episode counts.""" lang, _ = parse_url(url) - series = ArteClient(lang).resolve(url) - total = len(series.episodes) - print(f'== {series.title} ({series.id}) — {len(series.seasons)} season(s), {total} episode(s)') - if series.unavailable: - print(f' not available online: {", ".join(series.unavailable)}') - - meta = cfg.metadata - if meta.provider == 'tmdb' and not args.no_metadata: - if meta.key: - try: - apply_metadata(series, meta, forced_id=args.tmdb_id) - except TMDBError as e: - print(f' TMDB: {e} — keeping Arte metadata', file=sys.stderr) - else: - print(' TMDB: no API key ([metadata] api_key or TMDB_API_KEY) — using Arte metadata') - _filter(series, args) - ok = failed = 0 - for season in series.seasons: - print(f'-- Season {season.number}: {season.title}') - for ep in season.episodes: - dest = destination(series, ep, cfg) - exists = dest.exists() - arte_label = f'S{ep.arte_season:02d}E{ep.arte_number:02d}' - renumbered = f' (Arte {arte_label})' if arte_label != ep.label else '' - print(f' {ep.label} [{ep.id}] {ep.title}{renumbered}') - print(f' -> {dest}{" (exists)" if exists else ""}') - if args.list or (exists and not args.force): - ok += 1 - continue - try: - info, sel = plan(ep, cfg) - h = sel.video - print(f' video : {h["format_id"]} {h.get("height")}p {h.get("vcodec")}') - print(f' audio : {", ".join(a.title for a in sel.audio) or "(muxed)"}') - print(' subs : ' + (', '.join( - s.title + (' *' if i == sel.default_subtitle else '') - for i, s in enumerate(sel.subtitles)) or '-')) - for w in sel.warnings: - print(f' warning: {w}') - if not args.dry_run: - download(series, ep, info, sel, dest, cfg) - print(' done') - ok += 1 - except KeyboardInterrupt: - raise - except (DownloadError, SelectionError, subprocess.CalledProcessError) as e: - failed += 1 - print(f' FAILED: {e}', file=sys.stderr) - except Exception as e: # keep going with the next episode - failed += 1 - print(f' FAILED: {type(e).__name__}: {e}', file=sys.stderr) + for item in ArteClient(lang).resolve(url, as_series=args.as_series): + if isinstance(item, Movie): + kind = 'Documentary' if item.kind == 'documentary' else 'Film' + parts = f', {len(item.parts)}/{item.total_parts} part(s)' if item.multipart else '' + print(f'== {kind}: {item.title} ({item.year or "?"}) [{item.id}]{parts}') + else: + print(f'== {item.title} ({item.id}) — {len(item.seasons)} season(s), ' + f'{len(item.episodes)} episode(s)') + if item.unavailable: + print(f' not available online: {", ".join(item.unavailable)}') + _apply_metadata(item, cfg, args) + _filter(item, args) + + if isinstance(item, Movie): + for part in item.parts: + label = f'part {part.number}/{item.total_parts}' if item.multipart else 'film' + if _fetch(item, part, label, cfg, args): + ok += 1 + else: + failed += 1 + continue + for season in item.seasons: + print(f'-- Season {season.number}: {season.title}') + for ep in season.episodes: + arte_label = f'S{ep.arte_season:02d}E{ep.arte_number:02d}' + renumbered = f' (Arte {arte_label})' if arte_label != ep.label else '' + if _fetch(item, ep, ep.label, cfg, args, renumbered): + ok += 1 + else: + failed += 1 return ok, failed @@ -131,8 +181,9 @@ def main(argv: list[str] | None = None) -> int: except ConfigError as e: print(f'config error: {e}', file=sys.stderr) return 2 - if args.output: + if args.output: # a single root for series, films and documentaries cfg.output.directory = args.output + cfg.output.movies_directory = cfg.output.documentaries_directory = '' if not (args.list or args.dry_run) and not shutil.which('ffmpeg'): print('ffmpeg not found in PATH', file=sys.stderr) return 2 diff --git a/arte_dl/config.py b/arte_dl/config.py index bd42e11..069ba14 100644 --- a/arte_dl/config.py +++ b/arte_dl/config.py @@ -23,6 +23,13 @@ class OutputConfig: series_dir: str = '{series} ({year})' season_dir: str = 'Season {season:02d}' filename: str = '{series} - S{season:02d}E{episode:02d} - {title}' + # Roots for films and documentaries; empty: `directory` (documentaries: `movies_directory`) + movies_directory: str = '' + documentaries_directory: str = '' + movie_dir: str = '{title} ({year})' + movie_filename: str = '{title} ({year})' + # Appended to movie_filename for a documentary in several parts ("part1": Plex and Jellyfin) + part_suffix: str = ' - part{part}' @dataclass @@ -75,7 +82,17 @@ class Config: @property def output_dir(self) -> Path: - return Path(os.path.expandvars(self.output.directory)).expanduser() + return _expand(self.output.directory) + + def movie_root(self, kind: str) -> Path: + """Root directory for a "film" or a "documentary".""" + o = self.output + path = (o.documentaries_directory if kind == 'documentary' else '') or o.movies_directory + return _expand(path) if path else self.output_dir + + +def _expand(path: str) -> Path: + return Path(os.path.expandvars(path)).expanduser() def default_config_path() -> Path: diff --git a/arte_dl/download.py b/arte_dl/download.py index 32219ab..7fef14e 100644 --- a/arte_dl/download.py +++ b/arte_dl/download.py @@ -14,7 +14,7 @@ from pathlib import Path from yt_dlp import YoutubeDL from yt_dlp.utils import ISO639Utils -from .arte_api import Episode, Series +from .arte_api import Episode, Movie, Series from .config import Config from .selection import Selection, select from .subtitles import vtt_to_srt @@ -47,20 +47,37 @@ class EpisodeNumbers: _EMPTY_GROUP_RE = re.compile(r'\s*[(\[{]\s*(?:[a-z]+-)?\s*[)\]}]') -def destination(series: Series, ep: Episode, cfg: Config) -> Path: +def _render(templates: list[str], fields: dict) -> list[str]: + try: + return [_EMPTY_GROUP_RE.sub('', tpl.format(**fields)).strip() for tpl in templates] + except (KeyError, ValueError, AttributeError) as e: + raise DownloadError(f'Invalid output template: {e!r}') from None + + +def destination(item: Series | Movie, ep: Episode, cfg: Config) -> Path: + if isinstance(item, Movie): + return movie_destination(item, ep, cfg) + series = item fields = {'series': sanitize(ep.series), 'year': series.year or '', 'tmdb_id': series.tmdb_id or '', 'tvdb_id': series.tvdb_id or '', 'season': ep.season, 'episode': EpisodeNumbers(ep.numbers), 'title': sanitize(ep.title), 'id': ep.id} - try: - parts = [_EMPTY_GROUP_RE.sub('', tpl.format(**fields)).strip() for tpl in ( - cfg.output.series_dir, cfg.output.season_dir, cfg.output.filename)] - except (KeyError, ValueError, AttributeError) as e: - raise DownloadError(f'Invalid output template: {e!r}') from None - series_dir, season_dir, filename = parts + series_dir, season_dir, filename = _render( + [cfg.output.series_dir, cfg.output.season_dir, cfg.output.filename], fields) return cfg.output_dir / series_dir / season_dir / f'{filename}.mkv' +def movie_destination(movie: Movie, part: Episode, cfg: Config) -> Path: + o = cfg.output + fields = {'title': sanitize(movie.title), 'year': movie.year or '', + 'original_title': sanitize(movie.original_title or movie.title), + 'tmdb_id': movie.tmdb_id or '', 'imdb_id': movie.imdb_id or '', 'id': movie.id, + 'part': part.number} + filename_tpl = o.movie_filename + (o.part_suffix if movie.multipart else '') + movie_dir, filename = _render([o.movie_dir, filename_tpl], fields) + return cfg.movie_root(movie.kind) / movie_dir / f'{filename}.mkv' + + def _ydl_params(**extra) -> dict: return { 'quiet': True, @@ -93,7 +110,7 @@ def _lang3(lang: str | None) -> str: return (lang and ISO639Utils.short2long(lang)) or 'und' -def download(series: Series, ep: Episode, info: dict, sel: Selection, dest: Path, cfg: Config) -> None: +def download(item: Series | Movie, ep: Episode, info: dict, sel: Selection, dest: Path, cfg: Config) -> None: work = cfg.output_dir / '.arte-dl-tmp' / ep.id shutil.rmtree(work, ignore_errors=True) work.mkdir(parents=True) @@ -129,7 +146,7 @@ def download(series: Series, ep: Episode, info: dict, sel: Selection, dest: Path dest.parent.mkdir(parents=True, exist_ok=True) part = dest.with_name(dest.stem + '.part.mkv') - subprocess.run(_mux_command(media, sub_files, series, ep, sel, part), check=True) + subprocess.run(_mux_command(media, sub_files, item, ep, sel, part), check=True) part.replace(dest) shutil.rmtree(work, ignore_errors=True) try: @@ -138,7 +155,37 @@ def download(series: Series, ep: Episode, info: dict, sel: Selection, dest: Path pass -def _mux_command(media: Path, subs: list[Path], series: Series, ep: Episode, sel: Selection, +def _tags(item: Series | Movie, ep: Episode) -> dict[str, str]: + if isinstance(item, Movie): + title = item.title + if item.multipart: + title += f' ({ep.number}/{item.total_parts})' + if ep.title != item.title: + title += f' - {ep.title}' + return { + 'title': title, + 'part_number': str(ep.number) if item.multipart else '', + 'total_parts': str(item.total_parts) if item.multipart else '', + 'description': (ep.description if item.multipart else None) or item.description or '', + 'comment': ep.url, + 'tmdb': f'{item.tmdb_type}/{item.tmdb_id}' if item.tmdb_id else '', + 'imdb': item.imdb_id or '', + 'date': str(item.year or ''), + } + return { + 'title': ep.title, + 'show': ep.series, + 'season_number': str(ep.season), + 'episode_sort': str(ep.number), + 'episode_id': ep.label, + 'description': ep.description or '', + 'comment': ep.url, + 'tmdb': f'tv/{item.tmdb_id}' if item.tmdb_id else '', + 'date': str(item.year or ''), + } + + +def _mux_command(media: Path, subs: list[Path], item: Series | Movie, ep: Episode, sel: Selection, out: Path) -> list[str]: cmd = ['ffmpeg', '-hide_banner', '-loglevel', 'error', '-nostdin', '-y', '-i', str(media)] for path in subs: @@ -148,18 +195,7 @@ def _mux_command(media: Path, subs: list[Path], series: Series, ep: Episode, sel cmd += ['-map', f'{i + 1}:0'] cmd += ['-c', 'copy', '-c:s', 'srt', '-map_metadata', '-1'] - tags = { - 'title': ep.title, - 'show': ep.series, - 'season_number': str(ep.season), - 'episode_sort': str(ep.number), - 'episode_id': ep.label, - 'description': ep.description or '', - 'comment': ep.url, - 'tmdb': f'tv/{series.tmdb_id}' if series.tmdb_id else '', - 'date': str(series.year or ''), - } - for k, v in tags.items(): + for k, v in _tags(item, ep).items(): if v: cmd += ['-metadata', f'{k}={v}'] diff --git a/arte_dl/metadata.py b/arte_dl/metadata.py index cfef7a6..c05368f 100644 --- a/arte_dl/metadata.py +++ b/arte_dl/metadata.py @@ -4,6 +4,9 @@ Arte's numbering doesn't always follow the reference one: e.g. from season 6, "Meurtres à Sandhamn" episodes are 88 min on Arte but two 45 min episodes on TMDB, so Arte's S06E01 becomes S06E01-E02 (multi-episode file, understood by Plex and Jellyfin). + +Films and documentaries are matched against TMDB movies (documentaries also +against TV shows, where TMDB often files multi-part ones). """ from __future__ import annotations @@ -20,7 +23,7 @@ from difflib import SequenceMatcher from functools import lru_cache from pathlib import Path -from .arte_api import Episode, Series +from .arte_api import Episode, Movie, Series from .config import MetadataConfig API = 'https://api.themoviedb.org/3' @@ -75,17 +78,20 @@ class TMDBClient: raise AssertionError('unreachable') @lru_cache(maxsize=None) - def search(self, query: str, language: str) -> tuple[dict, ...]: - data = self._get('/search/tv', query=query, language=language, include_adult='false') + def search(self, query: str, language: str, kind: str = 'tv') -> tuple[dict, ...]: + data = self._get(f'/search/{kind}', query=query, language=language, include_adult='false') return tuple((data or {}).get('results') or ()) @lru_cache(maxsize=None) - def show(self, show_id: int) -> dict: - data = self._get(f'/tv/{show_id}', append_to_response='external_ids,translations') + def details(self, kind: str, tmdb_id: int) -> dict: + data = self._get(f'/{kind}/{tmdb_id}', append_to_response='external_ids,translations') if not data: - raise TMDBError(f'Unknown TMDB show {show_id}') + raise TMDBError(f'Unknown TMDB {kind} {tmdb_id}') return data + def show(self, show_id: int) -> dict: + return self.details('tv', show_id) + @lru_cache(maxsize=None) def season(self, show_id: int, number: int, language: str) -> dict | None: return self._get(f'/tv/{show_id}/season/{number}', language=language) @@ -98,14 +104,15 @@ def _ids_path() -> Path: return Path(base) / 'arte-dl' / 'tmdb-ids.json' -def load_ids() -> dict[str, int]: +def load_ids() -> dict[str, int | str]: + """Arte id -> TMDB show id, or "movie/ID" / "tv/ID" for films and documentaries.""" try: return json.loads(_ids_path().read_text()) except (FileNotFoundError, ValueError): return {} -def save_id(arte_id: str, tmdb_id: int) -> None: +def save_id(arte_id: str, tmdb_id: int | str) -> None: ids = load_ids() if ids.get(arte_id) == tmdb_id: return @@ -127,12 +134,21 @@ def _similarity(a: str, b: str) -> float: return SequenceMatcher(None, normalize(a), normalize(b)).ratio() -def score_candidate(series: Series, queries: list[str], result: dict) -> tuple[float, float]: +def _name(result: dict) -> str | None: + return result.get('name') or result.get('title') # TV show / movie + + +def _date(result: dict) -> str: + return result.get('first_air_date') or result.get('release_date') or '' + + +def score_candidate(series: Series | Movie, queries: list[str], result: dict) -> tuple[float, float]: """Returns (score, name similarity).""" sim = max((_similarity(q, name) for q in queries - for name in (result.get('name'), result.get('original_name')) if name), default=0) + for name in (_name(result), result.get('original_name') or result.get('original_title')) + if name), default=0) score = sim - year = (result.get('first_air_date') or '')[:4] + year = _date(result)[:4] if series.year and year.isdigit(): diff = abs(int(year) - series.year) score += 0.2 if diff == 0 else 0.1 if diff == 1 else -0.5 if diff > 2 else 0 @@ -141,29 +157,46 @@ def score_candidate(series: Series, queries: list[str], result: dict) -> tuple[f return score, sim -def find_show(client: TMDBClient, series: Series, language: str) -> tuple[int | None, str]: - """Search TMDB for the series. Returns (id or None, explanation).""" - queries = list(dict.fromkeys(q for q in (series.original_title, series.title) if q)) +def _queries(item: Series | Movie) -> list[str]: + queries = list(dict.fromkeys(q for q in (item.original_title, item.title) if q)) # "The Hack : sur écoute" -> also try "The Hack" queries += [q.split(' : ')[0] for q in queries if ' : ' in q and q.split(' : ')[0] not in queries] - candidates = {} - for q in queries: - for r in client.search(q, language)[:10]: - candidates[r['id']] = r - scored = sorted(((score_candidate(series, queries, r), r) for r in candidates.values()), + return queries + + +def _pick(item: Series | Movie, queries: list[str], + candidates: dict) -> tuple[object | None, str]: + """Best of {key: result}, or None with an explanation when unsure.""" + scored = sorted(((score_candidate(item, queries, r), key, r) for key, r in candidates.items()), key=lambda x: x[0][0], reverse=True) if not scored: return None, f'no TMDB result for {queries}' def describe(r): - return f'{r.get("name")} ({(r.get("first_air_date") or "?")[:4]}) id={r["id"]}' + return f'{_name(r)} ({(_date(r) or "?")[:4]}) id={r["id"]}' - (score, sim), best = scored[0] + (score, sim), key, best = scored[0] if sim < 0.8 or score < 0.95: return None, f'no confident match (best: {describe(best)}, score {score:.2f})' if len(scored) > 1 and scored[1][0][0] > score - 0.05: - return None, f'ambiguous: {describe(best)} / {describe(scored[1][1])}' - return best['id'], f'matched {describe(best)}' + return None, f'ambiguous: {describe(best)} / {describe(scored[1][2])}' + return key, f'matched {describe(best)}' + + +def find_show(client: TMDBClient, series: Series, language: str) -> tuple[int | None, str]: + """Search TMDB for the series. Returns (id or None, explanation).""" + queries = _queries(series) + candidates = {r['id']: r for q in queries for r in client.search(q, language)[:10]} + return _pick(series, queries, candidates) + + +def find_movie(client: TMDBClient, movie: Movie, language: str, + kinds: tuple[str, ...] = ('movie',)) -> tuple[tuple[str, int] | None, str]: + """Search TMDB for a film / documentary. Returns ((kind, id) or None, explanation).""" + queries = _queries(movie) + candidates = {(kind, r['id']): r for kind in kinds for q in queries + for r in client.search(q, language, kind)[:10]} + return _pick(movie, queries, candidates) # --- episode alignment ------------------------------------------------------------- @@ -241,6 +274,31 @@ class Metadata: def show_id(self) -> int: return self.show['id'] + def translated(self, lang: str, key: str) -> str | None: + """Show / movie `key` ("name", "title", "overview") in a TMDB language.""" + translations = (self.show.get('translations') or {}).get('translations') or [] + iso, _, region = lang.partition('-') + return next((t['data'].get(key) for t in translations if t.get('iso_639_1') == iso + and (not region or t.get('iso_3166_1') == region) and t['data'].get(key)), None) + + def movie_texts(self, arte_title: str, arte_description: str | None) -> tuple[str, str | None]: + """(title, synopsis) of a movie (or TV show) following the language priority.""" + key = 'title' if 'title' in self.show else 'name' + original = self.show.get(f'original_{key}') + title = overview = None + for lang in self.languages: + if lang == 'arte': + t, o = arte_title, arte_description + elif lang == 'original': + t, o = original, self.translated(self.tmdb_language(lang), 'overview') + else: + # Untranslated: TMDB (and Plex / Jellyfin) show the original title in that language + t, o = self.translated(lang, key) or original, self.translated(lang, 'overview') + title, overview = title or t, overview or o + if title and overview: + break + return title or arte_title, overview or arte_description + def tmdb_language(self, lang: str) -> str: return (self.show.get('original_language') or 'en') if lang == 'original' else lang @@ -292,7 +350,8 @@ def apply(series: Series, cfg: MetadataConfig, forced_id: int | None = None, client = TMDBClient(cfg.key) first_lang = next((l for l in cfg.languages if l not in ('arte', 'original')), 'en-US') - show_id = forced_id or load_ids().get(series.id) + remembered = load_ids().get(series.id) + show_id = forced_id or (remembered if isinstance(remembered, int) else None) if show_id: how = 'forced' if forced_id else 'remembered' else: @@ -328,3 +387,44 @@ def apply(series: Series, cfg: MetadataConfig, forced_id: int | None = None, nums = mapping[ep.id] ep.number, ep.last_number = nums[0], (nums[-1] if len(nums) > 1 else None) meta.localize(ep) + + +def parse_ref(ref: int | str | None, default_kind: str = 'movie') -> tuple[str, int] | None: + """123, "123", "movie/123", "tv/123" -> (kind, id).""" + if ref is None: + return None + kind, _, num = str(ref).rpartition('/') + return kind or default_kind, int(num) + + +def apply_movie(movie: Movie, cfg: MetadataConfig, forced: str | None = None, log=print) -> None: + """Title, year and synopsis of a film / documentary from TMDB. Keeps Arte data when unsure.""" + client = TMDBClient(cfg.key) + first_lang = next((l for l in cfg.languages if l not in ('arte', 'original')), 'en-US') + # A documentary may be filed on TMDB as a movie or as a TV (mini-)series + kinds = ('movie', 'tv') if movie.kind == 'documentary' else ('movie',) + + # An int remembered for a documentary comes from a run as a series (--as-series) + ref = parse_ref(forced) or parse_ref(load_ids().get(movie.id), 'tv') + if ref: + how = 'forced' if forced else 'remembered' + else: + ref, how = find_movie(client, movie, first_lang, kinds) + if not ref: + log(f' TMDB: {how} — keeping Arte metadata (use --tmdb-id to set it)') + return + kind, tmdb_id = ref + details = client.details(kind, tmdb_id) + save_id(movie.id, f'{kind}/{tmdb_id}') + + meta = Metadata(client, details, cfg.languages) + movie.tmdb_id, movie.tmdb_type = tmdb_id, kind + movie.imdb_id = (details.get('external_ids') or {}).get('imdb_id') or details.get('imdb_id') + date = _date(details) + movie.year = int(date[:4]) if date[:4].isdigit() else movie.year + movie.title, movie.description = meta.movie_texts(movie.title, movie.description) + for part in movie.parts: + part.series = movie.title + if not movie.multipart: + part.title = movie.title + log(f' TMDB: {how} — "{movie.title}" ({movie.year}) https://www.themoviedb.org/{kind}/{tmdb_id}') diff --git a/arte_dl/selection.py b/arte_dl/selection.py index 37f697b..6620185 100644 --- a/arte_dl/selection.py +++ b/arte_dl/selection.py @@ -123,7 +123,10 @@ def pick_video(formats: list[dict], cfg: Config) -> dict: if not videos: raise SelectionError('No video format found') capped = [f for f in videos if not cfg.video.max_height or f['height'] <= cfg.video.max_height] - return max(capped or videos, key=lambda f: ( + if not capped: # nothing that small: the lowest height available + lowest = min(f['height'] for f in videos) + capped = [f for f in videos if f['height'] == lowest] + return max(capped, key=lambda f: ( f['height'], _codec_rank(f.get('vcodec'), cfg.video.codecs), f.get('tbr') or 0)) diff --git a/config.example.toml b/config.example.toml index d8f363c..a11f513 100644 --- a/config.example.toml +++ b/config.example.toml @@ -2,8 +2,12 @@ # Toutes les options sont facultatives ; les valeurs ci-dessous sont les valeurs par défaut. [output] -# Racine de la vidéothèque (~ et $VARIABLES acceptés) +# Racine de la vidéothèque des séries (~ et $VARIABLES acceptés) directory = "." +# Racines des films et des documentaires ; vide : `directory` (documentaires : `movies_directory`) +# -o / --output remplace ces trois racines par une seule. +movies_directory = "" +documentaries_directory = "" # Champs disponibles : {series} {year} {tmdb_id} {tvdb_id} {season} {episode} {title} {id} # Un groupe vide (année ou id inconnus) est retiré : "Série ()" -> "Série". # Pour forcer l'identification par Jellyfin : "{series} ({year}) [tmdbid-{tmdb_id}]" @@ -11,6 +15,13 @@ series_dir = "{series} ({year})" season_dir = "Season {season:02d}" # {episode:02d} donne "01", ou "01-E02" pour un fichier qui contient deux épisodes TMDB filename = "{series} - S{season:02d}E{episode:02d} - {title}" +# Films et documentaires. Champs : {title} {year} {original_title} {tmdb_id} {imdb_id} {id} +# Pour Jellyfin : "{title} ({year}) [tmdbid-{tmdb_id}]", pour Plex : "{title} ({year}) {{tmdb-{tmdb_id}}}" +movie_dir = "{title} ({year})" +movie_filename = "{title} ({year})" +# Ajouté au nom d'un documentaire en plusieurs parties ({part} : numéro de la partie) ; +# "part1", "part2"… est reconnu par Plex et Jellyfin, qui les enchaînent comme un seul film. +part_suffix = " - part{part}" [video] # Hauteur maximale (216, 360, 432, 720, 1080) @@ -41,7 +52,7 @@ provider = "tmdb" # Clé API TMDB (v3) ou jeton d'accès en lecture (v4). La variable d'environnement # TMDB_API_KEY fonctionne aussi. Sans clé : données Arte uniquement. api_key = "" -# Ordre de priorité pour le nom de la série, les titres et les résumés d'épisodes. +# Ordre de priorité pour le nom de la série, les titres (épisodes, films) et les résumés. # "fr-FR", "en-US", "de"… -> langue TMDB # "arte" -> titres / résumés Arte # "original" -> langue originale de la série diff --git a/pyproject.toml b/pyproject.toml index 6d11514..bf60383 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "arte-dl" version = "0.1.0" -description = "Download whole arte.tv series with yt-dlp into a tidy Series/Season XX/ tree of MKV files" +description = "Download arte.tv series, films and documentaries with yt-dlp into a tidy Plex / Jellyfin tree of MKV files" readme = "README.md" requires-python = ">=3.10" dependencies = [ diff --git a/tests/test_movies.py b/tests/test_movies.py new file mode 100644 index 0000000..d743001 --- /dev/null +++ b/tests/test_movies.py @@ -0,0 +1,163 @@ +from pathlib import Path + +from arte_dl import metadata +from arte_dl.arte_api import ArteClient, Episode, Movie, kind_of, merge +from arte_dl.config import Config, MetadataConfig +from arte_dl.download import _tags, destination +from arte_dl.metadata import find_movie, parse_ref + + +# OPA programs, trimmed from the API (September 2026) +FILM = {'programId': '051404-000-A', 'catalogType': 'MOVIE', 'genre': {'code': 2, 'label': 'Cinéma'}, + 'title': 'Les vieux espions vous saluent bien', 'originalTitle': '', 'productionYear': 2017, + 'originalLanguage': {'iso6391Code': 'de'}, 'durationSeconds': 5132, 'shortDescription': 'film', + 'collections': [{'collectionId': 'RC-027882', 'catalogType': 'TOPIC', 'title': 'Comédie', + 'url': 'https://www.arte.tv/fr/videos/RC-027882/comedie/'}]} +GODFATHER = {'programId': '045559-000-A', 'catalogType': 'MINI_EPISODE', + 'genre': {'code': 2, 'label': 'Cinéma'}, 'title': 'Le parrain', + 'originalTitle': '(The Godfather)', 'productionYear': 1972, + 'originalLanguage': {'iso6391Code': 'en'}, + 'collections': [{'collectionId': 'RC-028368', 'catalogType': 'TOPIC'}]} +LVMH = {'programId': 'RC-028069', 'catalogType': 'MINI_SERIES', + 'genre': {'code': 1, 'label': 'Documentaires et reportages'}, 'title': "L'empire LVMH", + 'originalTitle': 'LVMH – das Imperium der Luxusmarken', 'productionYear': 2026, + 'shortDescription': 'saga'} +LVMH_PART2 = {'programId': '122704-002-A', 'catalogType': 'MINI_EPISODE', + 'genre': {'code': 1}, 'title': "L'empire LVMH (2/2)", + 'collections': [{'collectionId': 'RC-028069', 'catalogType': 'MINI_SERIES', + 'url': 'https://www.arte.tv/fr/videos/RC-028069/l-empire-lvmh/'}]} +LVMH_ITEMS = [ + {'providerId': '122704-001-A', 'title': "L'empire LVMH (1/2)", 'subtitle': 'Un morceau du rêve', + 'duration': {'seconds': 3596}}, + {'providerId': '122704-002-A', 'title': "L'empire LVMH (2/2)", 'subtitle': "L’État dans l'État", + 'duration': {'seconds': 3680}}, +] +TRILOGY = {'programId': 'RC-028368', 'catalogType': 'TOPIC', 'title': 'Le parrain - La trilogie'} + + +class FakeArte(ArteClient): + programs = {p['programId']: p for p in (FILM, GODFATHER, LVMH, LVMH_PART2, TRILOGY, + {**GODFATHER, 'programId': '045560-000-A', + 'title': 'Le parrain II', 'productionYear': 1975})} + playlists = {'RC-028069': LVMH_ITEMS, + 'RC-028368': [{'providerId': '045559-000-A'}, {'providerId': '045560-000-A'}]} + + def program(self, pid): + return self.programs[pid] + + def playlist(self, cid): + return {'items': self.playlists.get(cid, []), 'metadata': {}} + + +def test_kind_of(): + assert kind_of(FILM) == kind_of(GODFATHER) == 'film' + assert kind_of(LVMH) == 'documentary' + assert kind_of({'catalogType': 'SERIES', 'genre': {'code': 3}}) is None + + +def test_film_is_not_an_episode_of_its_topic(): + [movie] = FakeArte('fr').resolve('https://www.arte.tv/fr/videos/051404-000-A/x/') + assert isinstance(movie, Movie) and movie.kind == 'film' and not movie.multipart + assert (movie.title, movie.year, movie.original_language) == (FILM['title'], 2017, 'de') + + +def test_trilogy_gives_each_film(): + movies = FakeArte('fr').resolve('https://www.arte.tv/fr/videos/RC-028368/le-parrain/') + assert [(m.title, m.year, m.original_title) for m in movies] == [ + ('Le parrain', 1972, 'The Godfather'), ('Le parrain II', 1975, 'The Godfather')] + + +def test_documentary_in_parts(): + client = FakeArte('fr') + [doc] = client.resolve('https://www.arte.tv/fr/videos/RC-028069/l-empire-lvmh/') + assert (doc.kind, doc.total_parts, doc.multipart) == ('documentary', 2, True) + assert [(p.number, p.title) for p in doc.parts] == [(1, 'Un morceau du rêve'), (2, "L’État dans l'État")] + # A single part's URL: that part of the whole documentary + [doc] = client.resolve('https://www.arte.tv/fr/videos/122704-002-A/x/') + assert (doc.id, [p.number for p in doc.parts], doc.multipart) == ('RC-028069', [2], True) + # Or as a mini-series + [series] = client.resolve('https://www.arte.tv/fr/videos/RC-028069/l-empire-lvmh/', as_series=True) + assert [e.label for e in series.episodes] == ['S01E01', 'S01E02'] + + +def test_merge_parts(): + a = Movie('RC-1', 'Doc', 'fr', parts=[Episode('p2', 'u', 'Doc', 1, 2, 'b')], total_parts=2) + b = Movie('RC-1', 'Doc', 'fr', parts=[Episode('p1', 'u', 'Doc', 1, 1, 'a')], total_parts=2) + [m] = merge([a, b]) + assert [p.id for p in m.parts] == ['p1', 'p2'] + + +def test_movie_destination(): + cfg = Config() + cfg.output.directory = '/media/series' + cfg.output.movies_directory = '/media/films' + film = Movie('051404-000-A', 'Les vieux espions : saluts', 'fr', year=2017, + parts=[Episode('051404-000-A', 'u', 'x', 1, 1, 'x')]) + assert destination(film, film.parts[0], cfg) == Path( + '/media/films/Les vieux espions - saluts (2017)/Les vieux espions - saluts (2017).mkv') + doc = Movie('RC-028069', "L'empire LVMH", 'fr', kind='documentary', total_parts=2, + parts=[Episode('122704-002-A', 'u', 'x', 1, 2, 'Deux')]) + # documentaries_directory unset: movies_directory + assert destination(doc, doc.parts[0], cfg) == Path( + "/media/films/L'empire LVMH/L'empire LVMH - part2.mkv") + cfg.output.documentaries_directory = '/media/docs' + doc.tmdb_id, doc.year = 42, 2026 + cfg.output.movie_dir = '{title} ({year}) [tmdbid-{tmdb_id}]' + assert destination(doc, doc.parts[0], cfg) == Path( + "/media/docs/L'empire LVMH (2026) [tmdbid-42]/L'empire LVMH (2026) - part2.mkv") + tags = _tags(doc, doc.parts[0]) + assert (tags['title'], tags['part_number'], tags['tmdb']) == ("L'empire LVMH (2/2) - Deux", '2', 'movie/42') + + +class FakeTMDB: + def __init__(self, results, details): + self.results, self._details, self.searches = results, details, [] + + def search(self, query, language, kind='tv'): + self.searches.append((query, kind)) + return tuple(r for r in self.results.get(kind, ()) if query.lower() in str(r).lower()) + + def details(self, kind, tmdb_id): + return self._details[(kind, tmdb_id)] + + +GODFATHER_TMDB = {'id': 238, 'title': 'Le Parrain', 'original_title': 'The Godfather', + 'release_date': '1972-03-14', 'original_language': 'en'} + + +def test_find_movie(): + movie = Movie('045559-000-A', 'Le parrain', 'fr', original_title='The Godfather', + original_language='en', year=1972) + client = FakeTMDB({'movie': [GODFATHER_TMDB, {'id': 240, 'title': 'Le Parrain, 2e partie', + 'original_title': 'The Godfather Part II', + 'release_date': '1974-12-20'}]}, {}) + assert find_movie(client, movie, 'fr-FR')[0] == ('movie', 238) + # A documentary is also looked for among TV shows + doc = Movie('RC-1', 'Tchernobyl', 'fr', kind='documentary', year=2026) + client = FakeTMDB({'tv': [{'id': 7, 'name': 'Tchernobyl', 'first_air_date': '2026-01-01'}]}, {}) + assert find_movie(client, doc, 'fr-FR', ('movie', 'tv'))[0] == ('tv', 7) + + +def test_apply_movie(tmp_path, monkeypatch): + monkeypatch.setenv('XDG_DATA_HOME', str(tmp_path)) + details = {**GODFATHER_TMDB, 'external_ids': {'imdb_id': 'tt0068646'}, 'translations': {'translations': [ + {'iso_639_1': 'fr', 'iso_3166_1': 'FR', 'data': {'title': 'Le Parrain', 'overview': 'Corleone'}}]}} + client = FakeTMDB({'movie': [GODFATHER_TMDB]}, {('movie', 238): details}) + monkeypatch.setattr(metadata, 'TMDBClient', lambda key: client) + movie = Movie('045559-000-A', 'Le parrain', 'fr', original_title='The Godfather', year=1972, + description='arte', parts=[Episode('045559-000-A', 'u', 'Le parrain', 1, 1, 'Le parrain')]) + metadata.apply_movie(movie, MetadataConfig(api_key='k'), log=lambda _: None) + assert (movie.title, movie.description, movie.tmdb_id, movie.imdb_id) == ( + 'Le Parrain', 'Corleone', 238, 'tt0068646') + assert movie.parts[0].title == 'Le Parrain' + assert metadata.load_ids() == {'045559-000-A': 'movie/238'} + client.searches.clear() + metadata.apply_movie(movie, MetadataConfig(api_key='k'), log=lambda _: None) + assert not client.searches + + +def test_parse_ref(): + assert parse_ref('123') == ('movie', 123) + assert parse_ref('tv/5') == ('tv', 5) + assert parse_ref(55270, 'tv') == ('tv', 55270) + assert parse_ref(None) is None diff --git a/tests/test_selection.py b/tests/test_selection.py index 779e638..96566f0 100644 --- a/tests/test_selection.py +++ b/tests/test_selection.py @@ -94,3 +94,9 @@ def test_invalid_audio_spec(): cfg.audio.tracks = ['fr-xyz'] with pytest.raises(SelectionError): select(INFO, cfg) + + +def test_max_height_below_every_format_takes_the_lowest(): + cfg = Config() + cfg.video.max_height = 144 + assert select(INFO, cfg).video['format_id'] == 'VF-STF-427'