commit 2f0addb69be5a393776fd7f202a0fb3a1703697f Author: Justin Oro Date: Thu Aug 20 16:31:08 2026 -0700 Initial commit diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..a2e86ae --- /dev/null +++ b/.gitignore @@ -0,0 +1,8 @@ +.env +logs/ +backups/ +.trash/ +cache.json +__pycache__/ +*.pyc +.DS_Store diff --git a/README.md b/README.md new file mode 100644 index 0000000..e8efa31 --- /dev/null +++ b/README.md @@ -0,0 +1,197 @@ +# Plex Library Tool + +A single Python script that scans your Movies / TV Shows folders, looks each one up on [TMDb](https://www.themoviedb.org/), and renames everything into clean, Plex-friendly names: folders, video files, season folders, and subtitles. + +No installation, no dependencies, nothing to compile. It's one `.py` file that runs anywhere Python runs: Mac, Windows, or Linux. + +This guide assumes you've never used Python or GitHub before. If you already know your way around both, skip to [Quick Start](#quick-start). + +--- + +## What it does + +- Matches your existing folder names against TMDb and renames them to `Movie Name (Year)` / `Show Name (Year)` format +- Renames video files to match, and organizes TV episodes into `S01`, `S02`, etc. season folders +- Finds subtitle files, figures out which one matches your primary language (by filename, and by reading the file's content/metadata if the filename doesn't say), and renames it to match the video. Defaults to English, but follows whatever language you've set for TMDb results (see [Non-English users](#non-english-users)). +- Optionally cleans up junk files/folders (samples, `.nfo`, `.txt`, screenshots, unwanted-language subtitles) into a local trash folder. Nothing is deleted permanently, and every cleanup can be reversed. +- Fully customizable naming convention (`names.yaml`): change `S01` to `Season 01`, use dots instead of spaces, uppercase everything, rename the "Subs" folder to something else, etc. +- Every rename is logged, and can be undone with one command +- Works on Mac, Linux, and Windows + +Nothing this script does is destructive by default. It asks for confirmation before renaming anything (unless you pass `-y`), and everything it changes is logged so it can be undone. + +--- + +## Requirements + +- **Python 3.8 or later.** That's it. No other software or packages are required. +- **A free TMDb account**, to get an API key. Takes about 2 minutes, and is covered below. + +--- + +## Quick Start + +### 1. Install Python + +**Windows 11:** + +Open PowerShell (search "PowerShell" in the Start menu) and run: + +``` +winget install Python.Python.3.14 +``` + +Close and reopen PowerShell afterward, then check it worked: + +``` +python --version +``` + +If `winget` doesn't work on your machine, download the installer from [python.org/downloads](https://www.python.org/downloads/) instead. **Important:** on the first screen of the installer, check the box that says **"Add python.exe to PATH"** before clicking Install. This is the single most common thing people miss, and without it Windows won't recognize the `python` command. + +**Mac:** + +Macs usually already have Python 3 installed. Open Terminal (search "Terminal" in Spotlight) and run: + +``` +python3 --version +``` + +If that fails, download the installer from [python.org/downloads](https://www.python.org/downloads/) and run it. + +**Linux:** + +Almost every Linux distribution comes with Python 3 preinstalled. Confirm with: + +``` +python3 --version +``` + +If it's missing, install it with your distro's package manager, e.g. `sudo apt install python3` on Ubuntu/Debian. + +### 2. Download this repository + +If you've never used GitHub before: this page is a "repository" (a folder of files), and you don't need a GitHub account or `git` installed just to download it. + +1. Click the green **"Code"** button near the top of this page +2. Click **"Download ZIP"** +3. Once it's downloaded, extract/unzip it somewhere you'll remember (e.g. your Desktop or Downloads folder) + +(If you do have `git` installed and prefer it, `git clone` this repository's URL instead.) + +### 3. Get a free TMDb API key + +The script uses [TMDb](https://www.themoviedb.org/) (The Movie Database) to look up correct titles and years. This is free. + +1. Log in or create a free account: +2. Request an API key: +3. Choose **"Developer"** when asked what type of key +4. Fill in the short form (you can use placeholder info for personal/non-commercial use) +5. On the resulting settings page, copy the value labeled **"API Key"** (the short one, **not** the longer "API Read Access Token") + +You don't need to do anything with this key yet. The script will ask for it the first time you run it. + +### 4. Run it + +Open a terminal in the folder you extracted: + +- **Windows:** open the extracted folder in File Explorer, click the address bar, type `powershell`, and press Enter +- **Mac/Linux:** open Terminal, then `cd` into the folder, e.g. `cd ~/Downloads/plex-library-tool` + +Then run: + +``` +python plex-library-tool.py -r +``` + +(On Mac/Linux, use `python3` instead of `python` if `python` isn't recognized.) + +The first time you run it, it'll ask you to paste in the TMDb API key from step 3, and offer to save it to a local `.env` file so you're never asked again. That file stays on your computer and is never uploaded anywhere. + +### 5. Point it at your media + +After the API key step, the script will either: +- automatically find SMB/network shares mounted on your computer and let you pick one, or +- you can skip that entirely by giving it a path directly: + +``` +python plex-library-tool.py -r "/path/to/your/Movies" +``` + +``` +python plex-library-tool.py -r "D:\Movies" +``` + +It'll then walk through each folder, look it up, show you the proposed rename, and ask for confirmation before doing anything. + +**Tip:** before renaming anything for real, run it with `-t` (test mode) first to preview what it *would* do without changing anything: + +``` +python plex-library-tool.py -r "/path/to/your/Movies" -t +``` + +--- + +## Command reference + +| Flag | What it does | +|---|---| +| `-r`, `--rename [PATH]` | Scan and rename a share. Pass a path to skip the share-selection prompt. | +| `-c`, `--cleanup [PATH]` | Move junk files/folders (per `delete.yaml`) to a local trash folder. | +| `-t`, `--test [N]` | Preview only. No changes are made. Optionally limit how many folders are shown. | +| `-y`, `--yes` | Don't ask for confirmation before each rename. | +| `-f`, `--force` | Force a full scan even if nothing looks like it changed since the last run. | +| `-v`, `--verbose` | Print detailed diagnostic output about what the script is doing and why. | +| `-m`, `--manual-rename CURRENT NEW` | Manually rename one specific folder/file, bypassing TMDb entirely. | +| `--backup` | Snapshot current names to a log file without changing anything. | +| `--restore [LOGFILE]` | Undo a previous run using its log file. Pick from a list if no file is given. | +| `-u`, `--undo` | Instantly undo the most recent run, no need to look up a log filename. Can't be combined with any other flag. | + +Flags can be combined, e.g. `-rc` runs rename and cleanup back to back on the same share, `-rf` forces a full rename scan, `-ty` previews everything without prompting. + +--- + +## Configuration files + +These live alongside the script and are all optional. The script works out of the box with sensible defaults. Every file is fully documented with comments and examples inside it. + +- **`names.yaml`**: customize the naming convention. Folder/file name format, season folder naming (`S01` vs `Season 01`), separators (spaces vs dots vs underscores), uppercase/lowercase, whether to include resolution tags like `1080p`, and the subtitle folder name (e.g. rename "Subs" to "Subtitles" or any word in your own language). +- **`delete.yaml`**: what cleanup moves to trash. Folder name patterns (e.g. `sample`, `extras`), file name/extension patterns, and subtitle language rules (e.g. "only keep Spanish subtitles"). + +--- + +## Non-English users + +TMDb can return movie/show titles in your own language instead of English (e.g. Spanish, French, German titles). The first time you run the script, it detects your system's language and asks you to confirm before saving it to `.env`. You can also set it yourself at any time by adding a line to `.env`: + +``` +TMDB_LANGUAGE=es-MX +``` + +Use any TMDb-supported language code (ISO 639-1, optionally with a region, e.g. `en-US`, `fr-FR`, `de-DE`, `ja-JP`). This also decides which subtitle language is treated as primary: whichever subtitle file matches `TMDB_LANGUAGE` gets renamed to match the video (e.g. `Movie.es.srt` for Spanish), and everything else keeps its original filename. The script's own prompts and status messages (like "Renamed folder:") stay in English regardless of this setting. + +--- + +## Safety + +- **Nothing is renamed without asking first**, unless you pass `-y`. +- **Every rename is logged** to a `logs/` folder created next to the script. +- **Cleanup never permanently deletes anything.** Matched files/folders are moved to a local `.trash/` folder, not deleted, and you're prompted per item unless `-y` is used. +- **Undo anytime** with `-u` (most recent run) or `--restore ` (any past run). +- **Your TMDb API key is never uploaded anywhere.** It's stored locally in a `.env` file, which is excluded from Git via `.gitignore`. + +--- + +## Troubleshooting + +**`'python' is not recognized as an internal or external command`** (Windows) +Python isn't on your PATH. Reinstall using the winget command above, or rerun the python.org installer and make sure "Add python.exe to PATH" is checked. + +**`No SMB shares found.`** +The script couldn't auto-detect a network share. Pass the path directly instead: `python plex-library-tool.py -r "Z:\Movies"` (Windows) or `python plex-library-tool.py -r "/Volumes/Movies"` (Mac). + +**`No match found (Ignoring): `** +TMDb couldn't confidently match that folder to a title. This is intentionally conservative: the script would rather skip a folder than rename it wrong. You can rename it manually with `-m "current name" "new name"`, or clean up the raw folder name a bit and try again. + +**I want to undo something** +Run `python plex-library-tool.py -u` to instantly undo the most recent run, or `python plex-library-tool.py --restore` to pick from any past run. diff --git a/delete.yaml b/delete.yaml new file mode 100644 index 0000000..796ac2a --- /dev/null +++ b/delete.yaml @@ -0,0 +1,155 @@ +# delete.yaml +# +# Configures what cleanup (-c) moves to a local trash folder. Nothing here +# is deleted permanently -- matched items are moved to a .trash folder next +# to the script, and every move can be undone. All three sections below are +# optional and independent; enable only the ones you want. +# +# ============================================================================ +# folders +# ============================================================================ +# +# Folder names to move to trash during cleanup. +# +# - Matching is case-insensitive. +# - Matches the folder's full name -- e.g. "Screens" matches a folder named +# "Screens" or "screens", but not "My Screens" or "Screenshots". +# - Wildcards are supported: "*" matches any run of characters and "?" +# matches a single character, so you can match partial or variable names. +# For example "* Potato" matches "123 Potato" or "Mr Potato", and +# "Potato*" matches "Potato Salad" or "Potatoes". A pattern with no +# wildcard still has to match the whole folder name exactly. +# - Any folder anywhere in the scanned share with a matching name will be +# moved to trash, along with everything inside it. There is no +# confirmation beyond the normal cleanup run prompts, so double check +# this list before running for real. +# +# Uncomment and edit the list below to enable: +# +# folders: +# - Screens +# - Featurettes +# - Extras +# - Sample +# - Samples +# - Behind the Scenes +# - Deleted Scenes +# - Trailers +# - Interviews +# - Bonus +# +# ============================================================================ +# files +# ============================================================================ +# +# Files to move to trash during cleanup. +# +# - Matching is case-insensitive. +# - Entries can be: +# - a full filename, e.g. "potato.txt" matches only a file named +# exactly "potato.txt" (or "Potato.TXT", etc.) +# - an extension pattern, e.g. "*.txt" matches any file ending in .txt +# - a wildcard pattern, e.g. "* potato.txt" matches "123 potato.txt" or +# "Mr potato.txt"; "?otato.txt" matches "potato.txt" or "Xotato.txt" +# ("*" matches any run of characters, "?" matches a single character) +# - an exclusion pattern, prefixed with "!", e.g. "!*.en.*" protects any +# file matching that pattern from deletion, even if another entry +# above would otherwise match it. A file is only deleted if it +# matches at least one non-"!" entry and matches none of the "!" +# entries. +# - the special exclusion "!language:en" protects English subtitle +# files (.srt, .sub, .idx) from deletion specifically, even when the +# filename has no language tag at all. It's hardcoded to English -- +# the "subtitles:" section below is the friendlier, language-agnostic +# way to protect or target any language, including your own. +# - Video file extensions (mp4, mkv, avi, etc.) are never deleted by +# cleanup, even if a pattern here would otherwise match them. +# +# Uncomment and edit the list below to enable: +# +# files: +# - "*.txt" +# - "*.nfo" +# - "*.jpg" +# - "*.jpeg" +# - "*.png" +# - "*.ico" +# - "*.url" +# - "*.sfv" +# +# ============================================================================ +# subtitles +# ============================================================================ +# +# A separate, language-aware way to manage .srt/.sub/.idx files without +# needing to list those extensions under "files:" above -- this section +# handles them on its own. +# +# - Entries are language names (or common codes), matched case-insensitively: +# "English", "en", "eng" all mean the same thing. Covers ~80 languages +# (see the script's LANGUAGE_CANONICAL table for the full list). +# - Prefix an entry with "!" to KEEP that language and delete every other +# identified language. Use this if you only want one (or a few) +# languages kept -- replace "Spanish" below with whichever language you +# actually want to keep: +# +# subtitles: +# - "!Spanish" +# +# This deletes any subtitle file identified as a language other than +# Spanish, and leaves Spanish, and anything the tool can't confidently +# identify a language for, untouched. +# +# - An entry WITHOUT "!" means "delete just this language, leave every +# other language (including unidentified files) alone." Use this to +# strip out one unwanted language without being aggressive about +# everything else: +# +# subtitles: +# - "English" +# +# This deletes only English-identified subtitle files. Everything else, +# including unidentified files, is left as-is. +# +# - Language is identified from the filename first (e.g. "Movie.es.srt", +# "Movie.Spanish.srt"). If the filename has no language tag at all +# (e.g. a bare "Movie.srt"), the file's own content is read as a +# fallback: .idx files have a real "id: xx" language field, and .srt +# files are analyzed by script (Cyrillic, CJK, Arabic, etc.) and common +# word frequency for Latin-script languages. This is a best-effort +# heuristic for .srt content, not a guarantee, and always errs toward +# leaving a file alone if the result is unclear. +# +# Uncomment and edit to enable (replace "Spanish" with your own primary +# language, or whichever language you want to keep): +# +# subtitles: +# - "!Spanish" + + folders: + - Screens + - Featurettes + - Extras + - Sample + - Samples + - Behind the Scenes + - Deleted Scenes + - Trailers + - Interviews + - Bonus + - "* torrent" + - "* torrents" + +files: + - "*.txt" + - "*.nfo" + - "*.jpg" + - "*.jpeg" + - "*.png" + - "*.ico" + - "*.url" + - "*.sfv" + - "*.part" + +subtitles: + - "!English" diff --git a/names.yaml b/names.yaml new file mode 100644 index 0000000..7115b50 --- /dev/null +++ b/names.yaml @@ -0,0 +1,76 @@ +# names.yaml +# +# Customize the naming convention used when renaming folders and files. +# Uses Python's str.format() syntax: {token} for a plain value, or +# {token:02d} to zero-pad a number to 2 digits, {token:03d} for 3, etc. +# +# Any line you don't include (or leave commented out) falls back to the +# default shown below, so you only need to uncomment the ones you want +# to change. +# +# Available tokens per line: +# movie_folder {title} {year} {resolution} +# movie_folder_no_year {title} {resolution} (used when no year is found) +# show_folder {title} {year} {resolution} +# show_folder_no_year {title} {resolution} (used when no year is found) +# movie_file {title} {ext} {resolution} +# episode_file {title} {season} {episode} {ext} {resolution} +# season_folder {season} +# subtitle_file {stem} {ext} {lang} (stem = matching video's name, no extension; lang = TMDB_LANGUAGE's code, e.g. "en", "es", "fr") +# subtitle_folder (no tokens -- just the folder name, e.g. "Subs", "Subtitles", or a word in your own language) +# +# The subtitle identified as matching TMDB_LANGUAGE (set in .env, see the +# README) is treated as the primary one and renamed to match the video, +# e.g. "Movie.es.srt" if TMDB_LANGUAGE is Spanish. If none of the subtitle +# files can be matched to that language, an untagged one is used as a +# best guess. Everything else keeps its original filename and is just +# moved into the subtitle folder. +# +# {resolution} is opt-in -- it's detected from the original folder/file name +# (720p, 1080p, 2160p, 4K, etc.) but NONE of the default templates above +# reference it, so nothing changes unless you add {resolution} to a +# template yourself, e.g. movie_folder: "{title} ({year}) [{resolution}]". +# If no resolution is found, it renders as an empty string, and an empty +# bracket/paren pair left behind by that (like "[]" or "()") is +# automatically cleaned up. +# +# word_separator controls what replaces spaces INSIDE the {title} value +# itself (the title comes back from TMDb as "Movie Name", space-separated). +# Leave it as a single space for "Movie Name", or set it to "." for +# "Movie.Name", "_" for "Movie_Name", etc. This only affects {title} -- +# combine it with the templates below to fully control the look, e.g. +# word_separator: "." together with movie_folder: "{title}.{year}" gives +# you "Movie.Name.2020" instead of "Movie Name (2020)". +# +# name_case forces the case of every rendered folder/file name (the whole +# name, not just the title). Leave blank for the title case TMDb returns, +# or set to "lower" for "movie name (2020)" / "movie.name.2020.mkv", or +# "upper" for "MOVIE NAME (2020)" / "MOVIE.NAME.2020.MKV". +# +# Filesystem-illegal characters (< > : " / \ | ? *) are stripped from the +# result automatically, so it's safe to use punctuation like colons if you +# want -- it just won't survive into the final name. +# +# Examples: +# episode_file: "{title} - {season}x{episode:02d}.{ext}" -> "Show - 1x05.mkv" +# season_folder: "Season {season:02d}" -> "Season 01" +# movie_folder: "{title} [{year}]" -> "Movie [2020]" +# word_separator: "." -> "Movie.Name" instead of "Movie Name" +# name_case: "lower" -> "movie name (2020)" +# movie_folder: "{title} ({year}) [{resolution}]" -> "Movie (2020) [1080p]" +# subtitle_folder: "Subtitles" -> subs get consolidated into a "Subtitles" folder instead of "Subs" +# subtitle_folder: "Untertitel" -> or any word in your own language +# +# Uncomment and edit any of the lines below to enable: +# +# movie_folder: "{title} ({year})" +# movie_folder_no_year: "{title}" +# show_folder: "{title} ({year})" +# show_folder_no_year: "{title}" +# movie_file: "{title}.{ext}" +# episode_file: "{title} S{season:02d}E{episode:02d}.{ext}" +# season_folder: "S{season:02d}" +# subtitle_file: "{stem}.{lang}.{ext}" +# subtitle_folder: "Subs" +# word_separator: " " +# name_case: "" diff --git a/plex-library-tool.py b/plex-library-tool.py new file mode 100755 index 0000000..085b35a --- /dev/null +++ b/plex-library-tool.py @@ -0,0 +1,2483 @@ +#!/usr/bin/env python3 + +import argparse +import ctypes +import datetime +import fnmatch +import hashlib +import json +import os +import platform +import re +import shutil +import subprocess +import sys +import unicodedata +import urllib.error +import urllib.parse +import urllib.request +from pathlib import Path + +SCRIPT_DIR = Path(__file__).resolve().parent +API_KEY_FILE = SCRIPT_DIR / ".env" +LOG_DIR = SCRIPT_DIR / "logs" +BACKUP_DIR = SCRIPT_DIR / "backups" +CACHE_FILE = SCRIPT_DIR / "cache.json" +TMDB_BASE = "https://api.themoviedb.org/3" +DELETE_FILE = SCRIPT_DIR / "delete.yaml" +NAMES_FILE = SCRIPT_DIR / "names.yaml" +CLEANUP_TRASH_DIR = SCRIPT_DIR / ".trash" + +VERBOSE = False + + +def vprint(*args, **kwargs): + if VERBOSE: + print(*args, **kwargs) + + +VIDEO_EXTENSIONS = { + "mp4", "mkv", "avi", "mov", "wmv", "m4v", "mpg", "mpeg", "flv", "ts", "m2ts", "webm", +} + +SUBTITLE_EXTENSIONS = {"srt", "sub"} +SUBTITLE_FOLDER_NAMES = {"subs", "subtitles"} + +SEASON_EP_PATTERNS = [ + (re.compile(r'[Ss](\d{1,2})[Ee](\d{1,2})'), 'E'), + (re.compile(r'[Ss](\d{1,2})[Xx](\d{1,2})'), 'X'), + (re.compile(r'[Ss](\d{1,2})[Mm](\d{1,2})'), 'M'), + (re.compile(r'(?= 3 and parts[2] == "cifs": + mounts.append(parts[1]) + vprint(f" Found cifs mount: {parts[1]}") + except OSError as e: + vprint(f" Could not read /proc/mounts: {e}") + + try: + gvfs_base = os.environ.get("XDG_RUNTIME_DIR") or f"/run/user/{os.getuid()}" + gvfs_dir = Path(gvfs_base) / "gvfs" + vprint(f" Checking GVFS directory: {gvfs_dir}") + if gvfs_dir.is_dir(): + for entry in gvfs_dir.iterdir(): + if entry.is_dir() and entry.name.startswith("smb-share:"): + mounts.append(str(entry)) + vprint(f" Found GVFS smb-share mount: {entry}") + except OSError as e: + vprint(f" Could not read GVFS directory: {e}") + + elif system == "Windows": + DRIVE_REMOTE = 4 + try: + bitmask = ctypes.windll.kernel32.GetLogicalDrives() + for i in range(26): + if not (bitmask & (1 << i)): + continue + drive = f"{chr(ord('A') + i)}:\\" + drive_type = ctypes.windll.kernel32.GetDriveTypeW(ctypes.c_wchar_p(drive)) + if drive_type == DRIVE_REMOTE: + mounts.append(drive) + vprint(f" Found mapped network drive: {drive}") + except (AttributeError, OSError) as e: + vprint(f" Could not enumerate drives via GetLogicalDrives: {e}") + + try: + out = subprocess.run( + ["net", "use"], capture_output=True, text=True, check=True + ).stdout + for line in out.splitlines(): + m = re.search(r'([A-Za-z]:)\s+(\\\\[^\s]+)', line) + if m: + unc = m.group(2) + if unc not in mounts: + mounts.append(unc) + vprint(f" Found 'net use' mapping: {unc}") + except (subprocess.SubprocessError, OSError) as e: + vprint(f" 'net use' command failed: {e}") + + result = sorted(set(mounts)) + vprint(f"Total SMB mounts found: {len(result)}") + return result + + +def select_mount(mounts): + print("Available SMB shares:") + for i, m in enumerate(mounts, 1): + print(f" {i}) {m}") + while True: + sel = input(f"Select a share to scan [1-{len(mounts)}]: ").strip() + if sel.isdigit() and 1 <= int(sel) <= len(mounts): + return mounts[int(sel) - 1] + print("Invalid selection.") + + +def resolve_share(path_arg=None): + if isinstance(path_arg, str): + vprint(f"Path given on the command line: {path_arg}") + if not Path(path_arg).is_dir(): + print(f"Path not found: {path_arg}") + sys.exit(1) + return path_arg + + mounts = find_smb_mounts() + if not mounts: + print("No SMB shares found.") + sys.exit(1) + return select_mount(mounts) + + +def select_media_type(): + while True: + sel = input("Is this a Movies or TV Shows library? [M/T]: ").strip().lower() + if sel in ("m", "movie", "movies"): + return "movie" + if sel in ("t", "tv", "show", "shows", "tvshows"): + return "tv" + print("Please enter M or T.") + + +MOVIE_SHARE_KEYWORDS = ("movie", "film") +TV_SHARE_KEYWORDS = ("tv", "show", "series") + + +def infer_media_type(share): + name = Path(share).name.lower() + is_movie = any(k in name for k in MOVIE_SHARE_KEYWORDS) + is_tv = any(k in name for k in TV_SHARE_KEYWORDS) + if is_movie and not is_tv: + return "movie" + if is_tv and not is_movie: + return "tv" + return None + + +def determine_media_type(share): + inferred = infer_media_type(share) + if inferred: + label = "Movies" if inferred == "movie" else "TV Shows" + print(f"Detected library type from share name: {label}") + return inferred + return select_media_type() + + +def is_v4_token(api_key): + return api_key.startswith("eyJ") or len(api_key) > 60 + + +def tmdb_search(api_key, media_type, query): + v4 = is_v4_token(api_key) + + params = {"query": query, "language": get_tmdb_language()} + if not v4: + params["api_key"] = api_key + + endpoint = "search/movie" if media_type == "movie" else "search/tv" + url = f"{TMDB_BASE}/{endpoint}?{urllib.parse.urlencode(params)}" + + headers = {"Authorization": f"Bearer {api_key}"} if v4 else {} + req = urllib.request.Request(url, headers=headers) + + try: + with urllib.request.urlopen(req, timeout=10) as resp: + data = json.loads(resp.read().decode("utf-8")) + except (urllib.error.URLError, urllib.error.HTTPError, TimeoutError, OSError) as e: + return None, str(e) + + return data.get("results", []), None + + +def tmdb_alternative_titles(api_key, media_type, tmdb_id): + v4 = is_v4_token(api_key) + + params = {} + if not v4: + params["api_key"] = api_key + + endpoint = f"movie/{tmdb_id}/alternative_titles" if media_type == "movie" else f"tv/{tmdb_id}/alternative_titles" + url = f"{TMDB_BASE}/{endpoint}" + if params: + url += f"?{urllib.parse.urlencode(params)}" + + headers = {"Authorization": f"Bearer {api_key}"} if v4 else {} + req = urllib.request.Request(url, headers=headers) + + try: + with urllib.request.urlopen(req, timeout=10) as resp: + data = json.loads(resp.read().decode("utf-8")) + except (urllib.error.URLError, urllib.error.HTTPError, TimeoutError, OSError): + return [] + + key = "titles" if media_type == "movie" else "results" + entries = data.get(key, []) + return [e.get("title") or e.get("name") for e in entries if e.get("title") or e.get("name")] + + +def result_title(media_type, result): + return result.get("title") if media_type == "movie" else result.get("name") + + +def result_year(media_type, result): + date_field = result.get("release_date") if media_type == "movie" else result.get("first_air_date") + if date_field and len(date_field) >= 4: + return date_field[:4] + return None + + +def best_match(media_type, results, year, query=None): + if not results: + return None + + query_norm = clean_and_squeeze(query).lower() if query else None + + def is_exact_title(r): + return query_norm is not None and clean_and_squeeze(result_title(media_type, r) or "").lower() == query_norm + + if query_norm and year: + for r in results: + if is_exact_title(r) and result_year(media_type, r) == year: + return r + + if year: + for r in results: + if result_year(media_type, r) == year: + return r + return None + + if query_norm: + for r in results: + if is_exact_title(r): + return r + + return results[0] + + +def same_existing_path(dest, src): + try: + return dest.exists() and src.exists() and dest.samefile(src) + except OSError: + return False + + +def confirm(prompt): + answer = input(f"{prompt} [y/N] ").strip().lower() + return answer in ("y", "yes") + + +def confirm_delete_choice(prompt): + while True: + answer = input(f"{prompt} [y/N/a] ").strip().lower() + if answer in ("y", "yes"): + return "y" + if answer in ("a", "all"): + return "a" + if answer in ("n", "no", ""): + return "n" + print("Please enter y (yes), n (no), or a (all).") + + +JUNK_TOKENS = { + "bluray", "blueray", "bdrip", "brrip", "bdremux", "remux", + "webrip", "webdl", "web", "dl", "hdtv", "hdrip", "dvdrip", "dvd", + "hevc", "x264", "x265", "h264", "h265", "avc", + "aac", "ac3", "eac3", "dts", "atmos", "ddp", "dd", + "proper", "repack", "extended", "unrated", "uncut", "directors", "cut", + "internal", "limited", "theatrical", "multi", "dual", "audio", + "hdr", "sdr", "4k", "uhd", "10bit", "8bit", "hi10p", "hi444pp", + "deluxe", "boxset", "box", "set", "extras", "hd", + "complete", "collection", "series", + "se", "ce", "remastered", "anniversary", "edition", + "amzn", "nf", "dsnp", "hmax", "atvp", "pcok", "hulu", +} + +SEASON_RANGE_PATTERN = re.compile(r'\bSeasons?\b(?:\s+\d{1,2})+', re.IGNORECASE) +IN_FORMAT_PATTERN = re.compile(r'\bin\s+(?:full\s+)?(?:hd|4k|uhd|sd)\b', re.IGNORECASE) + + +def strip_junk_tokens(text): + text = re.sub(r'\b\d{3,4}p\b', ' ', text, flags=re.IGNORECASE) + text = SEASON_RANGE_PATTERN.sub(' ', text) + text = IN_FORMAT_PATTERN.sub(' ', text) + words = text.split() + kept = [w for w in words if w.lower() not in JUNK_TOKENS] + return squeeze_spaces(' '.join(kept)) + + +SEASON_TRUNCATE_PATTERN = re.compile(r'\bSeasons?\b\s+\d', re.IGNORECASE) +BARE_SEASON_TRUNCATE_PATTERN = re.compile(r'(? 0: + raw_name = raw_name[:cut] + + raw_name = AUDIO_CHANNELS_PATTERN.sub(' ', raw_name) + raw_name = strip_junk_trailing_paren(raw_name) + + match = RELEASE_GROUP_SUFFIX_PATTERN.search(raw_name) + if match: + raw_name = raw_name[:match.start()] + match.group(1) + + cleaned = clean_and_squeeze(raw_name) + if year: + without_year = squeeze_spaces(re.sub(rf'(?:"/\\|?*\x00-\x1f]') +EMPTY_GROUP_PATTERN = re.compile(r'[\(\[\{]\s*[\)\]\}]') + +RESOLUTION_PATTERN = re.compile(r'\b(4320p|2160p|1080p|720p|480p|360p|4K|8K|UHD)\b', re.IGNORECASE) +RESOLUTION_CANONICAL = { + "4320p": "4320p", "2160p": "2160p", "1080p": "1080p", "720p": "720p", + "480p": "480p", "360p": "360p", "4k": "4K", "8k": "8K", "uhd": "UHD", +} + +_NAME_TEMPLATES = None + + +def detect_resolution(name): + match = RESOLUTION_PATTERN.search(name) + if not match: + return None + return RESOLUTION_CANONICAL.get(match.group(1).lower()) + + +def sanitize_rendered_name(name): + name = FILESYSTEM_ILLEGAL_CHARS_PATTERN.sub('', name) + name = EMPTY_GROUP_PATTERN.sub('', name) + name = squeeze_spaces(name) + name = name.strip().rstrip('. ') + case = get_name_templates().get("name_case", "") + if case == "lower": + name = name.lower() + elif case == "upper": + name = name.upper() + return name + + +def load_name_templates(): + templates = dict(DEFAULT_NAME_TEMPLATES) + if not NAMES_FILE.exists(): + return templates + try: + lines = NAMES_FILE.read_text().splitlines() + except OSError as e: + vprint(f"Could not read {NAMES_FILE.name}: {e}") + return templates + + for line in lines: + stripped = line.strip() + if not stripped or stripped.startswith("#") or ":" not in stripped: + continue + key, _, value = stripped.partition(":") + key = key.strip() + value = value.strip() + if key not in DEFAULT_NAME_TEMPLATES: + continue + quoted = len(value) >= 2 and value[0] == value[-1] and value[0] in ('"', "'") + if quoted: + value = value[1:-1] + elif "#" in value: + value = value.split("#", 1)[0].strip() + if value or quoted: + templates[key] = value + vprint(f" Loaded naming template {key!r}: {value!r}") + + return templates + + +def get_name_templates(): + global _NAME_TEMPLATES + if _NAME_TEMPLATES is None: + _NAME_TEMPLATES = load_name_templates() + return _NAME_TEMPLATES + + +def render_name(template, fallback, **context): + try: + rendered = template.format(**context) + except (KeyError, ValueError, IndexError) as e: + vprint(f" Bad naming template {template!r}: {e}, falling back to default") + rendered = fallback.format(**context) + return sanitize_rendered_name(rendered) + + +def apply_word_separator(title): + templates = get_name_templates() + separator = templates.get("word_separator", " ") + if separator == " ": + return title + return title.replace(" ", separator) + + +def movie_folder_name(title, year, resolution=None): + templates = get_name_templates() + title = apply_word_separator(title) + resolution = resolution or "" + if year: + return render_name(templates["movie_folder"], DEFAULT_NAME_TEMPLATES["movie_folder"], title=title, year=int(year), resolution=resolution) + return render_name(templates["movie_folder_no_year"], DEFAULT_NAME_TEMPLATES["movie_folder_no_year"], title=title, resolution=resolution) + + +def show_folder_name(title, year, resolution=None): + templates = get_name_templates() + title = apply_word_separator(title) + resolution = resolution or "" + if year: + return render_name(templates["show_folder"], DEFAULT_NAME_TEMPLATES["show_folder"], title=title, year=int(year), resolution=resolution) + return render_name(templates["show_folder_no_year"], DEFAULT_NAME_TEMPLATES["show_folder_no_year"], title=title, resolution=resolution) + + +def movie_file_name(title, ext, resolution=None): + templates = get_name_templates() + title = apply_word_separator(title) + resolution = resolution or "" + return render_name(templates["movie_file"], DEFAULT_NAME_TEMPLATES["movie_file"], title=title, ext=ext, resolution=resolution) + + +def episode_file_name(title, season, episode, ext, resolution=None): + templates = get_name_templates() + title = apply_word_separator(title) + resolution = resolution or "" + return render_name( + templates["episode_file"], DEFAULT_NAME_TEMPLATES["episode_file"], + title=title, season=season, episode=episode, ext=ext, resolution=resolution, + ) + + +def season_folder_name(season): + templates = get_name_templates() + return render_name(templates["season_folder"], DEFAULT_NAME_TEMPLATES["season_folder"], season=season) + + +def subtitle_file_name(stem, ext, lang): + templates = get_name_templates() + return render_name(templates["subtitle_file"], DEFAULT_NAME_TEMPLATES["subtitle_file"], stem=stem, ext=ext, lang=lang) + + +def subtitle_folder_name(): + templates = get_name_templates() + return render_name(templates["subtitle_folder"], DEFAULT_NAME_TEMPLATES["subtitle_folder"]) + + +def folder_target_name(media_type, final_name, match_year, raw_name=None): + resolution = detect_resolution(raw_name) if raw_name else None + if media_type == "movie": + return movie_folder_name(final_name, match_year, resolution) + return show_folder_name(final_name, match_year, resolution) + + +def infer_year_from_files(folder): + for path in sorted(folder.rglob("*")): + if path.is_file() and path.suffix.lower().lstrip(".") in VIDEO_EXTENSIONS: + year = extract_year(path.stem) + if year: + return year + return None + + +def lookup_folder(api_key, media_type, raw_name, hint_year=None): + year = extract_year(raw_name) or hint_year + query = build_query(raw_name, year) + vprint(f"Looking up: {raw_name!r} -> query={query!r}, year={year!r}, media_type={media_type!r}") + + results, err = tmdb_search(api_key, media_type, query) + if err: + return None, None, f"Lookup failed: {raw_name} ({err})" + vprint(f" TMDb returned {len(results)} result(s)") + + match = best_match(media_type, results, year, query) + if not match: + return None, None, f"No match found (Ignoring): {raw_name}" + + query_norm = clean_and_squeeze(query).lower() + title = result_title(media_type, match) + final_name = clean_and_squeeze(title) + vprint(f" Best match: {title!r} (id={match.get('id')})") + + if final_name.lower() != query_norm: + vprint(" Title differs from query, checking alternative titles...") + for candidate in results[:ALT_TITLE_CHECK_LIMIT]: + alt_titles = tmdb_alternative_titles(api_key, media_type, candidate.get("id")) + hit = next((a for a in alt_titles if clean_and_squeeze(a).lower() == query_norm), None) + if hit: + match = candidate + final_name = clean_and_squeeze(hit) + vprint(f" Alternative title match: {hit!r} (id={candidate.get('id')})") + break + + match_year = result_year(media_type, match) + vprint(f" Resolved: {final_name!r} ({match_year})") + + return final_name, match_year, None + + +def list_video_files(folder): + return [ + item for item in sorted(folder.iterdir()) + if item.is_file() and item.suffix.lower().lstrip(".") in VIDEO_EXTENSIONS + ] + + +def stem_match_pattern(stem): + words = [w for w in re.split(r'[^A-Za-z0-9]+', stem) if w] + if not words: + return None + body = r'[\W_]*'.join(re.escape(w) for w in words) + return re.compile(r'^' + body + r'(?=[\W_])', re.IGNORECASE) + + +ENGLISH_LANGUAGE_TOKENS = {"en", "eng", "english"} + +OTHER_LANGUAGE_TOKENS_RELIABLE = { + "spanish", "fre", "fra", "french", + "ger", "deu", "german", "ita", "italian", + "portuguese", "dut", "nld", "dutch", + "jpn", "japanese", "zho", "chinese", "cmn", "mandarin", + "kor", "korean", "rus", "russian", "ara", "arabic", + "swe", "swedish", "norwegian", "danish", + "finnish", "polish", "tur", "turkish", + "heb", "hebrew", "hin", "hindi", "tha", "thai", + "cze", "ces", "czech", "ell", "greek", + "hun", "hungarian", "romanian", + "vie", "vietnamese", "ind", "indonesian", "ukr", "ukrainian", + "malayalam", "tam", "tamil", "tel", "telugu", "kan", "kannada", + "bengali", "panjabi", "punjabi", "marathi", + "guj", "gujarati", "urd", "urdu", "odia", "oriya", + "asm", "assamese", "sanskrit", "nep", "nepali", "sinhala", + "fas", "persian", "farsi", "pashto", + "kur", "kurdish", "aze", "azerbaijani", "kaz", "kazakh", + "uzb", "uzbek", "tgk", "tajik", "tuk", "turkmen", + "tgl", "fil", "filipino", "tagalog", "msa", "malay", + "khm", "khmer", "lao", "laotian", "mya", "bur", "burmese", "myanmar", + "mongolian", "swa", "swahili", "amh", "amharic", + "hau", "hausa", "yor", "yoruba", "ibo", "igbo", "zul", "zulu", + "xho", "xhosa", "sna", "shona", "somali", + "sqi", "alb", "albanian", "hye", "armenian", + "eus", "baq", "basque", "belarusian", "bul", "bulgarian", + "catalan", "hrv", "croatian", "estonian", + "glg", "galician", "kat", "georgian", "isl", "icelandic", + "gle", "irish", "lav", "latvian", "lithuanian", + "ltz", "luxembourgish", "mkd", "macedonian", "mlt", "maltese", + "slk", "slo", "slovak", "slv", "slovenian", + "cym", "wel", "welsh", "yid", "yiddish", "haitian", + "srp", "serbian", "bos", "bosnian", +} + +OTHER_LANGUAGE_TOKENS_RESTRICTED = { + "es", "fr", "de", "it", "pt", "nl", "ja", "zh", "ko", "ru", "ar", + "sv", "no", "da", "fi", "pl", "tr", "he", "hi", "th", "cs", "el", + "hu", "ro", "vi", "id", "uk", + "spa", "por", "chi", "dan", "pol", "fin", "gre", + "mal", "san", "sin", "per", "mon", "pan", "mar", "pus", "ron", + "may", "mac", "arm", "geo", "bel", "ice", "lit", "rum", +} + +OTHER_LANGUAGE_TOKENS = OTHER_LANGUAGE_TOKENS_RELIABLE | OTHER_LANGUAGE_TOKENS_RESTRICTED + + +LANGUAGE_CANONICAL = { + "en": {"en", "eng", "english"}, + "es": {"es", "spa", "spanish"}, + "fr": {"fr", "fre", "fra", "french"}, + "de": {"de", "ger", "deu", "german"}, + "it": {"it", "ita", "italian"}, + "pt": {"pt", "por", "portuguese"}, + "nl": {"nl", "dut", "nld", "dutch"}, + "ja": {"ja", "jpn", "japanese"}, + "zh": {"zh", "chi", "zho", "chinese", "cmn", "mandarin"}, + "ko": {"ko", "kor", "korean"}, + "ru": {"ru", "rus", "russian"}, + "ar": {"ar", "ara", "arabic"}, + "sv": {"sv", "swe", "swedish"}, + "no": {"no", "nor", "norwegian"}, + "da": {"da", "dan", "danish"}, + "fi": {"fi", "fin", "finnish"}, + "pl": {"pl", "pol", "polish"}, + "tr": {"tr", "tur", "turkish"}, + "he": {"he", "heb", "hebrew"}, + "hi": {"hi", "hin", "hindi"}, + "th": {"th", "tha", "thai"}, + "cs": {"cs", "cze", "ces", "czech"}, + "el": {"el", "gre", "ell", "greek"}, + "hu": {"hu", "hun", "hungarian"}, + "ro": {"ro", "ron", "rum", "romanian"}, + "vi": {"vi", "vie", "vietnamese"}, + "id": {"id", "ind", "indonesian"}, + "uk": {"uk", "ukr", "ukrainian"}, + "ml": {"mal", "malayalam"}, + "ta": {"tam", "tamil"}, + "te": {"tel", "telugu"}, + "kn": {"kan", "kannada"}, + "bn": {"ben", "bengali"}, + "pa": {"pan", "panjabi", "punjabi"}, + "mr": {"mar", "marathi"}, + "gu": {"guj", "gujarati"}, + "ur": {"urd", "urdu"}, + "or": {"ori", "odia", "oriya"}, + "as": {"asm", "assamese"}, + "sa": {"san", "sanskrit"}, + "ne": {"nep", "nepali"}, + "si": {"sin", "sinhala"}, + "fa": {"per", "fas", "persian", "farsi"}, + "ps": {"pus", "pashto"}, + "ku": {"kur", "kurdish"}, + "az": {"aze", "azerbaijani"}, + "kk": {"kaz", "kazakh"}, + "uz": {"uzb", "uzbek"}, + "tg": {"tgk", "tajik"}, + "tk": {"tuk", "turkmen"}, + "tl": {"tgl", "fil", "filipino", "tagalog"}, + "ms": {"may", "msa", "malay"}, + "km": {"khm", "khmer"}, + "lo": {"lao", "laotian"}, + "my": {"mya", "bur", "burmese", "myanmar"}, + "mn": {"mon", "mongolian"}, + "sw": {"swa", "swahili"}, + "am": {"amh", "amharic"}, + "ha": {"hau", "hausa"}, + "yo": {"yor", "yoruba"}, + "ig": {"ibo", "igbo"}, + "zu": {"zul", "zulu"}, + "xh": {"xho", "xhosa"}, + "sn": {"sna", "shona"}, + "so": {"som", "somali"}, + "sq": {"sqi", "alb", "albanian"}, + "hy": {"hye", "arm", "armenian"}, + "eu": {"eus", "baq", "basque"}, + "be": {"bel", "belarusian"}, + "bg": {"bul", "bulgarian"}, + "ca": {"cat", "catalan"}, + "hr": {"hrv", "croatian"}, + "et": {"est", "estonian"}, + "gl": {"glg", "galician"}, + "ka": {"kat", "geo", "georgian"}, + "is": {"isl", "ice", "icelandic"}, + "ga": {"gle", "irish"}, + "lv": {"lav", "latvian"}, + "lt": {"lit", "lithuanian"}, + "lb": {"ltz", "luxembourgish"}, + "mk": {"mkd", "mac", "macedonian"}, + "mt": {"mlt", "maltese"}, + "sk": {"slk", "slo", "slovak"}, + "sl": {"slv", "slovenian"}, + "cy": {"cym", "wel", "welsh"}, + "yi": {"yid", "yiddish"}, + "ht": {"hat", "haitian"}, + "sr": {"srp", "serbian"}, + "bs": {"bos", "bosnian"}, +} + +LANGUAGE_ALIAS_TO_CODE = { + alias: code for code, aliases in LANGUAGE_CANONICAL.items() for alias in aliases +} + + +SUBTITLE_DESCRIPTOR_WORDS = {"forced", "sdh", "cc", "full", "commentary"} + + +def language_tag_from_name(name): + stem = name.rsplit(".", 1)[0] if "." in name else name + segments = [s.lower() for s in re.split(r'[._\-\s]+', stem) if s] + if len(segments) < 2: + return None + + trimmed = list(segments) + while len(trimmed) > 1 and trimmed[-1] in SUBTITLE_DESCRIPTOR_WORDS: + trimmed.pop() + + tag_window = set(segments[-2:]) + last_only = {trimmed[-1]} if trimmed else set() + + has_en = bool(tag_window & ENGLISH_LANGUAGE_TOKENS) + has_foreign = bool(tag_window & OTHER_LANGUAGE_TOKENS_RELIABLE) or bool(last_only & OTHER_LANGUAGE_TOKENS_RESTRICTED) + + if has_en and not has_foreign: + return "en" + if has_foreign and not has_en: + return "foreign" + return None + + +def specific_language_from_name(name): + stem = name.rsplit(".", 1)[0] if "." in name else name + segments = [s.lower() for s in re.split(r'[._\-\s]+', stem) if s] + if len(segments) < 2: + return None + + trimmed = list(segments) + while len(trimmed) > 1 and trimmed[-1] in SUBTITLE_DESCRIPTOR_WORDS: + trimmed.pop() + + tag_window = segments[-2:] + last_only = trimmed[-1] if trimmed else None + + for seg in tag_window: + if seg in ENGLISH_LANGUAGE_TOKENS: + return "en" + if seg in OTHER_LANGUAGE_TOKENS_RELIABLE: + return LANGUAGE_ALIAS_TO_CODE.get(seg) + + if last_only and last_only in OTHER_LANGUAGE_TOKENS_RESTRICTED: + return LANGUAGE_ALIAS_TO_CODE.get(last_only) + + return None + + +def resolve_subtitle_language(path): + hint = specific_language_from_name(path.name) + if hint: + return hint + detected = detect_subtitle_language(path) + if detected and detected != "unknown": + return LANGUAGE_ALIAS_TO_CODE.get(detected, detected) + return None + + +def gather_video_subtitles(video_path, multi): + folder = video_path.parent + pattern = stem_match_pattern(video_path.stem) if multi else None + matches = [] + + search_dirs = [folder] + try: + for entry in sorted(folder.iterdir()): + if entry.is_dir() and entry.name.lower() in SUBTITLE_FOLDER_NAMES: + search_dirs.append(entry) + except OSError: + pass + + for search_dir in search_dirs: + try: + entries = sorted(search_dir.iterdir()) + except OSError: + continue + for item in entries: + if not item.is_file() or item == video_path: + continue + ext = item.suffix.lower().lstrip(".") + if ext not in SUBTITLE_EXTENSIONS: + continue + if multi and (pattern is None or not pattern.match(item.name)): + continue + matches.append(item) + + return matches + + +def compute_consolidated_subtitle_pairs(subtitles, new_video_path): + new_stem = new_video_path.stem + target_dir = new_video_path.parent / subtitle_folder_name() + primary_lang = get_tmdb_language().split("-")[0].lower() + + by_ext = {} + for item in subtitles: + by_ext.setdefault(item.suffix.lower(), []).append(item) + + canonical = {} + for ext, items in by_ext.items(): + detected = {i: resolve_subtitle_language(i) for i in items} + chosen = next((i for i in items if detected[i] == primary_lang), None) + if chosen is None: + chosen = next((i for i in items if detected[i] is None), None) + if chosen: + canonical[chosen] = subtitle_file_name(new_stem, ext.lstrip("."), primary_lang) + + pairs = [] + for item in subtitles: + dest_name = canonical.get(item, item.name) + dest = target_dir / dest_name + pairs.append((item, dest)) + + if item.suffix.lower() == ".sub": + idx_item = item.with_suffix(".idx") + if idx_item.exists(): + if item in canonical: + idx_dest = target_dir / (canonical[item].rsplit(".", 1)[0] + ".idx") + else: + idx_dest = target_dir / idx_item.name + pairs.append((idx_item, idx_dest)) + + return pairs + + +def rename_consolidated_subtitles(subtitles, new_video_path, log): + renamed = 0 + source_dirs = set() + for item, dest in compute_consolidated_subtitle_pairs(subtitles, new_video_path): + if dest == item: + continue + if dest.exists() and not same_existing_path(dest, item): + print(f"Skipping subtitle (target already exists): {item.name}") + continue + dest.parent.mkdir(exist_ok=True) + source_dir = item.parent + item.rename(dest) + log.record(item, dest) + label = "subtitle index" if dest.suffix.lower() == ".idx" else "subtitle" + if source_dir == dest.parent: + print(f"Renamed {label}: {item.name} -> {dest.name}") + else: + print(f"Moved {label}: {item.name} -> {dest.parent.name}/{dest.name}") + renamed += 1 + if source_dir.name.lower() in SUBTITLE_FOLDER_NAMES: + source_dirs.add(source_dir) + + for source_dir in source_dirs: + try: + if source_dir.exists() and not any(source_dir.iterdir()): + source_dir.rmdir() + print(f"Removed empty folder: {source_dir}") + except OSError: + pass + + return renamed + + +def preview_consolidated_subtitles(subtitles, new_video_path): + for item, dest in compute_consolidated_subtitle_pairs(subtitles, new_video_path): + if dest == item: + continue + label = "subtitle index" if dest.suffix.lower() == ".idx" else "subtitle" + if item.parent == dest.parent: + print(f"Renamed {label}: {item.name} -> {dest.name}") + else: + print(f"Moved {label}: {item.name} -> {dest.parent.name}/{dest.name}") + + +def organize_subtitle_folder(folder, log): + try: + entries = sorted(folder.iterdir()) + except OSError: + return + + canonical_name = subtitle_folder_name() + for entry in entries: + if not entry.is_dir() or entry.name == canonical_name: + continue + if entry.name.lower() not in SUBTITLE_FOLDER_NAMES: + continue + + dest = folder / canonical_name + if dest.exists() and not same_existing_path(dest, entry): + print(f"Skipping (target already exists): {entry.name} -> {canonical_name}") + continue + + entry.rename(dest) + log.record(entry, dest) + print(f"Renamed folder: {entry.name} -> {canonical_name}") + + +def preview_subtitle_folder(folder): + try: + entries = sorted(folder.iterdir()) + except OSError: + return + + canonical_name = subtitle_folder_name() + for entry in entries: + if not entry.is_dir() or entry.name == canonical_name: + continue + if entry.name.lower() not in SUBTITLE_FOLDER_NAMES: + continue + print(f"Renamed folder: {entry.name} -> {canonical_name}") + + +def list_subfolders(folder): + return sorted(p for p in folder.iterdir() if p.is_dir() and not p.name.startswith(".")) + + +def preview_season_folders(folder): + for sub in list_subfolders(folder): + season = parse_season_folder_name(sub.name) + if season is None: + continue + new_name = season_folder_name(season) + if new_name != sub.name: + print(f"Renamed folder: {sub.name} -> {new_name}") + + +def organize_season_folders(folder, log): + renamed = 0 + for sub in list_subfolders(folder): + season = parse_season_folder_name(sub.name) + if season is None: + continue + new_name = season_folder_name(season) + if new_name == sub.name: + continue + dest = folder / new_name + if dest.exists() and not same_existing_path(dest, sub): + print(f"Skipping (target already exists): {sub.name} -> {new_name}") + continue + sub.rename(dest) + log.record(sub, dest) + print(f"Renamed folder: {sub.name} -> {new_name}") + renamed += 1 + return renamed + + +def resolve_season_folder_file(item, season): + se = parse_season_episode(item.name) + if se: + return se[0], se[1], se[2] + return parse_season_only(item.name), parse_episode_only(item.name), 'E' + + +def preview_season_folder_files(folder, final_name): + for sub in list_subfolders(folder): + season = parse_season_folder_name(sub.name) + if season is None: + continue + + preview_subtitle_folder(sub) + + for item in list_video_files(sub): + ext = item.suffix.lower().lstrip(".") + file_season, episode, marker = resolve_season_folder_file(item, season) + target_season = file_season if file_season is not None else season + subtitles = gather_video_subtitles(item, True) + + if episode is None: + if target_season != season: + target_season_folder = season_folder_name(target_season) + dest = folder / target_season_folder / item.name + print(f"Move: {item.name} -> {target_season_folder}/{item.name}") + preview_consolidated_subtitles(subtitles, dest) + else: + print(f"No episode found, already in {season_folder_name(season)}: {item.name}") + continue + + new_name = episode_file_name(final_name, target_season, episode, ext, detect_resolution(item.name)) + if target_season != season: + target_season_folder = season_folder_name(target_season) + dest = folder / target_season_folder / new_name + print(f"Move: {item.name} -> {target_season_folder}/{new_name}") + preview_consolidated_subtitles(subtitles, dest) + elif new_name != item.name: + dest = sub / new_name + print(f"Renamed file: {item.name} -> {new_name}") + preview_consolidated_subtitles(subtitles, dest) + else: + preview_consolidated_subtitles(subtitles, item) + + +def rename_season_folder_files(folder, final_name, log): + renamed = 0 + skipped = 0 + + for sub in list_subfolders(folder): + season = parse_season_folder_name(sub.name) + if season is None: + continue + + organize_subtitle_folder(sub, log) + + for item in list_video_files(sub): + ext = item.suffix.lower().lstrip(".") + file_season, episode, marker = resolve_season_folder_file(item, season) + target_season = file_season if file_season is not None else season + target_season_folder = season_folder_name(target_season) + target_dir = folder / target_season_folder if target_season != season else sub + subtitles = gather_video_subtitles(item, True) + + if episode is None: + if target_season == season: + print(f"No episode found, already in {season_folder_name(season)}: {item.name}") + skipped += 1 + continue + new_name = item.name + else: + new_name = episode_file_name(final_name, target_season, episode, ext, detect_resolution(item.name)) + + dest = target_dir / new_name + if dest == item: + renamed += rename_consolidated_subtitles(subtitles, item, log) + continue + if dest.exists() and not same_existing_path(dest, item): + print(f"Skipping (target already exists): {item.name}") + skipped += 1 + continue + + if target_dir != sub: + target_dir.mkdir(exist_ok=True) + item.rename(dest) + log.record(item, dest) + if target_dir != sub: + print(f"Moved: {item.name} -> {target_season_folder}/{new_name}") + else: + print(f"Renamed file: {item.name} -> {new_name}") + renamed += 1 + renamed += rename_consolidated_subtitles(subtitles, dest, log) + + return renamed, skipped + + +def preview_video_files(folder, media_type, final_name, api_key): + files = list_video_files(folder) + + if media_type == "movie": + preview_subtitle_folder(folder) + multi = len(files) > 1 + for item in files: + ext = item.suffix.lower().lstrip(".") + name = final_name + if multi: + per_name, _, error = lookup_folder(api_key, media_type, item.stem) + if error: + print(error) + continue + name = per_name + new_name = movie_file_name(name, ext, detect_resolution(item.name)) + dest = folder / new_name + subtitles = gather_video_subtitles(item, multi) + if new_name != item.name: + print(f"Renamed file: {item.name} -> {new_name}") + preview_consolidated_subtitles(subtitles, dest) + else: + preview_consolidated_subtitles(subtitles, item) + return + + for item in files: + ext = item.suffix.lower().lstrip(".") + se = parse_season_episode(item.name) + subtitles = gather_video_subtitles(item, True) + if se: + season, episode, marker = se + new_name = episode_file_name(final_name, season, episode, ext, detect_resolution(item.name)) + target_season_folder = season_folder_name(season) + print(f"Move: {item.name} -> {target_season_folder}/{new_name}") + preview_consolidated_subtitles(subtitles, folder / target_season_folder / new_name) + continue + + season = parse_season_only(item.name) + if season is None: + print(f"No season/episode found, skipping: {item.name}") + continue + + target_season_folder = season_folder_name(season) + print(f"Move: {item.name} -> {target_season_folder}/{item.name}") + preview_consolidated_subtitles(subtitles, folder / target_season_folder / item.name) + + +def rename_video_files(folder, media_type, final_name, log, api_key): + renamed = 0 + skipped = 0 + + files = list_video_files(folder) + + if media_type == "movie": + organize_subtitle_folder(folder, log) + multi = len(files) > 1 + for item in files: + ext = item.suffix.lower().lstrip(".") + name = final_name + if multi: + per_name, _, error = lookup_folder(api_key, media_type, item.stem) + if error: + print(error) + skipped += 1 + continue + name = per_name + new_name = movie_file_name(name, ext, detect_resolution(item.name)) + dest = folder / new_name + subtitles = gather_video_subtitles(item, multi) + if dest == item: + renamed += rename_consolidated_subtitles(subtitles, item, log) + continue + if dest.exists() and not same_existing_path(dest, item): + print(f"Skipping (target already exists): {item.name}") + skipped += 1 + continue + item.rename(dest) + log.record(item, dest) + print(f"Renamed file: {item.name} -> {new_name}") + renamed += 1 + renamed += rename_consolidated_subtitles(subtitles, dest, log) + return renamed, skipped + + for item in files: + ext = item.suffix.lower().lstrip(".") + se = parse_season_episode(item.name) + subtitles = gather_video_subtitles(item, True) + + if se: + season, episode, marker = se + new_name = episode_file_name(final_name, season, episode, ext, detect_resolution(item.name)) + else: + season = parse_season_only(item.name) + if season is None: + print(f"No season/episode found, skipping: {item.name}") + skipped += 1 + continue + new_name = item.name + + target_season_folder = season_folder_name(season) + season_dir = folder / target_season_folder + dest = season_dir / new_name + + if dest == item: + renamed += rename_consolidated_subtitles(subtitles, item, log) + continue + if dest.exists() and not same_existing_path(dest, item): + print(f"Skipping (target already exists): {item.name}") + skipped += 1 + continue + + season_dir.mkdir(exist_ok=True) + item.rename(dest) + log.record(item, dest) + print(f"Moved: {item.name} -> {target_season_folder}/{new_name}") + renamed += 1 + renamed += rename_consolidated_subtitles(subtitles, dest, log) + + return renamed, skipped + + +def compute_folder_signature(folder): + entries = [] + for root, dirs, files in os.walk(folder): + rel_root = os.path.relpath(root, folder) + for name in dirs + files: + entries.append(os.path.normpath(os.path.join(rel_root, name))) + + hasher = hashlib.sha256() + for entry in sorted(entries): + hasher.update(entry.encode("utf-8")) + hasher.update(b"\n") + return hasher.hexdigest() + + +def load_scan_cache(): + if CACHE_FILE.exists(): + try: + return json.loads(CACHE_FILE.read_text()) + except (json.JSONDecodeError, OSError): + return {} + return {} + + +def save_scan_cache(cache): + CACHE_FILE.write_text(json.dumps(cache, indent=2)) + + +def parse_simple_yaml_list(path, key): + if not path.exists(): + return [] + + items = [] + in_list = False + for raw_line in path.read_text().splitlines(): + line = raw_line.strip() + if not line or line.startswith("#"): + continue + if line.startswith(f"{key}:"): + in_list = True + continue + if in_list and line.startswith("-"): + value = line[1:].strip().strip('"').strip("'") + if value: + items.append(value) + else: + in_list = False + + return items + + +def load_delete_folder_names(): + return [name.lower() for name in parse_simple_yaml_list(DELETE_FILE, "folders")] + + +TRAILING_WRAPPER_CHARS = ")]}'\"" + + +def split_patterns(patterns): + include = [] + exclude = [] + for pattern in patterns: + if pattern.startswith("!"): + exclude.append(pattern[1:]) + else: + include.append(pattern) + return include, exclude + + +def match_with_wrapper_stripping(name_lower, patterns): + if any(fnmatch.fnmatch(name_lower, pattern) for pattern in patterns): + return True + stripped = name_lower.rstrip(TRAILING_WRAPPER_CHARS) + if stripped and stripped != name_lower: + return any(fnmatch.fnmatch(stripped, pattern) for pattern in patterns) + return False + + +def matches_folder_name(name, patterns): + include, exclude = split_patterns(patterns) + if not match_with_wrapper_stripping(name.lower(), include): + return False + if exclude and match_with_wrapper_stripping(name.lower(), exclude): + return False + return True + + +def load_delete_file_patterns(): + return [p.lower() for p in parse_simple_yaml_list(DELETE_FILE, "files")] + + +def load_delete_subtitle_rules(): + keep_languages = set() + delete_languages = set() + for entry in parse_simple_yaml_list(DELETE_FILE, "subtitles"): + is_keep = entry.startswith("!") + name = entry[1:] if is_keep else entry + code = LANGUAGE_ALIAS_TO_CODE.get(name.strip().lower()) + if not code: + continue + if is_keep: + keep_languages.add(code) + else: + delete_languages.add(code) + return keep_languages, delete_languages + + +def subtitle_language_cleanup_target(item, keep_languages, delete_languages): + if not keep_languages and not delete_languages: + return False + ext = item.suffix.lower().lstrip(".") + if ext not in SUBTITLE_EXTENSIONS and ext != "idx": + return False + lang = resolve_subtitle_language(item) + if keep_languages: + if lang is None: + return False + return lang not in keep_languages + if lang is None: + return False + return lang in delete_languages + + +def parse_idx_language(path): + try: + text = path.read_text(errors="ignore") + except OSError: + return None + match = re.search(r'^\s*id:\s*([a-zA-Z]{2,3})', text, re.MULTILINE) + if match: + return match.group(1).lower() + return None + + +def count_unicode_range(text, lo, hi): + return sum(1 for ch in text if lo <= ord(ch) <= hi) + + +SUBTITLE_SCRIPT_RANGES = { + "ru": (0x0400, 0x04FF), + "el": (0x0370, 0x03FF), + "he": (0x0590, 0x05FF), + "ar": (0x0600, 0x06FF), + "th": (0x0E00, 0x0E7F), + "hi": (0x0900, 0x097F), + "ko": (0xAC00, 0xD7A3), + "ja": (0x3040, 0x30FF), + "zh": (0x4E00, 0x9FFF), +} + +SUBTITLE_STOPWORDS = { + "en": {"the", "and", "is", "of", "in", "to", "a", "that", "it", "you", "was", + "for", "on", "are", "with", "as", "this", "have", "be", "not", "but", + "he", "she", "they", "we", "his", "her", "at", "from", "by", "or", + "an", "if", "what", "so", "all", "can", "just", "one", "like", "get", + "know", "will", "would", "there", "when", "who", "how", "out", "up", + "about", "then", "them", "were", "been", "had", "do", "did", "yes", "no"}, + "fr": {"le", "la", "les", "de", "et", "un", "une", "des", "est", "que", "qui", + "pas", "pour", "dans", "ce", "il", "elle", "vous", "je", "nous", "au", + "du", "se", "ne", "tu", "on", "avec", "sur", "son", "sa", "ses", + "mais", "comme", "tout", "ça", "oui", "non"}, + "es": {"el", "la", "los", "las", "de", "y", "un", "una", "es", "que", "no", + "por", "con", "para", "en", "se", "su", "lo", "como", "más", "pero", + "le", "les", "yo", "tu", "este", "esta", "ese", "esa", "sí"}, + "de": {"der", "die", "das", "und", "ist", "nicht", "ein", "eine", "zu", "den", + "mit", "auf", "für", "sich", "du", "ich", "er", "sie", "wir", "es", + "war", "sind", "aber", "was", "wie", "wenn", "ja", "nein"}, + "it": {"il", "la", "di", "e", "un", "una", "che", "non", "per", "con", "del", + "della", "sono", "questo", "questa", "ma", "come", "io", "tu", "lui", + "lei", "noi", "sì", "no"}, + "pt": {"o", "a", "os", "as", "de", "e", "um", "uma", "que", "não", "por", + "com", "para", "em", "se", "seu", "sua", "como", "mas", "eu", "tu", + "ele", "ela", "nós", "sim"}, + "nl": {"de", "het", "een", "en", "is", "niet", "van", "dat", "je", "ik", + "hij", "zij", "wij", "met", "voor", "op", "aan", "maar", "zoals", + "ja", "nee"}, +} + + +def detect_srt_language(path): + try: + raw = path.read_bytes() + except OSError: + return "unknown" + + text = None + for encoding in ("utf-8-sig", "utf-8", "cp1252", "latin-1"): + try: + text = raw.decode(encoding) + break + except (UnicodeDecodeError, LookupError): + continue + if text is None: + return "unknown" + + lines = [] + for line in text.splitlines(): + line = line.strip() + if not line or line.isdigit() or "-->" in line: + continue + lines.append(line) + sample = " ".join(lines)[:20000] + if not sample.strip(): + return "unknown" + + for lang, (lo, hi) in SUBTITLE_SCRIPT_RANGES.items(): + if count_unicode_range(sample, lo, hi) >= 5: + vprint(f" {path.name}: detected non-Latin script -> {lang}") + return lang + + words = re.findall(r"[a-zà-öø-ÿ]+", sample.lower()) + total = len(words) + if total < 20: + vprint(f" {path.name}: only {total} word(s) sampled, too little to classify -> unknown") + return "unknown" + + scores = {lang: sum(1 for w in words if w in stopwords) for lang, stopwords in SUBTITLE_STOPWORDS.items()} + best_lang = max(scores, key=scores.get) + best_score = scores[best_lang] + en_score = scores.get("en", 0) + vprint(f" {path.name}: word count={total}, scores={scores}") + + if best_lang == "en": + vprint(f" {path.name}: classified as en") + return "en" + if best_score >= 5 and best_score >= en_score * 1.5 and best_score >= en_score + 3: + vprint(f" {path.name}: classified as {best_lang} (en_score={en_score}, {best_lang}_score={best_score})") + return best_lang + vprint(f" {path.name}: inconclusive (en_score={en_score}, best={best_lang}:{best_score}) -> unknown") + return "unknown" + + +def detect_subtitle_language(path): + ext = path.suffix.lower() + if ext == ".idx": + return parse_idx_language(path) or "unknown" + if ext == ".sub": + idx_path = path.with_suffix(".idx") + if idx_path.exists(): + lang = parse_idx_language(idx_path) + if lang: + return lang + return "unknown" + if ext == ".srt": + return detect_srt_language(path) + return "unknown" + + +LANGUAGE_EXCLUDE_TOKEN = "language:en" + + +def subtitle_language_hint_from_name(name): + return language_tag_from_name(name) + + +def matches_file_name(name, patterns, path=None): + include, exclude = split_patterns(patterns) + if not match_with_wrapper_stripping(name.lower(), include): + return False + + literal_exclude = [p for p in exclude if p != LANGUAGE_EXCLUDE_TOKEN] + if literal_exclude and match_with_wrapper_stripping(name.lower(), literal_exclude): + return False + + if LANGUAGE_EXCLUDE_TOKEN in exclude and path is not None: + hint = subtitle_language_hint_from_name(name) + if hint == "en": + return False + if hint is None: + detected = detect_subtitle_language(path) + if detected in ("en", "unknown"): + return False + + return True + + +def find_cleanup_targets(share, delete_folder_names, delete_file_patterns, keep_languages=None, delete_languages=None): + targets = [] + keep_languages = keep_languages or set() + delete_languages = delete_languages or set() + + def walk(folder): + vprint(f"Scanning folder: {folder}") + try: + entries = sorted(folder.iterdir()) + except OSError as e: + vprint(f" Could not read folder: {e}") + return + for entry in entries: + if entry.is_dir(): + if matches_folder_name(entry.name, delete_folder_names): + vprint(f" Match (folder name): {entry}") + targets.append(("folder", entry)) + continue + vprint(f" No match, descending into: {entry}") + walk(entry) + elif entry.is_file(): + ext = entry.suffix.lower().lstrip(".") + if ext in VIDEO_EXTENSIONS: + vprint(f" Skipping (protected video file): {entry}") + continue + if matches_file_name(entry.name, delete_file_patterns, entry): + vprint(f" Match (file pattern): {entry}") + targets.append(("file", entry)) + elif subtitle_language_cleanup_target(entry, keep_languages, delete_languages): + vprint(f" Match (subtitle language): {entry}") + targets.append(("file", entry)) + else: + vprint(f" No match: {entry}") + + walk(Path(share)) + return targets + + +def trash_path_for(share, item, timestamp): + rel = item.relative_to(share) + share_label = Path(share).name + return CLEANUP_TRASH_DIR / timestamp / share_label / rel + + +def remove_empty_folders(share, confirm_all, dry_run=False): + removed = 0 + skipped = 0 + removed_paths = set() + share_path = Path(share).resolve() + for root, dirs, files in os.walk(share, topdown=False): + root_path = Path(root) + if root_path.resolve() == share_path: + continue + try: + remaining = [c for c in root_path.iterdir() if c not in removed_paths] + except OSError as e: + vprint(f" Could not read {root_path}: {e}") + continue + if remaining: + continue + + if dry_run: + print(f"Would remove empty folder: {root_path}") + removed_paths.add(root_path) + removed += 1 + continue + + if not confirm_all: + choice = confirm_delete_choice(f"Remove empty folder: {root_path}?") + if choice == "a": + confirm_all = True + elif choice == "n": + print(f"Skipped: {root_path}") + skipped += 1 + continue + + try: + root_path.rmdir() + print(f"Removed empty folder: {root_path}") + removed_paths.add(root_path) + removed += 1 + except OSError as e: + vprint(f" Could not remove {root_path}: {e}") + return removed, skipped + + +def run_cleanup(args, log): + path_arg = args.cleanup if isinstance(args.cleanup, str) else None + share = resolve_share(path_arg) + + delete_folder_names = load_delete_folder_names() + delete_file_patterns = load_delete_file_patterns() + keep_languages, delete_languages = load_delete_subtitle_rules() + vprint(f"Loaded {len(delete_folder_names)} folder pattern(s) from {DELETE_FILE.name}: {delete_folder_names}") + vprint(f"Loaded {len(delete_file_patterns)} file pattern(s) from {DELETE_FILE.name}: {delete_file_patterns}") + vprint(f"Loaded subtitle rules: keep={keep_languages or 'none'}, delete={delete_languages or 'none'}") + + if not delete_folder_names and not delete_file_patterns and not keep_languages and not delete_languages: + print("No cleanup rules configured.") + print(f"Edit {DELETE_FILE.name} to enable cleanup.") + return + + print() + print(f"Scanning: {share}") + print() + + targets = find_cleanup_targets(share, delete_folder_names, delete_file_patterns, keep_languages, delete_languages) + vprint(f"Scan complete. {len(targets)} item(s) matched.") + + test_mode = args.test is not None + test_limit = args.test or 0 + + if test_mode: + shown = 0 + for kind, item in targets: + if test_limit > 0 and shown >= test_limit: + break + print(f"Would move to trash ({kind}): {item}") + shown += 1 + empty_preview, _ = remove_empty_folders(share, True, dry_run=True) + print() + if not targets and not empty_preview: + print("Nothing to clean up.") + else: + print( + f"Test mode: {len(targets)} item(s) would be moved to trash, " + f"{empty_preview} empty folder(s) would be removed. No changes were made." + ) + return + + if targets: + print(f"Found {len(targets)} item(s) to move to trash:") + for kind, item in targets: + print(f" [{kind}] {item}") + print() + + timestamp = datetime.datetime.now().strftime("%Y%m%d-%H%M%S") + moved_folders = 0 + moved_files = 0 + skipped_folders = 0 + skipped_files = 0 + confirm_all = args.yes + + for kind, item in targets: + if not item.exists(): + if kind == "folder": + skipped_folders += 1 + else: + skipped_files += 1 + continue + + if not confirm_all: + choice = confirm_delete_choice(f"Move to trash ({kind}): {item}?") + if choice == "a": + confirm_all = True + elif choice == "n": + print(f"Skipped: {item}") + if kind == "folder": + skipped_folders += 1 + else: + skipped_files += 1 + continue + + dest = trash_path_for(share, item, timestamp) + if dest.exists(): + print(f"Skipping (trash target already exists): {item}") + if kind == "folder": + skipped_folders += 1 + else: + skipped_files += 1 + continue + + dest.parent.mkdir(parents=True, exist_ok=True) + vprint(f"Moving {item} -> {dest}") + shutil.move(str(item), str(dest)) + log.record(item, dest) + print(f"Moved to trash: {item}") + if kind == "folder": + moved_folders += 1 + else: + moved_files += 1 + + empty_removed, empty_skipped = remove_empty_folders(share, confirm_all) + + folders_deleted = moved_folders + empty_removed + folders_deleted_skipped = skipped_folders + empty_skipped + + if not targets and not folders_deleted and not moved_files and not folders_deleted_skipped and not skipped_files: + print("Nothing to clean up.") + return + + print() + print("=== Summary ===") + print(f"Folders deleted: {folders_deleted}, skipped: {folders_deleted_skipped}") + print(f"Files deleted: {moved_files}, skipped: {skipped_files}") + + log_path = log.save(label=f"{Path(share).name}-cleanup") + if log_path: + print(f"Log saved: {log_path}") + print(f"Trash location: {CLEANUP_TRASH_DIR / timestamp}") + print(f"Run --restore {log_path.name} to undo, or delete the trash folder once you're confident.") + + +def list_loose_video_files(share): + return [ + item for item in sorted(Path(share).iterdir()) + if item.is_file() and item.suffix.lower().lstrip(".") in VIDEO_EXTENSIONS + ] + + +def process_loose_movie_files(share, api_key, log, test_mode, test_limit, args): + files = list_loose_video_files(share) + if not files: + return 0, 0, 0, 0 + + print(f"Found {len(files)} loose movie file(s) directly in {Path(share).name} with no folder of their own:") + print() + + folders_renamed = 0 + folders_skipped = 0 + files_renamed = 0 + files_skipped = 0 + shown = 0 + total = len(files) + + for index, item in enumerate(files, start=1): + if test_mode and test_limit > 0 and shown >= test_limit: + break + + print(f"[{index}/{total}] {item.name}") + final_name, match_year, error = lookup_folder(api_key, "movie", item.stem) + if error: + print(error) + shown += 1 + folders_skipped += 1 + continue + + folder_name = folder_target_name("movie", final_name, match_year, item.stem) + target_folder = Path(share) / folder_name + ext = item.suffix.lower().lstrip(".") + new_name = movie_file_name(final_name, ext, detect_resolution(item.name)) + dest = target_folder / new_name + subtitles = gather_video_subtitles(item, True) + + if test_mode: + print(f"Move: {item.name} -> {folder_name}/{new_name}") + preview_consolidated_subtitles(subtitles, dest) + shown += 1 + folders_renamed += 1 + files_renamed += 1 + continue + + if not args.yes and not confirm(f"Create '{folder_name}' and move '{item.name}' into it?"): + print(f"Skipped: {item.name}") + folders_skipped += 1 + continue + + if dest.exists() and not same_existing_path(dest, item): + print(f"Skipping (target already exists): {item.name}") + folders_skipped += 1 + continue + + target_folder.mkdir(exist_ok=True) + item.rename(dest) + log.record(item, dest) + print(f"Moved: {item.name} -> {folder_name}/{new_name}") + folders_renamed += 1 + files_renamed += 1 + files_renamed += rename_consolidated_subtitles(subtitles, dest, log) + + print() + return folders_renamed, folders_skipped, files_renamed, files_skipped + + +def run_scan(args, log): + path_arg = args.rename if isinstance(args.rename, str) else None + share = resolve_share(path_arg) + + cache = load_scan_cache() + share_key = str(Path(share).resolve()) + baseline = cache.get(share_key) + if not isinstance(baseline, dict): + baseline = {} + vprint(f"Baseline has {len(baseline)} folder(s) recorded") + + media_type = determine_media_type(share) + + print() + print(f"Scanning: {share}") + print() + + api_key = get_api_key() + get_tmdb_language() + + test_mode = args.test is not None + test_limit = args.test or 0 + + subfolders = sorted( + p for p in Path(share).iterdir() if p.is_dir() and not p.name.startswith(".") + ) + loose_files = list_loose_video_files(share) if media_type == "movie" else [] + + unchanged_count = 0 + if not args.force: + to_process = [] + for folder in subfolders: + signature = compute_folder_signature(folder) + if baseline.get(folder.name) == signature: + vprint(f"Unchanged, skipping: {folder.name}") + unchanged_count += 1 + continue + to_process.append(folder) + + if not to_process and not loose_files: + print(f"No changes detected since last scan ({unchanged_count} folder(s) unchanged). Use --force to force a full scan.") + return + + subfolders = to_process + if unchanged_count: + print(f"Skipping {unchanged_count} unchanged folder(s) ({len(subfolders)} to process). Use --force to force a full scan.") + print() + + examples_shown = 0 + folders_renamed = 0 + folders_skipped = 0 + files_renamed = 0 + files_skipped = 0 + + total_folders = len(subfolders) + + for index, folder in enumerate(subfolders, start=1): + if test_mode and test_limit > 0 and examples_shown >= test_limit: + break + + raw_name = folder.name + print(f"[{index}/{total_folders}] {raw_name}") + vprint(f"Processing folder: {folder}") + hint_year = None + if extract_year(raw_name) is None: + hint_year = infer_year_from_files(folder) + vprint(f" No year in folder name, inferred from files: {hint_year}") + final_name, match_year, error = lookup_folder(api_key, media_type, raw_name, hint_year) + + if error: + print(error) + examples_shown += 1 + folders_skipped += 1 + continue + + folder_name = folder_target_name(media_type, final_name, match_year, raw_name) + + if test_mode: + new_folder = folder.parent / folder_name + needs_rename = new_folder != folder + if needs_rename: + print(f"Renamed folder: {raw_name} -> {folder_name}") + else: + print(f"Skipping renaming (already correctly named): {raw_name}") + if media_type == "tv": + preview_season_folders(folder) + preview_season_folder_files(folder, final_name) + preview_video_files(folder, media_type, final_name, api_key) + examples_shown += 1 + continue + + new_folder = folder.parent / folder_name + needs_rename = new_folder != folder + + if needs_rename: + if not args.yes and not confirm(f"Rename '{raw_name}' -> '{folder_name}'?"): + print(f"Skipped: {raw_name}") + folders_skipped += 1 + continue + + if new_folder.exists() and not same_existing_path(new_folder, folder): + print(f"Skipping (target already exists): {raw_name} -> {folder_name}") + folders_skipped += 1 + continue + + folder.rename(new_folder) + log.record(folder, new_folder) + print(f"Renamed folder: {raw_name} -> {folder_name}") + folders_renamed += 1 + else: + print(f"Skipping renaming (already correctly named): {raw_name}") + folder = new_folder + + if media_type == "tv": + organize_season_folders(folder, log) + sub_renamed, sub_skipped = rename_season_folder_files(folder, final_name, log) + files_renamed += sub_renamed + files_skipped += sub_skipped + + renamed, skipped = rename_video_files(folder, media_type, final_name, log, api_key) + files_renamed += renamed + files_skipped += skipped + + baseline[folder.name] = compute_folder_signature(folder) + + if loose_files: + loose_folders_renamed, loose_folders_skipped, loose_files_renamed, loose_files_skipped = process_loose_movie_files( + share, api_key, log, test_mode, test_limit, args + ) + folders_renamed += loose_folders_renamed + folders_skipped += loose_folders_skipped + files_renamed += loose_files_renamed + files_skipped += loose_files_skipped + + print() + if test_mode: + print("Test mode: no changes were made.") + return + + print("=== Summary ===") + print(f"Folders renamed: {folders_renamed}, skipped: {folders_skipped}") + print(f"Files renamed: {files_renamed}, skipped: {files_skipped}") + + log_path = log.save(label=Path(share).name) + if log_path: + print(f"Log saved: {log_path}") + + current_names = { + p.name for p in Path(share).iterdir() if p.is_dir() and not p.name.startswith(".") + } + baseline = {name: sig for name, sig in baseline.items() if name in current_names} + cache[share_key] = baseline + save_scan_cache(cache) + + +def run_backup(args): + mounts = find_smb_mounts() + if not mounts: + print("No SMB shares found.") + sys.exit(1) + + share = select_mount(mounts) + + print() + print(f"Scanning: {share}") + print() + + paths = [] + for root, dirs, files in os.walk(share): + rel_root = os.path.relpath(root, share) + for name in sorted(dirs): + rel = name if rel_root == "." else os.path.join(rel_root, name) + rel = os.path.normpath(rel) + "/" + paths.append(rel) + vprint(f" Captured folder: {rel}") + for name in sorted(files): + rel = name if rel_root == "." else os.path.join(rel_root, name) + rel = os.path.normpath(rel) + paths.append(rel) + vprint(f" Captured file: {rel}") + + paths.sort() + + BACKUP_DIR.mkdir(exist_ok=True) + timestamp = datetime.datetime.now().strftime("%Y%m%d-%H%M%S") + label = re.sub(r'[^A-Za-z0-9]+', '_', Path(share).name).strip('_') + backup_path = BACKUP_DIR / f"{timestamp}-{label}-backup.json" + counter = 1 + while backup_path.exists(): + vprint(f" Backup filename already exists, trying next: {backup_path}") + backup_path = BACKUP_DIR / f"{timestamp}-{label}-backup-{counter}.json" + counter += 1 + + backup_path.write_text(json.dumps({"share": share, "timestamp": timestamp, "paths": paths}, indent=2)) + + print(f"Captured {len(paths)} folder(s)/file(s).") + print(f"Backup saved: {backup_path}") + + +def run_manual_rename(current_name, new_name, log): + src = Path(current_name) + dst = Path(new_name) + + if not src.exists(): + print(f"Path not found: {current_name}") + sys.exit(1) + if dst.exists() and not same_existing_path(dst, src): + print(f"Target already exists: {new_name}") + sys.exit(1) + + src.rename(dst) + log.record(src, dst) + print(f"Renamed: {current_name} -> {new_name}") + + log_path = log.save(label="manual") + if log_path: + print(f"Log saved: {log_path}") + + +def format_log_timestamp(name): + m = re.match(r'^(\d{8}-\d{6})', name) + if not m: + return None + try: + dt = datetime.datetime.strptime(m.group(1), "%Y%m%d-%H%M%S") + except ValueError: + return None + hour = dt.strftime("%I").lstrip("0") or "12" + return dt.strftime(f"%Y-%m-%d {hour}:%M %p") + + +def select_log_file(): + logs = sorted(LOG_DIR.glob("*.json"), key=lambda p: p.name, reverse=True) if LOG_DIR.exists() else [] + if not logs: + print("No logs found.") + sys.exit(1) + + print("Restore from log:") + for i, log_path in enumerate(logs, 1): + friendly = format_log_timestamp(log_path.name) + suffix = f" ({friendly})" if friendly else "" + print(f" {i}) {log_path.name}{suffix}") + while True: + sel = input(f"Select a log to restore [1-{len(logs)}]: ").strip() + if sel.isdigit() and 1 <= int(sel) <= len(logs): + return logs[int(sel) - 1] + print("Invalid selection.") + + +def find_latest_log(): + logs = sorted(LOG_DIR.glob("*.json"), key=lambda p: p.name, reverse=True) if LOG_DIR.exists() else [] + return logs[0] if logs else None + + +def run_undo(): + latest = find_latest_log() + if not latest: + print("No logs found.") + sys.exit(1) + friendly = format_log_timestamp(latest.name) + suffix = f" ({friendly})" if friendly else "" + print(f"Undoing last action: {latest.name}{suffix}") + print() + run_restore(str(latest)) + + +def run_restore(log_arg): + if isinstance(log_arg, str): + path = Path(log_arg) + if not path.is_absolute() and not path.exists(): + candidate = LOG_DIR / log_arg + if candidate.exists(): + path = candidate + + if not path.exists(): + print(f"Log file not found: {log_arg}") + sys.exit(1) + else: + path = select_log_file() + + vprint(f"Restoring from log: {path}") + entries = json.loads(path.read_text()) + vprint(f"Log contains {len(entries)} entry(ies), processing in reverse order") + + restored = 0 + skipped = 0 + + for entry in reversed(entries): + src = Path(entry["to"]) + dst = Path(entry["from"]) + + if not src.exists(): + print(f"Skipping (no longer exists): {src}") + skipped += 1 + continue + if dst.exists(): + print(f"Skipping (restore target already exists): {dst}") + skipped += 1 + continue + + dst.parent.mkdir(parents=True, exist_ok=True) + shutil.move(str(src), str(dst)) + print(f"Restored: {src} -> {dst}") + restored += 1 + + parent = src.parent + try: + if parent.exists() and not any(parent.iterdir()): + parent.rmdir() + print(f"Removed empty folder: {parent}") + except OSError: + pass + + print() + print(f"Restored: {restored}, skipped: {skipped}") + + +def build_parser(): + parser = argparse.ArgumentParser( + description="Scan SMB media shares, match against TMDb, and rename folders/files into Plex-friendly structure." + ) + parser.add_argument("-y", "--yes", action="store_true", help="Rename everything without prompting") + parser.add_argument( + "-f", "--force", + action="store_true", + help="Scan even if no changes were detected since the last run." + ) + parser.add_argument( + "-v", "--verbose", + action="store_true", + help="Print detailed information about everything the script is doing." + ) + parser.add_argument( + "-t", "--test", + nargs="?", const=0, type=int, default=None, metavar="N", + help="Preview matches and renames without making changes. Optional N limits how many folders are previewed." + ) + parser.add_argument( + "-r", "--rename", + nargs="?", const=True, default=False, metavar="PATH", + help="Scan a share, match against TMDb, and rename folders/files into Plex-friendly structure. " + "Optionally pass a path to skip the share-selection prompt." + ) + parser.add_argument( + "-m", "--manual-rename", + nargs=2, metavar=("CURRENT_NAME", "NEW_NAME"), + help="Manually rename a specific folder or file, bypassing TMDb lookup." + ) + parser.add_argument( + "--restore", + nargs="?", const=True, default=False, metavar="LOGFILE", + help="Restore original names from a saved rename log (filename or path). " + "If no logfile is given, choose from a list of available logs (most recent first)." + ) + parser.add_argument( + "--backup", + action="store_true", + help="Snapshot the current folder/file names on a share to a log file, without making any changes." + ) + parser.add_argument( + "-c", "--cleanup", + nargs="?", const=True, default=False, metavar="PATH", + help="Move junk folders/files (per delete.yaml) to a trash folder. " + "Optionally pass a path to skip the share-selection prompt." + ) + parser.add_argument( + "-u", "--undo", + action="store_true", + help="Restore from the most recent log file, without prompting. Cannot be combined with any other argument." + ) + return parser + + +BUNDLABLE_FLAGS = {"y", "f", "t", "r", "v", "c"} +PATH_TAKING_FLAGS = {"r", "c"} + + +def expand_bundled_flags(argv): + expanded = [] + for token in argv: + if ( + len(token) > 2 + and token[0] == "-" + and token[1] != "-" + and all(c in BUNDLABLE_FLAGS for c in token[1:]) + ): + letters = token[1:] + path_flag = next((c for c in letters if c in PATH_TAKING_FLAGS), None) + if path_flag: + letters = letters.replace(path_flag, "") + path_flag + expanded.extend(f"-{c}" for c in letters) + else: + expanded.append(token) + return expanded + + +def parse_args(): + return build_parser().parse_args(expand_bundled_flags(sys.argv[1:])) + + +def main(): + if len(sys.argv) == 1: + build_parser().print_help() + return + + args = parse_args() + + if args.undo: + other_args_used = ( + args.yes or args.force or args.verbose or args.test is not None + or args.rename or args.manual_rename or args.restore + or args.backup or args.cleanup + ) + if other_args_used: + print("--undo cannot be combined with any other argument.") + sys.exit(1) + run_undo() + return + + global VERBOSE + VERBOSE = args.verbose + + log = RenameLog() + + if args.restore: + run_restore(args.restore) + return + + if args.backup: + run_backup(args) + return + + if args.manual_rename: + run_manual_rename(args.manual_rename[0], args.manual_rename[1], log) + return + + if args.cleanup and args.rename: + path_arg = None + if isinstance(args.rename, str): + path_arg = args.rename + elif isinstance(args.cleanup, str): + path_arg = args.cleanup + share = resolve_share(path_arg) + args.rename = share + args.cleanup = share + run_scan(args, log) + print() + run_cleanup(args, log) + return + + if args.cleanup: + run_cleanup(args, log) + return + + if args.rename: + run_scan(args, log) + return + + build_parser().print_help() + + +if __name__ == "__main__": + try: + main() + except KeyboardInterrupt: + print() + print("Interrupted. Exiting.") + sys.exit(130)