diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..d738c53 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,36 @@ +name: CI + +on: + push: + branches: ["main", "copilot/**"] + pull_request: + branches: ["main"] + +jobs: + lint-and-test: + name: Lint & Test (Python ${{ matrix.python-version }}) + runs-on: ubuntu-latest + permissions: + contents: read + strategy: + matrix: + python-version: ["3.11", "3.12"] + + steps: + - uses: actions/checkout@v4 + + - name: Set up Python ${{ matrix.python-version }} + uses: actions/setup-python@v5 + with: + python-version: ${{ matrix.python-version }} + + - name: Install dependencies + run: | + python -m pip install --upgrade pip + pip install -e ".[dev]" + + - name: Lint with ruff + run: ruff check . + + - name: Run tests + run: pytest -v diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..464ff21 --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,22 @@ +# Changelog + +All notable changes to this project will be documented in this file. + +The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), +and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). + +## [1.0.0] - 2024-01-01 + +### Added + +- Initial release of `jellyfin_cleanup` CLI tool. +- Async scraping of the full Jellyfin library with configurable page size and concurrency. +- SQLite cache so subsequent runs can skip re-scraping. +- Bulk deletion via `DELETE /Items?ids=…` with per-item fallback on 404. +- Exponential backoff with jitter for retries on transient errors. +- `--dry-run` mode to preview matches without deleting. +- `--force-rescrape` and `--no-rescrape` flags to control cache behaviour non-interactively. +- `--yes` flag to skip the delete confirmation prompt. +- Support for multiple target path prefixes (positional or `--target-path`). +- `jellyfin-cleanup` console script entry point. +- GitHub Actions CI workflow (lint with ruff + pytest on Python 3.11 and 3.12). diff --git a/README.md b/README.md index cba8d28..6ffed05 100644 --- a/README.md +++ b/README.md @@ -1,2 +1,109 @@ # jellyfinCleaner -A py file to delete bad jellyfin library entries + +A command-line tool to find and delete Jellyfin library items by path prefix — useful when you have moved, renamed, or removed drives and need to clean up stale entries that Jellyfin still tracks. + +## Features + +- **Async scraping** — fetches your entire Jellyfin library in parallel pages and caches results in a local SQLite database. +- **SQLite cache** — avoids re-scraping on every run; prompts you to re-use cached data or refresh it. +- **Bulk deletion** — sends batched `DELETE /Items?ids=…` requests with configurable concurrency and batch size; falls back to per-item deletion on 404 responses. +- **Retry with backoff** — exponential backoff with jitter on transient errors (429 / 5xx / timeouts). +- **Dry-run mode** — preview what would be deleted without touching anything. +- **Resumable** — items that failed to delete are marked `failed` in the DB and will be retried automatically on the next run. + +## Requirements + +- Python ≥ 3.11 +- A Jellyfin server with API access + +## Installation + +```bash +# From source (recommended) +pip install . + +# Or in editable mode for development +pip install -e ".[dev]" +``` + +This installs the `jellyfin-cleanup` command. + +## Quick Start + +```bash +# Set your API key once (or pass --api-key on every run) +export JELLYFIN_API_KEY="your_api_key_here" + +# Preview items under a path (dry-run, no changes made) +jellyfin-cleanup --dry-run /mnt/old-drive/movies + +# Delete items (will prompt for confirmation) +jellyfin-cleanup /mnt/old-drive/movies + +# Delete items from multiple paths without confirmation prompts +jellyfin-cleanup --yes /mnt/old-drive/movies /mnt/old-drive/shows + +# Run against a remote Jellyfin instance +jellyfin-cleanup --url http://jellyfin.home:8096 --api-key abc123 /mnt/old-drive +``` + +## Usage + +``` +usage: jellyfin_cleanup [-h] [--target-path PATH] [--url URL] [--api-key KEY] + [--db FILE] [--page-size N] [--fetch-concurrency N] + [--delete-concurrency N] [--delete-batch-size N] + [--max-retries N] [--retry-backoff-base SECS] + [--retry-backoff-max SECS] [--timeout-connect SECS] + [--timeout-read SECS] [--timeout-write SECS] + [--timeout-pool SECS] [--force-rescrape] [--no-rescrape] + [--yes] [--dry-run] [--verbose] + [PATH ...] +``` + +### Key options + +| Option | Default | Description | +|---|---|---| +| `PATH …` (positional) | — | One or more path prefixes to target | +| `-t`, `--target-path PATH` | — | Path prefix (repeatable, merged with positional) | +| `-u`, `--url URL` | `http://127.0.0.1:8096` | Jellyfin base URL | +| `-k`, `--api-key KEY` | `JELLYFIN_API_KEY` env | Jellyfin API key | +| `--db FILE` | `jellyfin_cleanup.db` | SQLite cache file | +| `--page-size N` | `500` | Items per fetch page | +| `--fetch-concurrency N` | `3` | Parallel page-fetch requests | +| `--delete-concurrency N` | `5` | Parallel bulk-delete requests | +| `--delete-batch-size N` | `50` | Items per bulk-delete API call | +| `--max-retries N` | `5` | Max retries per request | +| `--force-rescrape` | `False` | Re-scrape even if cache exists | +| `--no-rescrape` | `False` | Always use cached data | +| `--yes`, `-y` | `False` | Skip delete confirmation prompt | +| `--dry-run` | `False` | Preview without deleting | +| `--verbose`, `-v` | `False` | Enable DEBUG logging | + +## Development + +```bash +# Install with dev extras +pip install -e ".[dev]" + +# Run tests +pytest -v + +# Lint +ruff check . +``` + +## How It Works + +1. **Connectivity check** — verifies the server is reachable and the API key is valid. +2. **Scrape** — pages through `GET /Items?Recursive=true&Fields=Path` and stores every item in a local SQLite database with its `delete_status = 'pending'`. +3. **Target matching** — queries the DB for items whose `path` starts with any of the specified prefixes and whose `delete_status` is `pending` or `failed`. +4. **Preview** — prints a grouped summary of matching items. +5. **Delete** — sends concurrent batched `DELETE /Items?ids=…` requests; records each outcome (`deleted`, `not_found`, or `failed`) back to the DB. +6. **Summary** — prints final DB statistics and warns if any items remain `failed`. + +## License + +GPL-3.0 — see [LICENSE](LICENSE). + diff --git a/jellyfin_cleanup.py b/jellyfin_cleanup.py new file mode 100644 index 0000000..8105cf0 --- /dev/null +++ b/jellyfin_cleanup.py @@ -0,0 +1,689 @@ +import argparse +import asyncio +import logging +import os +import random +import sqlite3 +import sys +import time +from contextlib import contextmanager +from datetime import UTC, datetime + +import httpx + +# --------------------------------------------------------------------------- +# CLI +# --------------------------------------------------------------------------- + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser( + prog="jellyfin_cleanup", + description="Find and delete Jellyfin library items by path prefix.", + formatter_class=argparse.ArgumentDefaultsHelpFormatter, + ) + + parser.add_argument( + "paths", + nargs="*", + metavar="PATH", + help="One or more path prefixes to target (e.g. /10TB2/tvShows /10TB/movies). " + "Overrides --target-path.", + ) + parser.add_argument( + "--target-path", + "-t", + action="append", + dest="target_paths", + metavar="PATH", + default=[], + help="Path prefix to target. Can be supplied multiple times.", + ) + parser.add_argument( + "--url", + "-u", + default="http://127.0.0.1:8096", + metavar="URL", + help="Jellyfin base URL.", + ) + parser.add_argument( + "--api-key", + "-k", + default=None, + metavar="KEY", + help="Jellyfin API key. Falls back to JELLYFIN_API_KEY env var.", + ) + parser.add_argument( + "--db", + default="jellyfin_cleanup.db", + metavar="FILE", + help="SQLite database path for caching scraped items.", + ) + parser.add_argument( + "--page-size", + type=int, + default=500, + metavar="N", + help="Items per fetch page.", + ) + parser.add_argument( + "--fetch-concurrency", + type=int, + default=3, + metavar="N", + help="Simultaneous page-fetch requests.", + ) + parser.add_argument( + "--delete-concurrency", + type=int, + default=5, + metavar="N", + help="Simultaneous bulk-delete requests.", + ) + parser.add_argument( + "--delete-batch-size", + type=int, + default=50, + metavar="N", + help="Items per bulk-delete API call.", + ) + parser.add_argument( + "--max-retries", + type=int, + default=5, + metavar="N", + help="Max retries per request before giving up.", + ) + parser.add_argument( + "--retry-backoff-base", + type=float, + default=1.0, + metavar="SECS", + help="Initial retry backoff in seconds (doubles + jitter each attempt).", + ) + parser.add_argument( + "--retry-backoff-max", + type=float, + default=30.0, + metavar="SECS", + help="Maximum retry backoff ceiling in seconds.", + ) + parser.add_argument( + "--timeout-connect", + type=float, + default=5.0, + metavar="SECS", + ) + parser.add_argument( + "--timeout-read", + type=float, + default=60.0, + metavar="SECS", + ) + parser.add_argument( + "--timeout-write", + type=float, + default=10.0, + metavar="SECS", + ) + parser.add_argument( + "--timeout-pool", + type=float, + default=10.0, + metavar="SECS", + ) + parser.add_argument( + "--force-rescrape", + action="store_true", + default=False, + help="Re-scrape Jellyfin even if cached data exists (skip the prompt).", + ) + parser.add_argument( + "--no-rescrape", + action="store_true", + default=False, + help="Never re-scrape; always use cached data (skip the prompt).", + ) + parser.add_argument( + "--yes", + "-y", + action="store_true", + default=False, + help="Skip the delete confirmation prompt.", + ) + parser.add_argument( + "--dry-run", + action="store_true", + default=False, + help="Preview matched items without deleting anything.", + ) + parser.add_argument( + "--verbose", + "-v", + action="store_true", + default=False, + help="Enable DEBUG logging.", + ) + + args = parser.parse_args() + + # Merge positional paths + --target-path into one deduplicated list + all_paths = list(dict.fromkeys(args.paths + args.target_paths)) + args.target_paths = all_paths + + # API key: argparse → env var + if not args.api_key: + args.api_key = os.environ.get("JELLYFIN_API_KEY") + if not args.api_key: + parser.error( + "API key required — pass --api-key or set JELLYFIN_API_KEY env var." + ) + + return args + + +# --------------------------------------------------------------------------- +# Logging +# --------------------------------------------------------------------------- + + +def setup_logging(verbose: bool) -> None: + logging.basicConfig( + level=logging.DEBUG if verbose else logging.INFO, + format="%(asctime)s %(levelname)-8s %(message)s", + datefmt="%H:%M:%S", + ) + + +log = logging.getLogger("jf_cleanup") + +# --------------------------------------------------------------------------- +# SQLite helpers +# --------------------------------------------------------------------------- + + +def db_connect(path: str) -> sqlite3.Connection: + conn = sqlite3.connect(path, check_same_thread=False) + conn.row_factory = sqlite3.Row + conn.execute("PRAGMA journal_mode=WAL") + conn.execute("PRAGMA synchronous=NORMAL") + conn.execute(""" + CREATE TABLE IF NOT EXISTS items ( + id TEXT PRIMARY KEY, + name TEXT NOT NULL, + type TEXT, + path TEXT, + scraped_at TEXT NOT NULL, + delete_status TEXT DEFAULT 'pending', + -- 'pending' | 'deleted' | 'not_found' | 'failed' + delete_attempted_at TEXT, + delete_error TEXT + ) + """) + conn.commit() + return conn + + +@contextmanager +def db_cursor(conn: sqlite3.Connection): + cur = conn.cursor() + try: + yield cur + conn.commit() + except Exception: + conn.rollback() + raise + finally: + cur.close() + + +def upsert_items(conn: sqlite3.Connection, items: list[dict], scraped_at: str) -> None: + with db_cursor(conn) as cur: + cur.executemany( + """ + INSERT INTO items (id, name, type, path, scraped_at) + VALUES (:id, :name, :type, :path, :scraped_at) + ON CONFLICT(id) DO UPDATE SET + name = excluded.name, + type = excluded.type, + path = excluded.path, + scraped_at = excluded.scraped_at + """, + [ + { + "id": item["Id"], + "name": item.get("Name", ""), + "type": item.get("Type", ""), + "path": item.get("Path", ""), + "scraped_at": scraped_at, + } + for item in items + ], + ) + + +def get_pending_targets( + conn: sqlite3.Connection, + target_paths: list[str], +) -> list[sqlite3.Row]: + """Return items under any of the target paths that still need deletion.""" + if not target_paths: + return [] + placeholders = " OR ".join("path LIKE ? || '%'" for _ in target_paths) + query = f""" + SELECT id, name, type, path + FROM items + WHERE ({placeholders}) + AND delete_status IN ('pending', 'failed') + ORDER BY path, type, name + """ + return conn.execute(query, target_paths).fetchall() + + +def mark_deleted(conn: sqlite3.Connection, item_ids: list[str]) -> None: + now = datetime.now(UTC).isoformat() + with db_cursor(conn) as cur: + cur.executemany( + "UPDATE items SET delete_status='deleted', delete_attempted_at=? WHERE id=?", + [(now, iid) for iid in item_ids], + ) + + +def mark_not_found(conn: sqlite3.Connection, item_ids: list[str]) -> None: + now = datetime.now(UTC).isoformat() + with db_cursor(conn) as cur: + cur.executemany( + "UPDATE items SET delete_status='not_found', delete_attempted_at=? WHERE id=?", + [(now, iid) for iid in item_ids], + ) + + +def mark_failed(conn: sqlite3.Connection, item_ids: list[str], error: str) -> None: + now = datetime.now(UTC).isoformat() + with db_cursor(conn) as cur: + cur.executemany( + """UPDATE items + SET delete_status='failed', delete_attempted_at=?, delete_error=? + WHERE id=?""", + [(now, error, iid) for iid in item_ids], + ) + + +def db_stats(conn: sqlite3.Connection) -> dict: + rows = conn.execute( + "SELECT delete_status, COUNT(*) AS n FROM items GROUP BY delete_status" + ).fetchall() + return {r["delete_status"]: r["n"] for r in rows} + + +# --------------------------------------------------------------------------- +# HTTP helpers +# --------------------------------------------------------------------------- + + +async def request_with_retry( + fn, + *args, + cfg: argparse.Namespace, + skip_retry_on: set[int] | None = None, + **kwargs, +) -> httpx.Response: + skip_retry_on = skip_retry_on or set() + last_exc: Exception | None = None + retryable = {429, 500, 502, 503, 504} + + for attempt in range(1, cfg.max_retries + 2): + try: + response = await fn(*args, **kwargs) + if response.status_code in skip_retry_on: + return response + if response.status_code in retryable: + raise httpx.HTTPStatusError( + f"Retryable {response.status_code}", + request=response.request, + response=response, + ) + response.raise_for_status() + return response + + except ( + httpx.TransportError, + httpx.TimeoutException, + httpx.HTTPStatusError, + ) as exc: + last_exc = exc + if attempt > cfg.max_retries: + break + ceiling = min( + cfg.retry_backoff_base * (2 ** (attempt - 1)), + cfg.retry_backoff_max, + ) + delay = random.uniform(0, ceiling) + log.warning( + "Attempt %d/%d failed (%s: %s) — retrying in %.1fs", + attempt, + cfg.max_retries + 1, + type(exc).__name__, + exc, + delay, + ) + await asyncio.sleep(delay) + + raise RuntimeError( + f"Request failed after {cfg.max_retries + 1} attempts: {last_exc}" + ) from last_exc + + +# --------------------------------------------------------------------------- +# Connectivity check +# --------------------------------------------------------------------------- + + +async def check_connectivity( + client: httpx.AsyncClient, cfg: argparse.Namespace +) -> None: + log.info("Checking connectivity → %s", cfg.url) + try: + r = await client.get("/System/Info/Public", timeout=cfg.timeout_connect) + info = r.json() + log.info( + "Server: %s version %s", + info.get("ServerName"), + info.get("Version"), + ) + except httpx.ConnectError as exc: + log.error("Connection failed: %s", exc) + sys.exit(1) + except httpx.TimeoutException: + log.error("Timeout reaching server") + sys.exit(1) + + r = await client.get("/System/Info", timeout=cfg.timeout_connect) + if r.status_code == 200: + log.info("API key valid ✓") + elif r.status_code == 401: + log.error("AUTH FAILED — API key rejected") + sys.exit(1) + elif r.status_code == 404: + log.error("404 on /System/Info — check base URL / path prefix") + sys.exit(1) + else: + log.error("Unexpected auth response %s: %s", r.status_code, r.text) + sys.exit(1) + + +# --------------------------------------------------------------------------- +# Scraping +# --------------------------------------------------------------------------- + + +async def fetch_page( + client: httpx.AsyncClient, + sem: asyncio.Semaphore, + start_index: int, + cfg: argparse.Namespace, +) -> tuple[list, int]: + async with sem: + r = await request_with_retry( + client.get, + "/Items", + cfg=cfg, + params={ + "Recursive": "true", + "Fields": "Path", + "Limit": cfg.page_size, + "StartIndex": start_index, + }, + ) + data = r.json() + return data.get("Items", []), data.get("TotalRecordCount", 0) + + +async def scrape_all_items( + client: httpx.AsyncClient, + conn: sqlite3.Connection, + cfg: argparse.Namespace, +) -> int: + scraped_at = datetime.now(UTC).isoformat() + + log.info("Scraping page 0 to get total record count...") + first_items, total = await fetch_page(client, asyncio.Semaphore(1), 0, cfg) + upsert_items(conn, first_items, scraped_at) + + if total <= cfg.page_size: + log.info("Scraped %d / %d items", len(first_items), total) + return total + + offsets = list(range(cfg.page_size, total, cfg.page_size)) + sem = asyncio.Semaphore(cfg.fetch_concurrency) + completed = 0 + t0 = time.monotonic() + + async def fetch_and_store(offset: int) -> None: + nonlocal completed + items, _ = await fetch_page(client, sem, offset, cfg) + upsert_items(conn, items, scraped_at) + completed += 1 + elapsed = time.monotonic() - t0 + done = min(cfg.page_size + completed * cfg.page_size, total) + rate = done / elapsed if elapsed > 0 else 0 + print(f" {done:>6}/{total} ({rate:.0f} items/s) ", end="\r", flush=True) + + log.info( + "Total %d items across %d pages — fetching (concurrency=%d)...", + total, + 1 + len(offsets), + cfg.fetch_concurrency, + ) + await asyncio.gather(*[fetch_and_store(off) for off in offsets]) + print() + log.info("Scrape complete — %d items stored in %s", total, cfg.db) + return total + + +# --------------------------------------------------------------------------- +# Deletion +# --------------------------------------------------------------------------- + + +async def _delete_individually( + client: httpx.AsyncClient, + conn: sqlite3.Connection, + batch: list[sqlite3.Row], + cfg: argparse.Namespace, +) -> None: + for row in batch: + item_id = row["id"] + name = row["name"] + try: + r = await request_with_retry( + client.delete, + f"/Items/{item_id}", + cfg=cfg, + skip_retry_on={404}, + ) + if r.status_code == 404: + log.info("[NOT FOUND — already gone] %s (%s)", name, item_id) + mark_not_found(conn, [item_id]) + else: + log.info("[DELETED] %s (%s)", name, item_id) + mark_deleted(conn, [item_id]) + except RuntimeError as exc: + log.error("[FAILED] %s (%s) — %s", name, item_id, exc) + mark_failed(conn, [item_id], str(exc)) + + +async def delete_batch( + client: httpx.AsyncClient, + sem: asyncio.Semaphore, + conn: sqlite3.Connection, + batch: list[sqlite3.Row], + cfg: argparse.Namespace, +) -> None: + ids = [row["id"] for row in batch] + + async with sem: + try: + r = await request_with_retry( + client.delete, + "/Items", + cfg=cfg, + params={"ids": ",".join(ids)}, + skip_retry_on={404}, + ) + if r.status_code == 404: + # One or more items missing — fall back to per-item so each + # gets its own status recorded correctly + log.debug( + "Bulk 404 on batch of %d — falling back to individual deletes", + len(ids), + ) + await _delete_individually(client, conn, batch, cfg) + else: + log.info("[DELETED batch of %d]", len(ids)) + mark_deleted(conn, ids) + + except RuntimeError as exc: + log.error("[FAILED batch of %d] %s", len(ids), exc) + mark_failed(conn, ids, str(exc)) + + +async def delete_targets( + client: httpx.AsyncClient, + conn: sqlite3.Connection, + targets: list[sqlite3.Row], + cfg: argparse.Namespace, +) -> None: + batches = [ + targets[i : i + cfg.delete_batch_size] + for i in range(0, len(targets), cfg.delete_batch_size) + ] + sem = asyncio.Semaphore(cfg.delete_concurrency) + log.info( + "Deleting %d items in %d batches (batch_size=%d, concurrency=%d)...", + len(targets), + len(batches), + cfg.delete_batch_size, + cfg.delete_concurrency, + ) + await asyncio.gather(*[delete_batch(client, sem, conn, b, cfg) for b in batches]) + + +# --------------------------------------------------------------------------- +# Main +# --------------------------------------------------------------------------- + + +async def main(cfg: argparse.Namespace) -> None: + conn = db_connect(cfg.db) + + timeout = httpx.Timeout( + connect=cfg.timeout_connect, + read=cfg.timeout_read, + write=cfg.timeout_write, + pool=cfg.timeout_pool, + ) + max_conn = max(cfg.fetch_concurrency, cfg.delete_concurrency) + 2 + + async with httpx.AsyncClient( + base_url=cfg.url, + headers={"X-Emby-Token": cfg.api_key, "Content-Type": "application/json"}, + timeout=timeout, + limits=httpx.Limits( + max_connections=max_conn, + max_keepalive_connections=max_conn - 2, + ), + ) as client: + await check_connectivity(client, cfg) + + # --- Scrape decision --- + existing = conn.execute("SELECT COUNT(*) FROM items").fetchone()[0] + do_scrape: bool + + if cfg.force_rescrape: + do_scrape = True + elif cfg.no_rescrape: + do_scrape = False + elif existing > 0: + ans = ( + input( + f"\n{existing} items cached in {cfg.db}. " + "Re-scrape from Jellyfin? [y/N]: " + ) + .strip() + .lower() + ) + do_scrape = ans == "y" + else: + do_scrape = True + + if do_scrape: + await scrape_all_items(client, conn, cfg) + else: + log.info("Using cached data from %s", cfg.db) + + # --- Find targets across all requested paths --- + if not cfg.target_paths: + log.error( + "No target paths specified. " + "Pass paths as positional arguments or use --target-path." + ) + sys.exit(1) + + log.info("Target paths: %s", cfg.target_paths) + targets = get_pending_targets(conn, cfg.target_paths) + log.info("Found %d pending/failed items across all target paths", len(targets)) + + if not targets: + log.info("Nothing to delete.") + log.info("DB stats: %s", db_stats(conn)) + return + + # --- Preview --- + print(f"\nItems to delete ({len(targets)}):") + # Group by path prefix for readability + for tp in cfg.target_paths: + group = [r for r in targets if r["path"].startswith(tp)] + if group: + print(f"\n [{tp}] ({len(group)} items)") + for row in group[:10]: + print(f" [{row['type']:12}] {row['name']}") + if len(group) > 10: + print(f" ... and {len(group) - 10} more") + + if cfg.dry_run: + log.info("Dry run — nothing deleted.") + return + + # --- Confirm --- + if not cfg.yes: + confirm = ( + input(f"\nDelete all {len(targets)} items? (yes/no): ").strip().lower() + ) + if confirm != "yes": + log.info("Aborted.") + return + + # --- Delete --- + await delete_targets(client, conn, targets, cfg) + + # --- Summary --- + stats = db_stats(conn) + log.info("Done. DB stats: %s", stats) + if stats.get("failed", 0): + log.warning( + "%d items still marked 'failed' — re-run to retry " + "(cached data will be reused, no re-scrape needed).", + stats["failed"], + ) + + +def main_sync() -> None: + """Entry point for the ``jellyfin-cleanup`` console script.""" + cfg = parse_args() + setup_logging(cfg.verbose) + asyncio.run(main(cfg)) + + +if __name__ == "__main__": + main_sync() diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..b33dd29 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,40 @@ +[build-system] +requires = ["setuptools>=68", "wheel"] +build-backend = "setuptools.build_meta" + +[project] +name = "jellyfin-cleanup" +version = "1.0.0" +description = "Find and delete Jellyfin library items by path prefix" +readme = "README.md" +license = { file = "LICENSE" } +requires-python = ">=3.11" +dependencies = [ + "httpx>=0.27", +] + +[project.scripts] +jellyfin-cleanup = "jellyfin_cleanup:main_sync" + +[project.optional-dependencies] +dev = [ + "pytest>=8.0", + "pytest-asyncio>=0.23", + "ruff>=0.4", + "respx>=0.21", +] + +[tool.setuptools] +py-modules = ["jellyfin_cleanup"] + +[tool.ruff] +line-length = 100 +target-version = "py311" + +[tool.ruff.lint] +select = ["E", "F", "W", "I", "UP"] +ignore = ["E501"] + +[tool.pytest.ini_options] +asyncio_mode = "auto" +testpaths = ["tests"] diff --git a/tests/test_cli.py b/tests/test_cli.py new file mode 100644 index 0000000..63cf495 --- /dev/null +++ b/tests/test_cli.py @@ -0,0 +1,170 @@ +"""Tests for CLI argument parsing in jellyfin_cleanup.""" + +import os +import sys +from unittest.mock import patch + +import pytest + +import jellyfin_cleanup + + +def _parse(args: list[str], env: dict | None = None) -> object: + """Run parse_args with the given argv and optional env overrides.""" + env_vars = {"JELLYFIN_API_KEY": "test-key"} + if env: + env_vars.update(env) + with patch.object(sys, "argv", ["jellyfin_cleanup"] + args): + with patch.dict(os.environ, env_vars, clear=False): + return jellyfin_cleanup.parse_args() + + +# --------------------------------------------------------------------------- +# API key handling +# --------------------------------------------------------------------------- + + +def test_api_key_from_flag(): + cfg = _parse(["--api-key", "mykey"], env={"JELLYFIN_API_KEY": ""}) + assert cfg.api_key == "mykey" + + +def test_api_key_from_env(): + cfg = _parse([], env={"JELLYFIN_API_KEY": "envkey"}) + assert cfg.api_key == "envkey" + + +def test_api_key_flag_takes_precedence_over_env(): + cfg = _parse(["--api-key", "flagkey"], env={"JELLYFIN_API_KEY": "envkey"}) + assert cfg.api_key == "flagkey" + + +def test_missing_api_key_exits(): + with patch.object(sys, "argv", ["jellyfin_cleanup"]): + with patch.dict(os.environ, {"JELLYFIN_API_KEY": ""}, clear=False): + # Remove key entirely + env = os.environ.copy() + env.pop("JELLYFIN_API_KEY", None) + with patch.dict(os.environ, env, clear=True): + with pytest.raises(SystemExit): + jellyfin_cleanup.parse_args() + + +# --------------------------------------------------------------------------- +# Path merging +# --------------------------------------------------------------------------- + + +def test_positional_paths(): + cfg = _parse(["/mnt/drive1", "/mnt/drive2"]) + assert cfg.target_paths == ["/mnt/drive1", "/mnt/drive2"] + + +def test_target_path_flag(): + cfg = _parse(["--target-path", "/mnt/drive1", "-t", "/mnt/drive2"]) + assert "/mnt/drive1" in cfg.target_paths + assert "/mnt/drive2" in cfg.target_paths + + +def test_positional_and_flag_merged(): + cfg = _parse(["/mnt/drive1", "--target-path", "/mnt/drive2"]) + assert set(cfg.target_paths) == {"/mnt/drive1", "/mnt/drive2"} + + +def test_duplicate_paths_deduplicated(): + cfg = _parse(["/mnt/drive1", "--target-path", "/mnt/drive1"]) + assert cfg.target_paths.count("/mnt/drive1") == 1 + + +def test_no_paths_is_empty(): + cfg = _parse([]) + assert cfg.target_paths == [] + + +# --------------------------------------------------------------------------- +# Defaults +# --------------------------------------------------------------------------- + + +def test_default_url(): + cfg = _parse([]) + assert cfg.url == "http://127.0.0.1:8096" + + +def test_custom_url(): + cfg = _parse(["--url", "http://nas:8096"]) + assert cfg.url == "http://nas:8096" + + +def test_default_page_size(): + cfg = _parse([]) + assert cfg.page_size == 500 + + +def test_default_delete_batch_size(): + cfg = _parse([]) + assert cfg.delete_batch_size == 50 + + +def test_default_flags_false(): + cfg = _parse([]) + assert cfg.dry_run is False + assert cfg.yes is False + assert cfg.verbose is False + assert cfg.force_rescrape is False + assert cfg.no_rescrape is False + + +def test_dry_run_flag(): + cfg = _parse(["--dry-run"]) + assert cfg.dry_run is True + + +def test_yes_flag(): + cfg = _parse(["--yes"]) + assert cfg.yes is True + + +def test_verbose_short_flag(): + cfg = _parse(["-v"]) + assert cfg.verbose is True + + +def test_force_rescrape(): + cfg = _parse(["--force-rescrape"]) + assert cfg.force_rescrape is True + + +def test_no_rescrape(): + cfg = _parse(["--no-rescrape"]) + assert cfg.no_rescrape is True + + +# --------------------------------------------------------------------------- +# Numeric parameters +# --------------------------------------------------------------------------- + + +def test_custom_page_size(): + cfg = _parse(["--page-size", "100"]) + assert cfg.page_size == 100 + + +def test_custom_max_retries(): + cfg = _parse(["--max-retries", "3"]) + assert cfg.max_retries == 3 + + +def test_custom_timeouts(): + cfg = _parse( + [ + "--timeout-connect", "2.5", + "--timeout-read", "30.0", + "--timeout-write", "5.0", + "--timeout-pool", "8.0", + ] + ) + assert cfg.timeout_connect == 2.5 + assert cfg.timeout_read == 30.0 + assert cfg.timeout_write == 5.0 + assert cfg.timeout_pool == 8.0 diff --git a/tests/test_db.py b/tests/test_db.py new file mode 100644 index 0000000..f7bbc74 --- /dev/null +++ b/tests/test_db.py @@ -0,0 +1,213 @@ +"""Tests for SQLite helper functions in jellyfin_cleanup.""" + + +import pytest + +from jellyfin_cleanup import ( + db_connect, + db_stats, + get_pending_targets, + mark_deleted, + mark_failed, + mark_not_found, + upsert_items, +) + +SCRAPED_AT = "2024-01-01T00:00:00+00:00" + + +@pytest.fixture() +def conn(tmp_path): + """In-memory SQLite connection for each test.""" + db_path = str(tmp_path / "test.db") + connection = db_connect(db_path) + yield connection + connection.close() + + +def _make_item( + item_id: str, + name: str = "Test", + type_: str = "Movie", + path: str = "/data/movies/Test", +) -> dict: + return {"Id": item_id, "Name": name, "Type": type_, "Path": path} + + +# --------------------------------------------------------------------------- +# db_connect +# --------------------------------------------------------------------------- + + +def test_db_connect_creates_table(tmp_path): + conn = db_connect(str(tmp_path / "fresh.db")) + tables = conn.execute( + "SELECT name FROM sqlite_master WHERE type='table'" + ).fetchall() + table_names = [r[0] for r in tables] + assert "items" in table_names + conn.close() + + +# --------------------------------------------------------------------------- +# upsert_items +# --------------------------------------------------------------------------- + + +def test_upsert_inserts_new_item(conn): + upsert_items(conn, [_make_item("abc")], SCRAPED_AT) + row = conn.execute("SELECT * FROM items WHERE id='abc'").fetchone() + assert row is not None + assert row["name"] == "Test" + assert row["path"] == "/data/movies/Test" + assert row["delete_status"] == "pending" + + +def test_upsert_updates_existing_item(conn): + upsert_items(conn, [_make_item("abc", name="Old Name")], SCRAPED_AT) + upsert_items(conn, [_make_item("abc", name="New Name")], SCRAPED_AT) + row = conn.execute("SELECT name FROM items WHERE id='abc'").fetchone() + assert row["name"] == "New Name" + + +def test_upsert_multiple_items(conn): + items = [_make_item(f"id{i}") for i in range(5)] + upsert_items(conn, items, SCRAPED_AT) + count = conn.execute("SELECT COUNT(*) FROM items").fetchone()[0] + assert count == 5 + + +def test_upsert_missing_optional_fields(conn): + """Items without Name/Type/Path should not raise.""" + upsert_items(conn, [{"Id": "x1"}], SCRAPED_AT) + row = conn.execute("SELECT * FROM items WHERE id='x1'").fetchone() + assert row["name"] == "" + assert row["type"] == "" + assert row["path"] == "" + + +# --------------------------------------------------------------------------- +# get_pending_targets +# --------------------------------------------------------------------------- + + +def test_get_pending_targets_empty_paths(conn): + upsert_items(conn, [_make_item("abc")], SCRAPED_AT) + result = get_pending_targets(conn, []) + assert result == [] + + +def test_get_pending_targets_matches_prefix(conn): + upsert_items( + conn, + [ + _make_item("a1", path="/mnt/drive1/movies/Film A"), + _make_item("a2", path="/mnt/drive1/shows/Show B"), + _make_item("b1", path="/mnt/drive2/movies/Film C"), + ], + SCRAPED_AT, + ) + result = get_pending_targets(conn, ["/mnt/drive1"]) + ids = {r["id"] for r in result} + assert ids == {"a1", "a2"} + + +def test_get_pending_targets_multiple_paths(conn): + upsert_items( + conn, + [ + _make_item("a1", path="/drive1/movie"), + _make_item("b1", path="/drive2/show"), + _make_item("c1", path="/drive3/other"), + ], + SCRAPED_AT, + ) + result = get_pending_targets(conn, ["/drive1", "/drive2"]) + ids = {r["id"] for r in result} + assert ids == {"a1", "b1"} + + +def test_get_pending_targets_excludes_deleted(conn): + upsert_items(conn, [_make_item("a1", path="/mnt/movies/x")], SCRAPED_AT) + mark_deleted(conn, ["a1"]) + result = get_pending_targets(conn, ["/mnt/movies"]) + assert result == [] + + +def test_get_pending_targets_includes_failed(conn): + upsert_items(conn, [_make_item("a1", path="/mnt/movies/x")], SCRAPED_AT) + mark_failed(conn, ["a1"], "timeout") + result = get_pending_targets(conn, ["/mnt/movies"]) + assert len(result) == 1 + + +def test_get_pending_targets_excludes_not_found(conn): + upsert_items(conn, [_make_item("a1", path="/mnt/movies/x")], SCRAPED_AT) + mark_not_found(conn, ["a1"]) + result = get_pending_targets(conn, ["/mnt/movies"]) + assert result == [] + + +# --------------------------------------------------------------------------- +# mark_deleted / mark_not_found / mark_failed +# --------------------------------------------------------------------------- + + +def test_mark_deleted(conn): + upsert_items(conn, [_make_item("d1")], SCRAPED_AT) + mark_deleted(conn, ["d1"]) + row = conn.execute("SELECT delete_status FROM items WHERE id='d1'").fetchone() + assert row["delete_status"] == "deleted" + + +def test_mark_not_found(conn): + upsert_items(conn, [_make_item("n1")], SCRAPED_AT) + mark_not_found(conn, ["n1"]) + row = conn.execute("SELECT delete_status FROM items WHERE id='n1'").fetchone() + assert row["delete_status"] == "not_found" + + +def test_mark_failed(conn): + upsert_items(conn, [_make_item("f1")], SCRAPED_AT) + mark_failed(conn, ["f1"], "some error") + row = conn.execute( + "SELECT delete_status, delete_error FROM items WHERE id='f1'" + ).fetchone() + assert row["delete_status"] == "failed" + assert row["delete_error"] == "some error" + + +def test_mark_deleted_sets_timestamp(conn): + upsert_items(conn, [_make_item("t1")], SCRAPED_AT) + mark_deleted(conn, ["t1"]) + row = conn.execute( + "SELECT delete_attempted_at FROM items WHERE id='t1'" + ).fetchone() + assert row["delete_attempted_at"] is not None + + +# --------------------------------------------------------------------------- +# db_stats +# --------------------------------------------------------------------------- + + +def test_db_stats_empty(conn): + assert db_stats(conn) == {} + + +def test_db_stats_counts(conn): + items = [_make_item(f"id{i}") for i in range(4)] + upsert_items(conn, items, SCRAPED_AT) + mark_deleted(conn, ["id0", "id1"]) + mark_not_found(conn, ["id2"]) + stats = db_stats(conn) + assert stats["pending"] == 1 + assert stats["deleted"] == 2 + assert stats["not_found"] == 1 + + +def test_db_stats_failed(conn): + upsert_items(conn, [_make_item("e1")], SCRAPED_AT) + mark_failed(conn, ["e1"], "err") + stats = db_stats(conn) + assert stats.get("failed") == 1