Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 4 additions & 4 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -65,7 +65,7 @@ usage: jellyfin_cleanup [-h] [--target-path PATH] [--url URL] [--api-key KEY]
[--retry-backoff-max SECS] [--timeout-connect SECS]
[--timeout-read SECS] [--timeout-write SECS]
[--timeout-pool SECS] [--force-rescrape] [--no-rescrape]
[--yes] [--dry-run] [--verbose]
[--yes] [--dry-run] [--badData] [--verbose]
[PATH ...]
```

Expand All @@ -87,6 +87,7 @@ usage: jellyfin_cleanup [-h] [--target-path PATH] [--url URL] [--api-key KEY]
| `--no-rescrape` | `False` | Always use cached data |
| `--yes`, `-y` | `False` | Skip delete confirmation prompt |
| `--dry-run` | `False` | Preview without deleting |
| `--badData` | `False` | Ignore path filters and target entries with bad metadata (missing season/episode values or missing media versions) |
| `--verbose`, `-v` | `False` | Enable DEBUG logging |

## Development
Expand All @@ -105,13 +106,12 @@ ruff check .
## How It Works

1. **Connectivity check** — verifies the server is reachable and the API key is valid.
2. **Scrape** — pages through `GET /Items?Recursive=true&Fields=Path` and stores every item in a local SQLite database with its `delete_status = 'pending'`.
3. **Target matching** — queries the DB for items whose `path` starts with any of the specified prefixes and whose `delete_status` is `pending` or `failed`.
2. **Scrape** — pages through `GET /Items` (including path + metadata fields) and stores every item in a local SQLite database with its `delete_status = 'pending'`.
3. **Target matching** — either queries the DB by `path` prefix (default) or, with `--badData`, finds entries with invalid episode/season metadata or no media versions.
4. **Preview** — prints a grouped summary of matching items.
5. **Delete** — sends concurrent batched `DELETE /Items?ids=…` requests; records each outcome (`deleted`, `not_found`, or `failed`) back to the DB.
6. **Summary** — prints final DB statistics and warns if any items remain `failed`.

## License

GPL-3.0 — see [LICENSE](LICENSE).

2 changes: 2 additions & 0 deletions jellyfin_cleanup/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
Database,
db_connect,
db_stats,
get_bad_data_targets,
get_pending_targets,
mark_deleted,
mark_failed,
Expand All @@ -30,6 +31,7 @@
# Database (backward-compat free functions)
"db_connect",
"db_stats",
"get_bad_data_targets",
"get_pending_targets",
"mark_deleted",
"mark_failed",
Expand Down
8 changes: 8 additions & 0 deletions jellyfin_cleanup/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -147,6 +147,14 @@ def parse_args() -> argparse.Namespace:
default=False,
help="Preview matched items without deleting anything.",
)
parser.add_argument(
"--badData",
"--bad-data",
dest="bad_data",
action="store_true",
default=False,
help="Target entries with bad metadata (missing season/episode data or no versions).",

Copilot AI Apr 8, 2026

Copy link

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The --badData help text doesn’t mention that it ignores any provided PATH/--target-path filters (the program does this in core). Consider updating the flag description to explicitly state that path filters are ignored in this mode so users don’t think both filters apply.

Suggested change
help="Target entries with bad metadata (missing season/episode data or no versions).",
help="Target entries with bad metadata (missing season/episode data or no versions). "
"Ignores any positional PATH values and --target-path filters in this mode.",

Copilot uses AI. Check for mistakes.
)
parser.add_argument(
"--verbose",
"-v",
Expand Down
2 changes: 1 addition & 1 deletion jellyfin_cleanup/client.py
Original file line number Diff line number Diff line change
Expand Up @@ -157,7 +157,7 @@ async def fetch_page(
"/Items",
params={
"Recursive": "true",
"Fields": "Path",
"Fields": "Path,IndexNumber,ParentIndexNumber,MediaSources",
"Limit": self.cfg.page_size,
"StartIndex": start_index,
},
Expand Down
41 changes: 28 additions & 13 deletions jellyfin_cleanup/core.py
Original file line number Diff line number Diff line change
Expand Up @@ -54,17 +54,23 @@ async def main(cfg: argparse.Namespace) -> None:
else:
log.info("Using cached data from %s", cfg.db)

# --- Find targets across all requested paths ---
if not cfg.target_paths:
# --- Find targets ---
if not cfg.bad_data and not cfg.target_paths:
log.error(
"No target paths specified. "
"Pass paths as positional arguments or use --target-path."
)
sys.exit(1)

log.info("Target paths: %s", cfg.target_paths)
targets = db.get_pending_targets(cfg.target_paths)
log.info("Found %d pending/failed items across all target paths", len(targets))
if cfg.bad_data:
if cfg.target_paths:
log.info("--badData/--bad-data set; ignoring provided target paths.")
targets = db.get_bad_data_targets()
log.info("Found %d pending/failed items with bad metadata", len(targets))
else:
log.info("Target paths: %s", cfg.target_paths)
targets = db.get_pending_targets(cfg.target_paths)
log.info("Found %d pending/failed items across all target paths", len(targets))
Comment on lines +65 to +73

Copilot AI Apr 8, 2026

Copy link

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

In --badData mode the target selection depends on additional scraped metadata fields; if the user opts to reuse cached data (prompt “Re-scrape?” → no, or --no-rescrape) and the cache predates these fields, the selection can be misleading/over-inclusive. Consider detecting whether the cache includes non-NULL values for the new metadata columns and forcing a re-scrape (or warning+exit) when it doesn’t.

Copilot uses AI. Check for mistakes.

if not targets:
log.info("Nothing to delete.")
Expand All @@ -73,14 +79,23 @@ async def main(cfg: argparse.Namespace) -> None:

# --- Preview ---
print(f"\nItems to delete ({len(targets)}):")
for tp in cfg.target_paths:
group = [r for r in targets if r["path"].startswith(tp)]
if group:
print(f"\n [{tp}] ({len(group)} items)")
for row in group[:10]:
print(f" [{row['type']:12}] {row['name']}")
if len(group) > 10:
print(f" ... and {len(group) - 10} more")
if cfg.bad_data:
for row in targets[:30]:
print(
f" [{row['type']:12}] {row['name']} ({row['bad_reason']})"
f" - {row['path'] or '<no path>'}"
)
if len(targets) > 30:
print(f" ... and {len(targets) - 30} more")
else:
for tp in cfg.target_paths:
group = [r for r in targets if r["path"].startswith(tp)]
if group:
print(f"\n [{tp}] ({len(group)} items)")
for row in group[:10]:
print(f" [{row['type']:12}] {row['name']}")
if len(group) > 10:
print(f" ... and {len(group) - 10} more")

if cfg.dry_run:
log.info("Dry run — nothing deleted.")
Expand Down
135 changes: 103 additions & 32 deletions jellyfin_cleanup/database.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,19 +16,7 @@ def __init__(self, path: str) -> None:
self._conn.row_factory = sqlite3.Row
self._conn.execute("PRAGMA journal_mode=WAL")
self._conn.execute("PRAGMA synchronous=NORMAL")
self._conn.execute("""
CREATE TABLE IF NOT EXISTS items (
id TEXT PRIMARY KEY,
name TEXT NOT NULL,
type TEXT,
path TEXT,
scraped_at TEXT NOT NULL,
delete_status TEXT DEFAULT 'pending',
-- 'pending' | 'deleted' | 'not_found' | 'failed'
delete_attempted_at TEXT,
delete_error TEXT
)
""")
_initialize_items_table(self._conn)
self._conn.commit()

# -- context-manager support ------------------------------------------
Expand Down Expand Up @@ -64,13 +52,34 @@ def upsert_items(self, items: list[dict], scraped_at: str) -> None:
with self._cursor() as cur:
cur.executemany(
"""
INSERT INTO items (id, name, type, path, scraped_at)
VALUES (:id, :name, :type, :path, :scraped_at)
INSERT INTO items (
id,
name,
type,
path,
scraped_at,
index_number,
parent_index_number,
media_source_count
)
VALUES (
:id,
:name,
:type,
:path,
:scraped_at,
:index_number,
:parent_index_number,
:media_source_count
)
ON CONFLICT(id) DO UPDATE SET
name = excluded.name,
type = excluded.type,
path = excluded.path,
scraped_at = excluded.scraped_at
name = excluded.name,
type = excluded.type,
path = excluded.path,
scraped_at = excluded.scraped_at,
index_number = excluded.index_number,
parent_index_number = excluded.parent_index_number,
media_source_count = excluded.media_source_count
""",
[
{
Expand All @@ -79,6 +88,9 @@ def upsert_items(self, items: list[dict], scraped_at: str) -> None:
"type": item.get("Type", ""),
"path": item.get("Path", ""),
"scraped_at": scraped_at,
"index_number": item.get("IndexNumber"),
"parent_index_number": item.get("ParentIndexNumber"),
"media_source_count": _media_source_count(item),
}
for item in items
],
Expand Down Expand Up @@ -127,6 +139,36 @@ def mark_failed(self, item_ids: list[str], error: str) -> None:
[(now, error, iid) for iid in item_ids],
)

def get_bad_data_targets(self) -> list[sqlite3.Row]:
return self._conn.execute(
"""
SELECT
id,
name,
type,
path,
CASE
WHEN type='Episode'
AND (parent_index_number IS NULL OR index_number IS NULL)
THEN 'missing season or episode number'
WHEN type='Season'
AND index_number IS NULL
THEN 'missing season number'
WHEN type IN ('Episode', 'Movie', 'Video')
AND COALESCE(media_source_count, 0) = 0
THEN 'no media versions'
END AS bad_reason
FROM items
WHERE delete_status IN ('pending', 'failed')
AND (
(type='Episode' AND (parent_index_number IS NULL OR index_number IS NULL))
OR (type='Season' AND index_number IS NULL)
OR (type IN ('Episode', 'Movie', 'Video') AND COALESCE(media_source_count, 0) = 0)
Comment on lines +152 to +166

Copilot AI Apr 8, 2026

Copy link

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

get_bad_data_targets treats NULL metadata as definitively “bad”. After migrating an existing cache DB (created before these columns existed), index_number, parent_index_number, and media_source_count will be NULL for most/all rows, so --badData can match nearly everything (especially due to COALESCE(media_source_count, 0) = 0). This is risky because it can drive unintended mass deletions when users choose to reuse cached data. Consider introducing a schema/user_version and forcing a rescrape (or warning+exit) when cached rows don’t have these fields populated, and/or adjusting the predicate so “unknown/not-scraped” isn’t treated as “bad” (e.g., don’t COALESCE NULL to 0 for versions).

Suggested change
AND (parent_index_number IS NULL OR index_number IS NULL)
THEN 'missing season or episode number'
WHEN type='Season'
AND index_number IS NULL
THEN 'missing season number'
WHEN type IN ('Episode', 'Movie', 'Video')
AND COALESCE(media_source_count, 0) = 0
THEN 'no media versions'
END AS bad_reason
FROM items
WHERE delete_status IN ('pending', 'failed')
AND (
(type='Episode' AND (parent_index_number IS NULL OR index_number IS NULL))
OR (type='Season' AND index_number IS NULL)
OR (type IN ('Episode', 'Movie', 'Video') AND COALESCE(media_source_count, 0) = 0)
AND (
(parent_index_number IS NULL AND index_number IS NOT NULL)
OR (parent_index_number IS NOT NULL AND index_number IS NULL)
)
THEN 'missing season or episode number'
WHEN type IN ('Episode', 'Movie', 'Video')
AND media_source_count = 0
THEN 'no media versions'
END AS bad_reason
FROM items
WHERE delete_status IN ('pending', 'failed')
AND (
(
type='Episode'
AND (
(parent_index_number IS NULL AND index_number IS NOT NULL)
OR (parent_index_number IS NOT NULL AND index_number IS NULL)
)
)
OR (type IN ('Episode', 'Movie', 'Video') AND media_source_count = 0)

Copilot uses AI. Check for mistakes.
)
ORDER BY type, path, name
"""
).fetchall()

def stats(self) -> dict:
rows = self._conn.execute(
"SELECT delete_status, COUNT(*) AS n FROM items GROUP BY delete_status"
Expand Down Expand Up @@ -155,19 +197,7 @@ def db_connect(path: str) -> sqlite3.Connection:
conn.row_factory = sqlite3.Row
conn.execute("PRAGMA journal_mode=WAL")
conn.execute("PRAGMA synchronous=NORMAL")
conn.execute("""
CREATE TABLE IF NOT EXISTS items (
id TEXT PRIMARY KEY,
name TEXT NOT NULL,
type TEXT,
path TEXT,
scraped_at TEXT NOT NULL,
delete_status TEXT DEFAULT 'pending',
-- 'pending' | 'deleted' | 'not_found' | 'failed'
delete_attempted_at TEXT,
delete_error TEXT
)
""")
_initialize_items_table(conn)
conn.commit()
return conn

Expand All @@ -187,6 +217,12 @@ def get_pending_targets(
return db.get_pending_targets(target_paths)


def get_bad_data_targets(conn: sqlite3.Connection) -> list[sqlite3.Row]:
db = Database.__new__(Database)
db._conn = conn
return db.get_bad_data_targets()


def mark_deleted(conn: sqlite3.Connection, item_ids: list[str]) -> None:
db = Database.__new__(Database)
db._conn = conn
Expand All @@ -209,3 +245,38 @@ def db_stats(conn: sqlite3.Connection) -> dict:
db = Database.__new__(Database)
db._conn = conn
return db.stats()


def _initialize_items_table(conn: sqlite3.Connection) -> None:
conn.execute("""
CREATE TABLE IF NOT EXISTS items (
id TEXT PRIMARY KEY,
name TEXT NOT NULL,
type TEXT,
path TEXT,
scraped_at TEXT NOT NULL,
delete_status TEXT DEFAULT 'pending',
-- 'pending' | 'deleted' | 'not_found' | 'failed'
delete_attempted_at TEXT,
delete_error TEXT,
index_number INTEGER,
parent_index_number INTEGER,
media_source_count INTEGER
)
""")
columns = {
row["name"] for row in conn.execute("PRAGMA table_info(items)").fetchall()
}
if "index_number" not in columns:
conn.execute("ALTER TABLE items ADD COLUMN index_number INTEGER")
if "parent_index_number" not in columns:
conn.execute("ALTER TABLE items ADD COLUMN parent_index_number INTEGER")
if "media_source_count" not in columns:
conn.execute("ALTER TABLE items ADD COLUMN media_source_count INTEGER")


def _media_source_count(item: dict) -> int | None:
media_sources = item.get("MediaSources")
if media_sources is None:
return None
return len(media_sources)
6 changes: 6 additions & 0 deletions tests/test_cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -109,6 +109,7 @@ def test_default_delete_batch_size():
def test_default_flags_false():
cfg = _parse([])
assert cfg.dry_run is False
assert cfg.bad_data is False
assert cfg.yes is False
assert cfg.verbose is False
assert cfg.force_rescrape is False
Expand All @@ -125,6 +126,11 @@ def test_yes_flag():
assert cfg.yes is True


def test_bad_data_flag():
cfg = _parse(["--badData"])
assert cfg.bad_data is True


def test_verbose_short_flag():
cfg = _parse(["-v"])
assert cfg.verbose is True
Expand Down
Loading
Loading