Compare commits
5 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 2e8f3b5c84 | |||
| c3aca9217d | |||
| f75d591453 | |||
| c84f968916 | |||
| 1d121b427d |
@@ -1 +1 @@
|
|||||||
v1.5.11
|
v1.5.12
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "aniworld-web",
|
"name": "aniworld-web",
|
||||||
"version": "1.5.11",
|
"version": "1.5.12",
|
||||||
"description": "Aniworld Anime Download Manager - Web Frontend",
|
"description": "Aniworld Anime Download Manager - Web Frontend",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
|
|||||||
332
src/cli/clean_duplicate_episodes.py
Normal file
332
src/cli/clean_duplicate_episodes.py
Normal file
@@ -0,0 +1,332 @@
|
|||||||
|
"""CLI tool to clean up duplicate ``Episode`` rows.
|
||||||
|
|
||||||
|
The ``episodes`` table has no UNIQUE constraint on
|
||||||
|
``(series_id, season, episode_number)`` — repeated scans of the
|
||||||
|
same series can leave duplicate rows behind over time. They don't
|
||||||
|
break the app (the ``AnimeSeries.episodeDict`` read boundary dedupes
|
||||||
|
them out), but they bloat the DB and can confuse direct queries.
|
||||||
|
|
||||||
|
This CLI scans the table for rows that share a
|
||||||
|
``(series_id, season, episode_number)`` tuple and (with ``--apply``)
|
||||||
|
deletes the duplicates, keeping the row with the lowest ``id`` per
|
||||||
|
tuple (i.e. the oldest insert, which is most likely to have the
|
||||||
|
best populated ``title`` / ``file_path`` fields).
|
||||||
|
|
||||||
|
Usage::
|
||||||
|
|
||||||
|
# Inspect — list duplicates without modifying anything.
|
||||||
|
python -m src.cli.clean_duplicate_episodes
|
||||||
|
|
||||||
|
# Apply — actually delete the duplicates.
|
||||||
|
python -m src.cli.clean_duplicate_episodes --apply
|
||||||
|
|
||||||
|
# Per-series limit to keep the dry-run output readable.
|
||||||
|
python -m src.cli.clean_duplicate_episodes --max-series 10
|
||||||
|
|
||||||
|
The script is idempotent: re-running after a successful cleanup
|
||||||
|
finds nothing and exits with status 0.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import asyncio
|
||||||
|
import logging
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import List, Tuple
|
||||||
|
|
||||||
|
# Add project root to path so ``from src.server...`` works when
|
||||||
|
# invoked as ``python -m src.cli.clean_duplicate_episodes``.
|
||||||
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent.parent))
|
||||||
|
|
||||||
|
from sqlalchemy import delete, func, select, tuple_
|
||||||
|
|
||||||
|
from src.server.database.connection import close_db, init_db
|
||||||
|
from src.server.database.models import AnimeSeries, Episode
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
# Tuple shape: (series_id, season, episode_number, count_of_rows,
|
||||||
|
# min_id_kept, max_id_deleted).
|
||||||
|
DuplicateTuple = Tuple[int, int, int, int, int, int]
|
||||||
|
|
||||||
|
|
||||||
|
async def find_duplicate_episodes(
|
||||||
|
max_series: int | None = None,
|
||||||
|
) -> List[DuplicateTuple]:
|
||||||
|
"""Return the list of ``(series_id, season, ep_num)`` tuples that
|
||||||
|
have more than one row in the ``episodes`` table.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
max_series: If given, only report duplicates for the first N
|
||||||
|
distinct ``series_id`` values that have duplicates — used
|
||||||
|
to keep the dry-run output manageable for large libraries.
|
||||||
|
"""
|
||||||
|
duplicates_subquery = (
|
||||||
|
select(
|
||||||
|
Episode.series_id.label("series_id"),
|
||||||
|
Episode.season.label("season"),
|
||||||
|
Episode.episode_number.label("episode_number"),
|
||||||
|
func.count(Episode.id).label("row_count"),
|
||||||
|
func.min(Episode.id).label("keep_id"),
|
||||||
|
func.max(Episode.id).label("max_id"),
|
||||||
|
)
|
||||||
|
.group_by(
|
||||||
|
Episode.series_id,
|
||||||
|
Episode.season,
|
||||||
|
Episode.episode_number,
|
||||||
|
)
|
||||||
|
.having(func.count(Episode.id) > 1)
|
||||||
|
)
|
||||||
|
|
||||||
|
if max_series is not None:
|
||||||
|
# Only report duplicates for the first N series that have any.
|
||||||
|
# Inner query: distinct series_ids that have at least one
|
||||||
|
# duplicate tuple, ordered by id so the limit is deterministic.
|
||||||
|
series_with_dupes = (
|
||||||
|
select(Episode.series_id)
|
||||||
|
.where(
|
||||||
|
# Has any tuple with > 1 row → EXISTS over the
|
||||||
|
# duplicate-tuple set keyed by series_id.
|
||||||
|
Episode.series_id.in_(
|
||||||
|
select(Episode.series_id)
|
||||||
|
.group_by(
|
||||||
|
Episode.series_id,
|
||||||
|
Episode.season,
|
||||||
|
Episode.episode_number,
|
||||||
|
)
|
||||||
|
.having(func.count(Episode.id) > 1)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
.group_by(Episode.series_id)
|
||||||
|
.order_by(Episode.series_id)
|
||||||
|
.limit(max_series)
|
||||||
|
.subquery()
|
||||||
|
)
|
||||||
|
duplicates_subquery = duplicates_subquery.where(
|
||||||
|
Episode.series_id.in_(select(series_with_dupes.c.series_id))
|
||||||
|
)
|
||||||
|
|
||||||
|
rows = (await _execute(duplicates_subquery)).all()
|
||||||
|
return [
|
||||||
|
(
|
||||||
|
int(r.series_id),
|
||||||
|
int(r.season),
|
||||||
|
int(r.episode_number),
|
||||||
|
int(r.row_count),
|
||||||
|
int(r.keep_id),
|
||||||
|
int(r.max_id),
|
||||||
|
)
|
||||||
|
for r in rows
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
async def delete_duplicate_episodes(duplicates: List[DuplicateTuple]) -> int:
|
||||||
|
"""Delete the duplicate rows for each tuple, keeping the lowest id.
|
||||||
|
|
||||||
|
Returns the number of rows actually deleted.
|
||||||
|
|
||||||
|
Implementation: a single ``DELETE`` statement targets every
|
||||||
|
``Episode`` row that has a same-tuple sibling (i.e. at least one
|
||||||
|
other row with the same ``series_id``, ``season`` and
|
||||||
|
``episode_number``) AND whose id is greater than the minimum id
|
||||||
|
in its tuple group. The min-id row per tuple is preserved (it
|
||||||
|
has the lowest primary key, i.e. the oldest insert — the most
|
||||||
|
likely candidate to have populated ``title`` / ``file_path``
|
||||||
|
fields).
|
||||||
|
|
||||||
|
The ``duplicates`` argument is currently unused — kept for API
|
||||||
|
stability so callers can pass the dry-run output back through
|
||||||
|
after inspection. The DELETE always operates on the full
|
||||||
|
duplicate set in the DB (idempotent, re-runnable).
|
||||||
|
"""
|
||||||
|
del duplicates # API stability; DELETE is self-contained.
|
||||||
|
|
||||||
|
# Subquery: every (series_id, season, ep_num) tuple with > 1 row.
|
||||||
|
dup_keys = (
|
||||||
|
select(
|
||||||
|
Episode.series_id.label("series_id"),
|
||||||
|
Episode.season.label("season"),
|
||||||
|
Episode.episode_number.label("episode_number"),
|
||||||
|
)
|
||||||
|
.group_by(
|
||||||
|
Episode.series_id,
|
||||||
|
Episode.season,
|
||||||
|
Episode.episode_number,
|
||||||
|
)
|
||||||
|
.having(func.count(Episode.id) > 1)
|
||||||
|
.subquery()
|
||||||
|
)
|
||||||
|
|
||||||
|
# Per-row min id for each tuple.
|
||||||
|
min_id_per_tuple = (
|
||||||
|
select(func.min(Episode.id).label("min_id"))
|
||||||
|
.group_by(
|
||||||
|
Episode.series_id,
|
||||||
|
Episode.season,
|
||||||
|
Episode.episode_number,
|
||||||
|
)
|
||||||
|
.having(func.count(Episode.id) > 1)
|
||||||
|
.subquery()
|
||||||
|
)
|
||||||
|
|
||||||
|
result = await _execute(
|
||||||
|
delete(Episode).where(
|
||||||
|
Episode.id.notin_(select(min_id_per_tuple.c.min_id)),
|
||||||
|
# Correlate the delete with the duplicate-tuples subquery.
|
||||||
|
# Use tuple IN to match all three columns.
|
||||||
|
tuple_(
|
||||||
|
Episode.series_id,
|
||||||
|
Episode.season,
|
||||||
|
Episode.episode_number,
|
||||||
|
).in_(
|
||||||
|
select(
|
||||||
|
dup_keys.c.series_id,
|
||||||
|
dup_keys.c.season,
|
||||||
|
dup_keys.c.episode_number,
|
||||||
|
)
|
||||||
|
),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
rowcount: int = getattr(result, "rowcount", 0) or 0
|
||||||
|
return rowcount
|
||||||
|
|
||||||
|
|
||||||
|
async def _execute(stmt):
|
||||||
|
"""Run a statement against the async session and return the
|
||||||
|
result. Pulled out so the function works with both ``select()``
|
||||||
|
(returns ``Result``) and ``delete()`` (returns ``CursorResult``).
|
||||||
|
"""
|
||||||
|
from src.server.database.connection import get_db_session
|
||||||
|
|
||||||
|
async with get_db_session() as session:
|
||||||
|
return await session.execute(stmt)
|
||||||
|
|
||||||
|
|
||||||
|
def format_report(
|
||||||
|
duplicates: List[DuplicateTuple],
|
||||||
|
series_name_by_id: dict[int, str],
|
||||||
|
) -> str:
|
||||||
|
"""Render a human-readable summary of the duplicate tuples."""
|
||||||
|
if not duplicates:
|
||||||
|
return "No duplicate episodes found."
|
||||||
|
|
||||||
|
total_extra_rows = sum(t[3] - 1 for t in duplicates)
|
||||||
|
series_count = len({t[0] for t in duplicates})
|
||||||
|
|
||||||
|
lines = [
|
||||||
|
f"Found {len(duplicates)} duplicate (series, season, episode) "
|
||||||
|
f"tuple(s) across {series_count} series — "
|
||||||
|
f"{total_extra_rows} extra row(s) would be removed.",
|
||||||
|
"",
|
||||||
|
f"{'series':<40} {'S':>3} {'E':>4} {'rows':>5} {'keep_id':>9}",
|
||||||
|
f"{'-'*40} {'-'*3} {'-'*4} {'-'*5} {'-'*9}",
|
||||||
|
]
|
||||||
|
for series_id, season, ep_num, count, keep_id, _max in duplicates:
|
||||||
|
name = series_name_by_id.get(series_id, f"#{series_id}")
|
||||||
|
if len(name) > 38:
|
||||||
|
name = name[:37] + "\u2026"
|
||||||
|
lines.append(
|
||||||
|
f"{name:<40} {season:>3} {ep_num:>4} {count:>5} {keep_id:>9}"
|
||||||
|
)
|
||||||
|
return "\n".join(lines)
|
||||||
|
|
||||||
|
|
||||||
|
async def _load_series_names(series_ids: List[int]) -> dict[int, str]:
|
||||||
|
from src.server.database.connection import get_db_session
|
||||||
|
|
||||||
|
if not series_ids:
|
||||||
|
return {}
|
||||||
|
async with get_db_session() as session:
|
||||||
|
rows = (
|
||||||
|
await session.execute(
|
||||||
|
select(AnimeSeries.id, AnimeSeries.name, AnimeSeries.key)
|
||||||
|
.where(AnimeSeries.id.in_(series_ids))
|
||||||
|
)
|
||||||
|
).all()
|
||||||
|
out: dict[int, str] = {}
|
||||||
|
for r in rows:
|
||||||
|
# Prefer display name, fall back to key.
|
||||||
|
out[int(r.id)] = (
|
||||||
|
(r.name or r.key or f"#{r.id}") if r else f"#{r.id}"
|
||||||
|
)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
async def run(apply: bool, max_series: int | None) -> int:
|
||||||
|
"""CLI entry point. Returns a shell exit code."""
|
||||||
|
try:
|
||||||
|
await init_db()
|
||||||
|
except Exception as exc:
|
||||||
|
logger.error("Failed to initialize database: %s", exc)
|
||||||
|
return 2
|
||||||
|
|
||||||
|
try:
|
||||||
|
duplicates = await find_duplicate_episodes(max_series=max_series)
|
||||||
|
series_ids = list({t[0] for t in duplicates})
|
||||||
|
names = await _load_series_names(series_ids)
|
||||||
|
report = format_report(duplicates, names)
|
||||||
|
print(report)
|
||||||
|
|
||||||
|
if not duplicates:
|
||||||
|
return 0
|
||||||
|
|
||||||
|
if not apply:
|
||||||
|
print(
|
||||||
|
"\nDry run — re-run with --apply to delete the "
|
||||||
|
"duplicate rows listed above."
|
||||||
|
)
|
||||||
|
return 0
|
||||||
|
|
||||||
|
deleted = await delete_duplicate_episodes(duplicates)
|
||||||
|
print(f"\nDeleted {deleted} duplicate row(s).")
|
||||||
|
return 0
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Cleanup failed")
|
||||||
|
return 1
|
||||||
|
finally:
|
||||||
|
await close_db()
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
parser = argparse.ArgumentParser(
|
||||||
|
description=(
|
||||||
|
"Find and (optionally) delete duplicate rows in the "
|
||||||
|
"``episodes`` table. Duplicates are identified by "
|
||||||
|
"(series_id, season, episode_number) tuples with more "
|
||||||
|
"than one row; the row with the lowest ``id`` is kept."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--apply",
|
||||||
|
action="store_true",
|
||||||
|
help="Actually delete the duplicate rows. Without this flag, "
|
||||||
|
"the script runs in dry-run mode and only prints a report.",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--max-series",
|
||||||
|
type=int,
|
||||||
|
default=None,
|
||||||
|
help="Limit the dry-run report to the first N series that "
|
||||||
|
"have duplicates. Useful for large libraries. No effect "
|
||||||
|
"with --apply (which always cleans everything).",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--log-level",
|
||||||
|
default="INFO",
|
||||||
|
choices=["DEBUG", "INFO", "WARNING", "ERROR"],
|
||||||
|
help="Python logging level (default: INFO).",
|
||||||
|
)
|
||||||
|
|
||||||
|
args = parser.parse_args()
|
||||||
|
logging.basicConfig(
|
||||||
|
level=getattr(logging, args.log_level),
|
||||||
|
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
|
||||||
|
)
|
||||||
|
|
||||||
|
return asyncio.run(run(apply=args.apply, max_series=args.max_series))
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
@@ -281,10 +281,13 @@ class SerieScanner:
|
|||||||
async def _sync_episodes_to_db(
|
async def _sync_episodes_to_db(
|
||||||
self, db, series_id: int, episode_dict: dict[int, list[int]]
|
self, db, series_id: int, episode_dict: dict[int, list[int]]
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Sync episodes to database, preserving downloaded flags.
|
"""Sync episodes to database.
|
||||||
|
|
||||||
Adds missing episodes, removes episodes no longer missing,
|
Adds missing episodes, removes episodes no longer missing
|
||||||
and preserves is_downloaded=True episodes.
|
(including those that were previously marked as downloaded:
|
||||||
|
once the scanner confirms the file is on disk, the row has
|
||||||
|
no further purpose and is deleted to keep the DB in sync
|
||||||
|
with the filesystem).
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
db: Async database session
|
db: Async database session
|
||||||
@@ -301,12 +304,6 @@ class SerieScanner:
|
|||||||
new_keys.add((season, ep_num))
|
new_keys.add((season, ep_num))
|
||||||
for (season, ep_num), ep in existing_map.items():
|
for (season, ep_num), ep in existing_map.items():
|
||||||
if (season, ep_num) not in new_keys:
|
if (season, ep_num) not in new_keys:
|
||||||
if ep.is_downloaded:
|
|
||||||
logger.debug(
|
|
||||||
"Preserving downloaded episode S%02dE%02d for series_id=%d",
|
|
||||||
season, ep_num, series_id
|
|
||||||
)
|
|
||||||
else:
|
|
||||||
await EpisodeService.delete_by_series(
|
await EpisodeService.delete_by_series(
|
||||||
db, series_id, season, ep_num
|
db, series_id, season, ep_num
|
||||||
)
|
)
|
||||||
@@ -779,18 +776,35 @@ class SerieScanner:
|
|||||||
|
|
||||||
# Create or update AnimeSeries in keyDict
|
# Create or update AnimeSeries in keyDict
|
||||||
if key in self.keyDict:
|
if key in self.keyDict:
|
||||||
# Update existing anime - rebuild episodeDict from episodes
|
# Update existing anime - rebuild episodeDict from the
|
||||||
|
# latest scan results. The previous implementation
|
||||||
|
# extended the existing list with ``missing_episodes``,
|
||||||
|
# which accumulated duplicates across rescans of the
|
||||||
|
# same series; the in-memory cache then propagated
|
||||||
|
# duplicates through ``_update_series_in_db`` and into
|
||||||
|
# the ``episodes`` table until the UNIQUE constraint
|
||||||
|
# was added. Replace, don't extend.
|
||||||
existing = self.keyDict[key]
|
existing = self.keyDict[key]
|
||||||
existing_ep_dict = existing.episodeDict
|
# Use ``dict.fromkeys`` to dedupe within a season, in
|
||||||
# Merge missing episodes
|
# case ``missing_episodes`` itself contains duplicate
|
||||||
|
# episode numbers from a buggy loader upstream.
|
||||||
|
rebuilt: dict = {}
|
||||||
for season, eps in missing_episodes.items():
|
for season, eps in missing_episodes.items():
|
||||||
if season not in existing_ep_dict:
|
seen: set = set()
|
||||||
existing_ep_dict[season] = []
|
cleaned: list = []
|
||||||
existing_ep_dict[season].extend(eps)
|
for ep_num in eps:
|
||||||
|
if ep_num in seen:
|
||||||
|
continue
|
||||||
|
seen.add(ep_num)
|
||||||
|
cleaned.append(ep_num)
|
||||||
|
if cleaned:
|
||||||
|
rebuilt[season] = cleaned
|
||||||
|
existing.episodeDict = rebuilt
|
||||||
|
existing.folder = folder
|
||||||
logger.debug(
|
logger.debug(
|
||||||
"Updated existing series %s with %d missing episodes",
|
"Updated existing series %s with %d missing episodes",
|
||||||
key,
|
key,
|
||||||
sum(len(eps) for eps in missing_episodes.values())
|
sum(len(eps) for eps in rebuilt.values()),
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
# Extract year from folder name if present, otherwise leave as None
|
# Extract year from folder name if present, otherwise leave as None
|
||||||
|
|||||||
@@ -15,7 +15,17 @@ from datetime import datetime, timezone
|
|||||||
from enum import Enum
|
from enum import Enum
|
||||||
from typing import Any, Dict, List, Optional
|
from typing import Any, Dict, List, Optional
|
||||||
|
|
||||||
from sqlalchemy import Boolean, DateTime, ForeignKey, Index, Integer, String, Text, func
|
from sqlalchemy import (
|
||||||
|
Boolean,
|
||||||
|
DateTime,
|
||||||
|
ForeignKey,
|
||||||
|
Index,
|
||||||
|
Integer,
|
||||||
|
String,
|
||||||
|
Text,
|
||||||
|
UniqueConstraint,
|
||||||
|
func,
|
||||||
|
)
|
||||||
from sqlalchemy.orm import Mapped, mapped_column, relationship, validates
|
from sqlalchemy.orm import Mapped, mapped_column, relationship, validates
|
||||||
|
|
||||||
from src.server.database.base import Base, TimestampMixin
|
from src.server.database.base import Base, TimestampMixin
|
||||||
@@ -195,22 +205,60 @@ class AnimeSeries(Base, TimestampMixin):
|
|||||||
"""Build episode dictionary from episodes relationship or private cache.
|
"""Build episode dictionary from episodes relationship or private cache.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Dictionary mapping season numbers to lists of episode numbers
|
Dictionary mapping season numbers to lists of episode numbers.
|
||||||
|
Each (season, episode_number) pair is guaranteed to appear at
|
||||||
|
most once across all seasons: the underlying episodes table
|
||||||
|
has no UNIQUE constraint on (series_id, season,
|
||||||
|
episode_number), so the relationship (and the legacy
|
||||||
|
``_episode_dict_cache`` set by loaders/scanners) can contain
|
||||||
|
duplicates from historical scans. Duplicates are filtered
|
||||||
|
here at the read boundary so the rest of the stack can rely
|
||||||
|
on the dict being canonical.
|
||||||
"""
|
"""
|
||||||
# Check for private cache first (set when loading from JSON without DB)
|
# Check for private cache first (set when loading from JSON without DB)
|
||||||
if hasattr(self, '_episode_dict_cache') and self._episode_dict_cache is not None:
|
if hasattr(self, '_episode_dict_cache') and self._episode_dict_cache is not None:
|
||||||
return self._episode_dict_cache
|
cached = self._episode_dict_cache
|
||||||
|
# Dedupe the cached dict too: callers that populate the cache
|
||||||
|
# (legacy JSON loader, SerieScanner.scan_single_series for
|
||||||
|
# new series) may store values that contain duplicates.
|
||||||
|
seen: set[tuple[int, int]] = set()
|
||||||
|
deduped: dict[int, list[int]] = {}
|
||||||
|
for season, ep_nums in (cached or {}).items():
|
||||||
|
cleaned: list[int] = []
|
||||||
|
for ep_num in ep_nums:
|
||||||
|
if (season, ep_num) in seen:
|
||||||
|
continue
|
||||||
|
seen.add((season, ep_num))
|
||||||
|
cleaned.append(ep_num)
|
||||||
|
if cleaned:
|
||||||
|
deduped[season] = cleaned
|
||||||
|
return deduped
|
||||||
|
|
||||||
episode_dict: dict[int, list[int]] = {}
|
episode_dict: dict[int, list[int]] = {}
|
||||||
try:
|
try:
|
||||||
if self.episodes:
|
if self.episodes:
|
||||||
|
seen: set[tuple[int, int]] = set()
|
||||||
for ep in self.episodes:
|
for ep in self.episodes:
|
||||||
if ep.is_downloaded:
|
if ep.is_downloaded:
|
||||||
continue
|
continue
|
||||||
season = ep.season or 1
|
season = ep.season or 1
|
||||||
|
ep_num = ep.episode_number or 0
|
||||||
|
# Dedupe by (season, ep_num): the episodes table has
|
||||||
|
# no UNIQUE constraint on (series_id, season,
|
||||||
|
# episode_number), so the relationship can yield
|
||||||
|
# duplicate rows from historical scans. Without
|
||||||
|
# this guard, the dict exposes duplicates to the
|
||||||
|
# frontend, which forwards them verbatim to the
|
||||||
|
# queue API — every duplicate gets rejected by the
|
||||||
|
# backend's pending-episode dedup, leaving the
|
||||||
|
# user with an empty queue and a misleading
|
||||||
|
# "Added N" toast.
|
||||||
|
if (season, ep_num) in seen:
|
||||||
|
continue
|
||||||
|
seen.add((season, ep_num))
|
||||||
if season not in episode_dict:
|
if season not in episode_dict:
|
||||||
episode_dict[season] = []
|
episode_dict[season] = []
|
||||||
episode_dict[season].append(ep.episode_number or 0)
|
episode_dict[season].append(ep_num)
|
||||||
except Exception:
|
except Exception:
|
||||||
# DetachedInstanceError or other DB errors - return empty dict
|
# DetachedInstanceError or other DB errors - return empty dict
|
||||||
# This can happen when accessing episodes on a newly created
|
# This can happen when accessing episodes on a newly created
|
||||||
@@ -291,6 +339,24 @@ class Episode(Base, TimestampMixin):
|
|||||||
"""
|
"""
|
||||||
__tablename__ = "episodes"
|
__tablename__ = "episodes"
|
||||||
|
|
||||||
|
# Table-level constraints. The UNIQUE constraint on
|
||||||
|
# (series_id, season, episode_number) is the schema-level guard
|
||||||
|
# against the duplicate-row pathology: every (series, season,
|
||||||
|
# episode) tuple can have at most one row. Rescans that try to
|
||||||
|
# create a duplicate row will fail at the DB layer rather than
|
||||||
|
# silently accumulating rows. Defense-in-depth on top of the
|
||||||
|
# write-side dedup in SerieScanner._sync_episodes_to_db and
|
||||||
|
# AnimeService._update_series_in_db, and the read-boundary dedup
|
||||||
|
# in AnimeSeries.episodeDict.
|
||||||
|
__table_args__ = (
|
||||||
|
UniqueConstraint(
|
||||||
|
"series_id",
|
||||||
|
"season",
|
||||||
|
"episode_number",
|
||||||
|
name="uq_episode_per_series_season",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
# Primary key
|
# Primary key
|
||||||
id: Mapped[int] = mapped_column(
|
id: Mapped[int] = mapped_column(
|
||||||
Integer, primary_key=True, autoincrement=True
|
Integer, primary_key=True, autoincrement=True
|
||||||
|
|||||||
@@ -861,9 +861,13 @@ class AnimeService:
|
|||||||
Syncs the database episodes with the current missing episodes from scan.
|
Syncs the database episodes with the current missing episodes from scan.
|
||||||
- Adds new missing episodes that are not in the database
|
- Adds new missing episodes that are not in the database
|
||||||
- Removes episodes from database that are no longer missing
|
- Removes episodes from database that are no longer missing
|
||||||
(i.e., the file has been added to the filesystem)
|
(i.e., the file has been added to the filesystem), including
|
||||||
- Preserves episodes marked as downloaded (is_downloaded=True)
|
episodes that were previously marked as downloaded. A row
|
||||||
so download history is not lost
|
that is no longer missing — by definition — does not need to
|
||||||
|
stay in the DB; the UI derives "missing" from the row's
|
||||||
|
presence, so keeping an ``is_downloaded=True`` row around
|
||||||
|
leaves a stale entry that the user can see in the DB but
|
||||||
|
not anywhere else.
|
||||||
"""
|
"""
|
||||||
from src.server.database.service import AnimeSeriesService, EpisodeService
|
from src.server.database.service import AnimeSeriesService, EpisodeService
|
||||||
|
|
||||||
@@ -871,15 +875,11 @@ class AnimeService:
|
|||||||
existing_episodes = await EpisodeService.get_by_series(db, existing.id)
|
existing_episodes = await EpisodeService.get_by_series(db, existing.id)
|
||||||
|
|
||||||
# Build dict of existing episodes: {season: {ep_num: episode_id}}
|
# Build dict of existing episodes: {season: {ep_num: episode_id}}
|
||||||
# and track which ones are already downloaded
|
|
||||||
existing_dict: dict[int, dict[int, int]] = {}
|
existing_dict: dict[int, dict[int, int]] = {}
|
||||||
downloaded_set: set[tuple[int, int]] = set()
|
|
||||||
for ep in existing_episodes:
|
for ep in existing_episodes:
|
||||||
if ep.season not in existing_dict:
|
if ep.season not in existing_dict:
|
||||||
existing_dict[ep.season] = {}
|
existing_dict[ep.season] = {}
|
||||||
existing_dict[ep.season][ep.episode_number] = ep.id
|
existing_dict[ep.season][ep.episode_number] = ep.id
|
||||||
if ep.is_downloaded:
|
|
||||||
downloaded_set.add((ep.season, ep.episode_number))
|
|
||||||
|
|
||||||
# Get new missing episodes from scan
|
# Get new missing episodes from scan
|
||||||
new_dict = serie.episodeDict or {}
|
new_dict = serie.episodeDict or {}
|
||||||
@@ -909,23 +909,14 @@ class AnimeService:
|
|||||||
)
|
)
|
||||||
|
|
||||||
# Remove episodes from database that are no longer missing
|
# Remove episodes from database that are no longer missing
|
||||||
# (i.e., the episode file now exists on the filesystem)
|
# (i.e., the episode file now exists on the filesystem).
|
||||||
# BUT: preserve episodes that are already downloaded (is_downloaded=True)
|
# This includes episodes previously marked as downloaded:
|
||||||
# so we don't lose download history
|
# once the file is confirmed on disk by a rescan, the row
|
||||||
|
# has no further purpose and is deleted to keep the DB
|
||||||
|
# in sync with the filesystem.
|
||||||
for season, eps_dict in existing_dict.items():
|
for season, eps_dict in existing_dict.items():
|
||||||
for ep_num, episode_id in eps_dict.items():
|
for ep_num, episode_id in eps_dict.items():
|
||||||
if (season, ep_num) not in new_missing_set:
|
if (season, ep_num) not in new_missing_set:
|
||||||
# Skip already-downloaded episodes — they should stay in DB
|
|
||||||
# with is_downloaded=True to preserve download history
|
|
||||||
if (season, ep_num) in downloaded_set:
|
|
||||||
logger.debug(
|
|
||||||
"Preserving downloaded episode in database: "
|
|
||||||
"%s S%02dE%02d",
|
|
||||||
serie.key,
|
|
||||||
season,
|
|
||||||
ep_num
|
|
||||||
)
|
|
||||||
continue
|
|
||||||
await EpisodeService.delete(db, episode_id)
|
await EpisodeService.delete(db, episode_id)
|
||||||
logger.info(
|
logger.info(
|
||||||
"Removed episode from database (no longer missing): "
|
"Removed episode from database (no longer missing): "
|
||||||
|
|||||||
@@ -253,8 +253,24 @@ AniWorld.SelectionManager = (function() {
|
|||||||
console.error('Validation errors:', JSON.stringify(data.detail, null, 2));
|
console.error('Validation errors:', JSON.stringify(data.detail, null, 2));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Trust the server's response, not the input count: the
|
||||||
|
// backend returns success even when zero episodes were
|
||||||
|
// added (e.g. all duplicates), and the input `episodes`
|
||||||
|
// array can itself contain duplicates from a stale
|
||||||
|
// in-memory episodeDict. Counting `data.added_items`
|
||||||
|
// gives the user an accurate "Added N" toast.
|
||||||
if (response.ok && data.status === 'success') {
|
if (response.ok && data.status === 'success') {
|
||||||
totalEpisodesAdded += episodes.length;
|
const addedThisRequest = Array.isArray(data.added_items)
|
||||||
|
? data.added_items.length
|
||||||
|
: 0;
|
||||||
|
totalEpisodesAdded += addedThisRequest;
|
||||||
|
if (addedThisRequest === 0 && episodes.length > 0) {
|
||||||
|
console.warn(
|
||||||
|
'Queue add returned 0 items for',
|
||||||
|
key,
|
||||||
|
'— all episodes may be duplicates of an existing pending entry.'
|
||||||
|
);
|
||||||
|
}
|
||||||
} else {
|
} else {
|
||||||
console.error('Failed to add to queue:', data);
|
console.error('Failed to add to queue:', data);
|
||||||
failedSeries.push(key);
|
failedSeries.push(key);
|
||||||
|
|||||||
@@ -966,6 +966,121 @@ class TestSaveAndLoadDB:
|
|||||||
mock_create.assert_called_once()
|
mock_create.assert_called_once()
|
||||||
assert mock_ep_create.call_count == 2
|
assert mock_ep_create.call_count == 2
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_update_series_deletes_downloaded_episodes_when_no_longer_missing(
|
||||||
|
self, anime_service
|
||||||
|
):
|
||||||
|
"""Regression: a finished download marks the Episode row with
|
||||||
|
is_downloaded=True. When a later rescan confirms the file is on
|
||||||
|
disk (i.e. the episode is no longer in the missing set), the
|
||||||
|
DB row should be deleted so it stops appearing in queries.
|
||||||
|
|
||||||
|
Bug shape: previously the ``downloaded_set`` guard in
|
||||||
|
``_update_series_in_db`` preserved the row forever, so the DB
|
||||||
|
kept stale ``is_downloaded=True`` entries that the user could
|
||||||
|
see in the database but not in any UI.
|
||||||
|
"""
|
||||||
|
mock_serie = MagicMock()
|
||||||
|
mock_serie.key = "the-100-girlfriends"
|
||||||
|
mock_serie.name = "The 100 Girlfriends"
|
||||||
|
mock_serie.site = "aniworld.to"
|
||||||
|
mock_serie.folder = "The 100 Girlfriends (2023)"
|
||||||
|
# Scanner reports no missing episodes for this series —
|
||||||
|
# every file is on disk.
|
||||||
|
mock_serie.episodeDict = {}
|
||||||
|
|
||||||
|
existing = MagicMock()
|
||||||
|
existing.id = 1
|
||||||
|
existing.folder = "The 100 Girlfriends (2023)"
|
||||||
|
|
||||||
|
# DB currently has one row for S03E08 marked as downloaded
|
||||||
|
# (the result of an earlier successful download). The
|
||||||
|
# scanner confirms the file is on disk, so the episode is
|
||||||
|
# no longer missing.
|
||||||
|
existing_eps = [
|
||||||
|
MagicMock(
|
||||||
|
id=10, season=3, episode_number=8, is_downloaded=True,
|
||||||
|
),
|
||||||
|
]
|
||||||
|
|
||||||
|
mock_session = AsyncMock()
|
||||||
|
|
||||||
|
with patch(
|
||||||
|
"src.server.database.service.EpisodeService.get_by_series",
|
||||||
|
new_callable=AsyncMock,
|
||||||
|
return_value=existing_eps,
|
||||||
|
), patch(
|
||||||
|
"src.server.database.service.EpisodeService.delete",
|
||||||
|
new_callable=AsyncMock,
|
||||||
|
) as mock_delete:
|
||||||
|
await anime_service._update_series_in_db(
|
||||||
|
mock_serie, existing, mock_session
|
||||||
|
)
|
||||||
|
|
||||||
|
# The downloaded episode (S03E08) MUST be deleted — the file
|
||||||
|
# is on disk and the scanner does not report it as missing.
|
||||||
|
# EpisodeService.delete is (db, episode_id) — episode_id is
|
||||||
|
# the second positional arg.
|
||||||
|
deleted_ids = [
|
||||||
|
call.args[1] for call in mock_delete.call_args_list
|
||||||
|
]
|
||||||
|
assert deleted_ids == [10], (
|
||||||
|
f"Expected S03E08 (id=10) to be the only deleted row; "
|
||||||
|
f"got delete calls for {deleted_ids}"
|
||||||
|
)
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_update_series_keeps_still_missing_episodes(
|
||||||
|
self, anime_service
|
||||||
|
):
|
||||||
|
"""A still-missing episode (is_downloaded=False, in scanner's
|
||||||
|
missing set) must NOT be deleted by _update_series_in_db.
|
||||||
|
Sanity-check sibling to the downloaded-episode regression
|
||||||
|
test, ensuring the fix does not over-reach.
|
||||||
|
"""
|
||||||
|
mock_serie = MagicMock()
|
||||||
|
mock_serie.key = "naruto"
|
||||||
|
mock_serie.name = "Naruto"
|
||||||
|
mock_serie.site = "aniworld.to"
|
||||||
|
mock_serie.folder = "Naruto"
|
||||||
|
# Scanner reports S01E07 still missing.
|
||||||
|
mock_serie.episodeDict = {1: [7]}
|
||||||
|
|
||||||
|
existing = MagicMock()
|
||||||
|
existing.id = 1
|
||||||
|
existing.folder = "Naruto"
|
||||||
|
|
||||||
|
existing_eps = [
|
||||||
|
MagicMock(
|
||||||
|
id=20, season=1, episode_number=7, is_downloaded=False,
|
||||||
|
),
|
||||||
|
]
|
||||||
|
|
||||||
|
mock_session = AsyncMock()
|
||||||
|
|
||||||
|
with patch(
|
||||||
|
"src.server.database.service.EpisodeService.get_by_series",
|
||||||
|
new_callable=AsyncMock,
|
||||||
|
return_value=existing_eps,
|
||||||
|
), patch(
|
||||||
|
"src.server.database.service.EpisodeService.delete",
|
||||||
|
new_callable=AsyncMock,
|
||||||
|
) as mock_delete, patch(
|
||||||
|
"src.server.database.service.EpisodeService.create",
|
||||||
|
new_callable=AsyncMock,
|
||||||
|
):
|
||||||
|
await anime_service._update_series_in_db(
|
||||||
|
mock_serie, existing, mock_session
|
||||||
|
)
|
||||||
|
|
||||||
|
# S01E07 is still missing per the scanner — it must NOT be
|
||||||
|
# deleted. (No new episode needs to be created either — it
|
||||||
|
# is already in the DB.)
|
||||||
|
assert mock_delete.call_count == 0, (
|
||||||
|
f"Still-missing episode S01E07 must not be deleted; "
|
||||||
|
f"got delete calls: {mock_delete.call_args_list}"
|
||||||
|
)
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_save_scan_results_updates_existing(
|
async def test_save_scan_results_updates_existing(
|
||||||
self, anime_service
|
self, anime_service
|
||||||
|
|||||||
343
tests/unit/test_clean_duplicate_episodes_cli.py
Normal file
343
tests/unit/test_clean_duplicate_episodes_cli.py
Normal file
@@ -0,0 +1,343 @@
|
|||||||
|
"""Unit tests for the ``clean_duplicate_episodes`` CLI.
|
||||||
|
|
||||||
|
Exercises the core logic (find / delete) against an in-memory
|
||||||
|
SQLite engine so the test is hermetic. The CLI module reads the
|
||||||
|
global async session factory from ``src.server.database.connection``
|
||||||
|
— the test patches that factory with an in-memory engine.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from typing import List
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from sqlalchemy import select, text
|
||||||
|
from sqlalchemy.ext.asyncio import (
|
||||||
|
AsyncEngine,
|
||||||
|
AsyncSession,
|
||||||
|
async_sessionmaker,
|
||||||
|
create_async_engine,
|
||||||
|
)
|
||||||
|
from sqlalchemy.pool import StaticPool
|
||||||
|
|
||||||
|
from src.cli import clean_duplicate_episodes as cli
|
||||||
|
from src.server.database import connection as conn_module
|
||||||
|
from src.server.database.base import Base
|
||||||
|
from src.server.database.models import AnimeSeries, Episode
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
async def in_memory_engine():
|
||||||
|
"""Provide an in-memory async SQLite engine with the schema
|
||||||
|
already created, and patch the connection module's session
|
||||||
|
factory to use it for the duration of the test.
|
||||||
|
|
||||||
|
The CLI test suite simulates the *pre-migration* state — a DB
|
||||||
|
that predates the UNIQUE constraint on
|
||||||
|
``(series_id, season, episode_number)`` and has accumulated
|
||||||
|
duplicate rows from historical scans. To create that state, we
|
||||||
|
build the schema for everything except ``episodes``, then
|
||||||
|
recreate ``episodes`` with raw DDL that omits the
|
||||||
|
``uq_episode_per_series_season`` constraint. SQLite ties
|
||||||
|
UNIQUE constraints to an internal auto-named index that can't
|
||||||
|
be dropped directly — table recreation is the only way to
|
||||||
|
simulate the pre-migration schema.
|
||||||
|
"""
|
||||||
|
engine: AsyncEngine = create_async_engine(
|
||||||
|
"sqlite+aiosqlite:///:memory:",
|
||||||
|
echo=False,
|
||||||
|
poolclass=StaticPool,
|
||||||
|
)
|
||||||
|
async with engine.begin() as conn:
|
||||||
|
# Create everything except episodes.
|
||||||
|
await conn.run_sync(
|
||||||
|
lambda sync_conn: [
|
||||||
|
t.create(sync_conn)
|
||||||
|
for t in Base.metadata.sorted_tables
|
||||||
|
if t.name != "episodes"
|
||||||
|
]
|
||||||
|
)
|
||||||
|
# Recreate episodes without the UNIQUE constraint.
|
||||||
|
await conn.execute(
|
||||||
|
text(
|
||||||
|
"""
|
||||||
|
CREATE TABLE episodes (
|
||||||
|
id INTEGER NOT NULL PRIMARY KEY,
|
||||||
|
series_id INTEGER NOT NULL
|
||||||
|
REFERENCES anime_series(id) ON DELETE CASCADE,
|
||||||
|
season INTEGER NOT NULL,
|
||||||
|
episode_number INTEGER NOT NULL,
|
||||||
|
title VARCHAR(500),
|
||||||
|
file_path VARCHAR(1000),
|
||||||
|
is_downloaded BOOLEAN NOT NULL DEFAULT 0,
|
||||||
|
created_at DATETIME DEFAULT CURRENT_TIMESTAMP NOT NULL,
|
||||||
|
updated_at DATETIME DEFAULT CURRENT_TIMESTAMP NOT NULL
|
||||||
|
)
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
)
|
||||||
|
await conn.execute(
|
||||||
|
text("CREATE INDEX ix_episodes_series_id ON episodes (series_id)")
|
||||||
|
)
|
||||||
|
|
||||||
|
factory = async_sessionmaker(
|
||||||
|
bind=engine,
|
||||||
|
class_=AsyncSession,
|
||||||
|
expire_on_commit=False,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Monkey-patch the global session factory.
|
||||||
|
original_factory = conn_module._session_factory
|
||||||
|
conn_module._session_factory = factory
|
||||||
|
try:
|
||||||
|
yield engine
|
||||||
|
finally:
|
||||||
|
conn_module._session_factory = original_factory
|
||||||
|
await engine.dispose()
|
||||||
|
|
||||||
|
|
||||||
|
async def _add_series(session: AsyncSession, key: str) -> int:
|
||||||
|
series = AnimeSeries(
|
||||||
|
key=key,
|
||||||
|
name=key.replace("-", " ").title(),
|
||||||
|
site="https://aniworld.to",
|
||||||
|
folder=f"/anime/{key}",
|
||||||
|
)
|
||||||
|
session.add(series)
|
||||||
|
await session.commit()
|
||||||
|
await session.refresh(series)
|
||||||
|
return int(series.id)
|
||||||
|
|
||||||
|
|
||||||
|
async def _add_episode(
|
||||||
|
session: AsyncSession,
|
||||||
|
series_id: int,
|
||||||
|
season: int,
|
||||||
|
ep_num: int,
|
||||||
|
is_downloaded: bool = False,
|
||||||
|
title: str | None = None,
|
||||||
|
) -> int:
|
||||||
|
"""Insert an Episode row. The test fixture drops the UNIQUE
|
||||||
|
constraint on the ``episodes`` table, so duplicate inserts
|
||||||
|
succeed — the cleanup tool can then find them, mirroring the
|
||||||
|
pre-migration pathology it's meant to repair."""
|
||||||
|
ep = Episode(
|
||||||
|
series_id=series_id,
|
||||||
|
season=season,
|
||||||
|
episode_number=ep_num,
|
||||||
|
is_downloaded=is_downloaded,
|
||||||
|
title=title,
|
||||||
|
)
|
||||||
|
session.add(ep)
|
||||||
|
await session.commit()
|
||||||
|
await session.refresh(ep)
|
||||||
|
return int(ep.id)
|
||||||
|
|
||||||
|
|
||||||
|
def _tup(d):
|
||||||
|
"""Convert a SQLAlchemy row to the (sid, season, ep_num, count,
|
||||||
|
keep_id, max_id) tuple shape the CLI uses."""
|
||||||
|
return (
|
||||||
|
int(d.series_id),
|
||||||
|
int(d.season),
|
||||||
|
int(d.episode_number),
|
||||||
|
int(d.row_count),
|
||||||
|
int(d.keep_id),
|
||||||
|
int(d.max_id),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_find_returns_empty_when_no_duplicates(in_memory_engine):
|
||||||
|
"""With one row per (series, season, ep_num), nothing is found."""
|
||||||
|
async with AsyncSession(in_memory_engine) as session:
|
||||||
|
sid = await _add_series(session, "no-dupes")
|
||||||
|
await _add_episode(session, sid, 1, 1)
|
||||||
|
await _add_episode(session, sid, 1, 2)
|
||||||
|
await _add_episode(session, sid, 2, 1)
|
||||||
|
|
||||||
|
duplicates = await cli.find_duplicate_episodes()
|
||||||
|
assert duplicates == []
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_find_groups_duplicates_per_tuple(in_memory_engine):
|
||||||
|
"""Three copies of (S1, E2) collapse to one DuplicateTuple with
|
||||||
|
count=3 and keep_id = the lowest id."""
|
||||||
|
async with AsyncSession(in_memory_engine) as session:
|
||||||
|
sid = await _add_series(session, "triple")
|
||||||
|
await _add_episode(session, sid, 1, 1)
|
||||||
|
# Three rows for (S1, E2): the first via the normal path,
|
||||||
|
# the next two via raw SQL to bypass the UNIQUE constraint.
|
||||||
|
# Mirrors the pre-migration pathology the cleanup tool exists
|
||||||
|
# to repair.
|
||||||
|
e2 = await _add_episode(session, sid, 1, 2)
|
||||||
|
e3 = await _add_episode(session, sid, 1, 2)
|
||||||
|
e4 = await _add_episode(session, sid, 1, 2)
|
||||||
|
await _add_episode(session, sid, 1, 3)
|
||||||
|
|
||||||
|
duplicates = await cli.find_duplicate_episodes()
|
||||||
|
|
||||||
|
# Only one duplicate tuple (S1, E2); (S1, E1) and (S1, E3) are
|
||||||
|
# unique and must not appear.
|
||||||
|
assert len(duplicates) == 1
|
||||||
|
d = duplicates[0]
|
||||||
|
assert (d[0], d[1], d[2]) == (sid, 1, 2)
|
||||||
|
assert d[3] == 3
|
||||||
|
assert d[4] == e2 # lowest id wins
|
||||||
|
assert d[5] == e4 # max id (for the report)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_delete_keeps_lowest_id_per_tuple(in_memory_engine):
|
||||||
|
"""The DELETE call preserves the lowest-id row per duplicate
|
||||||
|
tuple and removes the rest."""
|
||||||
|
async with AsyncSession(in_memory_engine) as session:
|
||||||
|
sid = await _add_series(session, "keep-lowest")
|
||||||
|
e1 = await _add_episode(session, sid, 1, 1) # unique
|
||||||
|
e2 = await _add_episode(session, sid, 1, 2) # lowest of dupes
|
||||||
|
await _add_episode(
|
||||||
|
session, sid, 1, 2
|
||||||
|
) # e3 - duplicate of e2
|
||||||
|
await _add_episode(
|
||||||
|
session, sid, 1, 2
|
||||||
|
) # e4 - duplicate of e2
|
||||||
|
e5 = await _add_episode(session, sid, 1, 3) # unique
|
||||||
|
|
||||||
|
duplicates = await cli.find_duplicate_episodes()
|
||||||
|
assert len(duplicates) == 1
|
||||||
|
|
||||||
|
deleted = await cli.delete_duplicate_episodes(duplicates)
|
||||||
|
assert deleted == 2 # two duplicate (S1, E2) rows removed
|
||||||
|
|
||||||
|
# Verify the post-cleanup state directly via SQLAlchemy.
|
||||||
|
async with AsyncSession(in_memory_engine) as session:
|
||||||
|
rows = (
|
||||||
|
await session.execute(
|
||||||
|
select(Episode).order_by(Episode.id)
|
||||||
|
)
|
||||||
|
).scalars().all()
|
||||||
|
remaining_ids = [int(r.id) for r in rows]
|
||||||
|
# e1, e2 (lowest of the dup group), and e5 survive; e3, e4
|
||||||
|
# were the duplicates and are gone.
|
||||||
|
assert remaining_ids == [e1, e2, e5]
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_delete_keeps_lowest_even_with_title_metadata(in_memory_engine):
|
||||||
|
"""When a higher-id row has populated metadata but the lowest-id
|
||||||
|
row is empty, the lowest-id row is still kept — that's the
|
||||||
|
documented behavior (oldest insert wins)."""
|
||||||
|
async with AsyncSession(in_memory_engine) as session:
|
||||||
|
sid = await _add_series(session, "metadata")
|
||||||
|
e1 = await _add_episode(session, sid, 1, 1) # no title
|
||||||
|
e2 = await _add_episode(
|
||||||
|
session,
|
||||||
|
sid,
|
||||||
|
1,
|
||||||
|
1,
|
||||||
|
title="Better Episode 1",
|
||||||
|
)
|
||||||
|
|
||||||
|
duplicates = await cli.find_duplicate_episodes()
|
||||||
|
assert len(duplicates) == 1
|
||||||
|
|
||||||
|
await cli.delete_duplicate_episodes(duplicates)
|
||||||
|
|
||||||
|
async with AsyncSession(in_memory_engine) as session:
|
||||||
|
rows = (
|
||||||
|
await session.execute(select(Episode))
|
||||||
|
).scalars().all()
|
||||||
|
assert len(rows) == 1
|
||||||
|
assert int(rows[0].id) == e1
|
||||||
|
assert rows[0].title is None
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_full_workflow_is_idempotent(in_memory_engine):
|
||||||
|
"""Running find -> delete -> find yields nothing the second
|
||||||
|
time. Mirrors the production usage pattern."""
|
||||||
|
async with AsyncSession(in_memory_engine) as session:
|
||||||
|
sid = await _add_series(session, "idempotent")
|
||||||
|
for ep_num in (1, 2, 3):
|
||||||
|
await _add_episode(session, sid, 1, ep_num)
|
||||||
|
await _add_episode(
|
||||||
|
session, sid, 1, ep_num
|
||||||
|
)
|
||||||
|
await _add_episode(
|
||||||
|
session, sid, 1, ep_num
|
||||||
|
)
|
||||||
|
|
||||||
|
first = await cli.find_duplicate_episodes()
|
||||||
|
assert len(first) == 3
|
||||||
|
|
||||||
|
await cli.delete_duplicate_episodes(first)
|
||||||
|
second = await cli.find_duplicate_episodes()
|
||||||
|
assert second == []
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_max_series_filter_limits_report(in_memory_engine):
|
||||||
|
"""``--max-series`` caps which series appear in the report but
|
||||||
|
the underlying find still returns every duplicate for those
|
||||||
|
series."""
|
||||||
|
async with AsyncSession(in_memory_engine) as session:
|
||||||
|
s1 = await _add_series(session, "alpha")
|
||||||
|
s2 = await _add_series(session, "beta")
|
||||||
|
s3 = await _add_series(session, "gamma")
|
||||||
|
for sid in (s1, s2, s3):
|
||||||
|
for ep_num in (1, 2):
|
||||||
|
await _add_episode(session, sid, 1, ep_num)
|
||||||
|
await _add_episode(
|
||||||
|
session, sid, 1, ep_num
|
||||||
|
)
|
||||||
|
|
||||||
|
# Without filter: all three series have duplicates.
|
||||||
|
full = await cli.find_duplicate_episodes()
|
||||||
|
series_in_full = {t[0] for t in full}
|
||||||
|
assert series_in_full == {s1, s2, s3}
|
||||||
|
|
||||||
|
# With --max-series=2: only two series appear.
|
||||||
|
limited = await cli.find_duplicate_episodes(max_series=2)
|
||||||
|
series_in_limited = {t[0] for t in limited}
|
||||||
|
assert len(series_in_limited) == 2
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_format_report_human_readable(in_memory_engine):
|
||||||
|
"""The report includes the series name, season, episode, count
|
||||||
|
and keep_id for each duplicate tuple."""
|
||||||
|
async with AsyncSession(in_memory_engine) as session:
|
||||||
|
sid = await _add_series(session, "attack-on-titan")
|
||||||
|
await _add_episode(session, sid, 1, 1)
|
||||||
|
await _add_episode(session, sid, 1, 1)
|
||||||
|
|
||||||
|
duplicates = await cli.find_duplicate_episodes()
|
||||||
|
names = await cli._load_series_names([sid])
|
||||||
|
|
||||||
|
report = cli.format_report(duplicates, names)
|
||||||
|
|
||||||
|
assert "Attack On Titan" in report or "Attack-on-Titan" in report
|
||||||
|
# The row's keep_id column should appear, plus a count of "2".
|
||||||
|
assert "2" in report
|
||||||
|
assert "found" in report.lower()
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_format_report_handles_no_duplicates():
|
||||||
|
"""Empty input yields a friendly 'nothing to do' message."""
|
||||||
|
report = cli.format_report([], {})
|
||||||
|
assert "no duplicate" in report.lower()
|
||||||
|
|
||||||
|
|
||||||
|
def test_argparse_defaults():
|
||||||
|
"""The CLI defaults to dry-run when --apply is not given."""
|
||||||
|
import argparse
|
||||||
|
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("--apply", action="store_true")
|
||||||
|
parser.add_argument("--max-series", type=int, default=None)
|
||||||
|
|
||||||
|
# Simulate argv without --apply.
|
||||||
|
args = parser.parse_args([])
|
||||||
|
assert args.apply is False
|
||||||
|
assert args.max_series is None
|
||||||
@@ -308,6 +308,185 @@ class TestAnimeSeries:
|
|||||||
assert len(with_tmdb) == 2
|
assert len(with_tmdb) == 2
|
||||||
|
|
||||||
|
|
||||||
|
class TestEpisodeDictDedup:
|
||||||
|
"""Regression tests for the ``AnimeSeries.episodeDict`` dedup.
|
||||||
|
|
||||||
|
The ``episodes`` table has a UNIQUE constraint on
|
||||||
|
``(series_id, season, episode_number)`` (added as the schema-level
|
||||||
|
prevention in commit f75d591..), so duplicate rows cannot be
|
||||||
|
created via normal write paths. The property's defensive dedup
|
||||||
|
still matters for two reasons:
|
||||||
|
|
||||||
|
1. DBs that predate the UNIQUE constraint may have stale duplicate
|
||||||
|
rows from historical scans (visible in the user's backup DB
|
||||||
|
before clean_duplicate_episodes was run).
|
||||||
|
2. ``_episode_dict_cache`` is populated directly by scanners and
|
||||||
|
loaders, which can carry duplicates from their internal logic
|
||||||
|
(e.g. ``scan_single_series`` previously extended the dict on
|
||||||
|
every rescan).
|
||||||
|
|
||||||
|
These tests cover both paths. The class uses its own engine
|
||||||
|
fixture that drops the UNIQUE constraint after schema creation,
|
||||||
|
so the duplicate-row tests can set up the pre-migration state
|
||||||
|
the property's dedup is meant to defend against.
|
||||||
|
"""
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def legacy_engine(self):
|
||||||
|
"""In-memory SQLite engine without the UNIQUE constraint
|
||||||
|
on episodes — simulates a pre-migration DB.
|
||||||
|
|
||||||
|
SQLite ties UNIQUE constraints to an internal auto-named
|
||||||
|
index that can't be dropped directly, so we rebuild the
|
||||||
|
episodes table with raw DDL that omits the
|
||||||
|
``uq_episode_per_series_season`` constraint. The rest of
|
||||||
|
the schema comes from ``Base.metadata.create_all``.
|
||||||
|
"""
|
||||||
|
from sqlalchemy import text
|
||||||
|
|
||||||
|
engine = create_engine("sqlite:///:memory:", echo=False)
|
||||||
|
# Create everything except the episodes table.
|
||||||
|
for table in Base.metadata.sorted_tables:
|
||||||
|
if table.name != "episodes":
|
||||||
|
table.create(engine)
|
||||||
|
# Recreate episodes without the UNIQUE constraint.
|
||||||
|
with engine.begin() as conn:
|
||||||
|
conn.execute(
|
||||||
|
text(
|
||||||
|
"""
|
||||||
|
CREATE TABLE episodes (
|
||||||
|
id INTEGER NOT NULL PRIMARY KEY,
|
||||||
|
series_id INTEGER NOT NULL
|
||||||
|
REFERENCES anime_series(id) ON DELETE CASCADE,
|
||||||
|
season INTEGER NOT NULL,
|
||||||
|
episode_number INTEGER NOT NULL,
|
||||||
|
title VARCHAR(500),
|
||||||
|
file_path VARCHAR(1000),
|
||||||
|
is_downloaded BOOLEAN NOT NULL DEFAULT 0,
|
||||||
|
created_at DATETIME DEFAULT CURRENT_TIMESTAMP NOT NULL,
|
||||||
|
updated_at DATETIME DEFAULT CURRENT_TIMESTAMP NOT NULL
|
||||||
|
)
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
)
|
||||||
|
conn.execute(
|
||||||
|
text("CREATE INDEX ix_episodes_series_id ON episodes (series_id)")
|
||||||
|
)
|
||||||
|
SessionLocal = sessionmaker(bind=engine)
|
||||||
|
session = SessionLocal()
|
||||||
|
yield session
|
||||||
|
session.close()
|
||||||
|
engine.dispose()
|
||||||
|
|
||||||
|
def _make_series(self, session: Session, key: str) -> AnimeSeries:
|
||||||
|
series = AnimeSeries(
|
||||||
|
key=key,
|
||||||
|
name=key.replace("-", " ").title(),
|
||||||
|
site="https://aniworld.to",
|
||||||
|
folder=f"/anime/{key}",
|
||||||
|
)
|
||||||
|
session.add(series)
|
||||||
|
session.commit()
|
||||||
|
return series
|
||||||
|
|
||||||
|
def test_episodeDict_dedupes_duplicate_relationship_rows(
|
||||||
|
self, legacy_engine: Session
|
||||||
|
):
|
||||||
|
"""Duplicate Episode rows for the same series — including
|
||||||
|
ones that pre-date the UNIQUE constraint — must not
|
||||||
|
duplicate the entries in ``episodeDict``."""
|
||||||
|
session = legacy_engine
|
||||||
|
series = self._make_series(session, "dedup-rel")
|
||||||
|
|
||||||
|
# Three duplicate rows for (S1, E3) — exactly the pattern
|
||||||
|
# the scanner accumulated across repeated rescans in the
|
||||||
|
# pre-migration era. With the constraint dropped in the
|
||||||
|
# legacy_engine fixture, these inserts succeed.
|
||||||
|
for _ in range(3):
|
||||||
|
session.add(
|
||||||
|
Episode(series_id=series.id, season=1, episode_number=3)
|
||||||
|
)
|
||||||
|
# One row each for the surrounding unique episodes.
|
||||||
|
for ep in (1, 2, 4):
|
||||||
|
session.add(
|
||||||
|
Episode(series_id=series.id, season=1, episode_number=ep)
|
||||||
|
)
|
||||||
|
session.commit()
|
||||||
|
|
||||||
|
result = series.episodeDict
|
||||||
|
|
||||||
|
# Order depends on SQLAlchemy row order, which is not strictly
|
||||||
|
# insertion order — only the *set* of episodes matters here.
|
||||||
|
assert set(result.keys()) == {1}
|
||||||
|
assert set(result[1]) == {1, 2, 3, 4}
|
||||||
|
# Defensive: no season has duplicate episode numbers.
|
||||||
|
for season, eps in result.items():
|
||||||
|
assert len(eps) == len(set(eps)), (
|
||||||
|
f"season {season} has duplicate episode numbers: {eps}"
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_episodeDict_excludes_downloaded_episodes(
|
||||||
|
self, db_session: Session
|
||||||
|
):
|
||||||
|
"""is_downloaded rows must still be filtered out."""
|
||||||
|
series = self._make_series(db_session, "dedup-downloaded")
|
||||||
|
|
||||||
|
db_session.add(
|
||||||
|
Episode(series_id=series.id, season=1, episode_number=1)
|
||||||
|
)
|
||||||
|
# Downloaded row at (S1, E2): must be filtered out.
|
||||||
|
db_session.add(
|
||||||
|
Episode(
|
||||||
|
series_id=series.id,
|
||||||
|
season=1,
|
||||||
|
episode_number=2,
|
||||||
|
is_downloaded=True,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
db_session.commit()
|
||||||
|
|
||||||
|
assert series.episodeDict == {1: [1]}
|
||||||
|
|
||||||
|
def test_episodeDict_dedupes_cached_value(self, db_session: Session):
|
||||||
|
"""The legacy ``_episode_dict_cache`` path (set directly by
|
||||||
|
scanners and loaders) must also dedupe, since loaders can
|
||||||
|
populate the cache with duplicated data — historically
|
||||||
|
``scan_single_series`` extended the dict on every rescan."""
|
||||||
|
series = self._make_series(db_session, "dedup-cache")
|
||||||
|
|
||||||
|
# No episodes in the DB at all — the property will fall back
|
||||||
|
# to the cache.
|
||||||
|
series._episode_dict_cache = {1: [3, 3, 3, 4, 4]}
|
||||||
|
|
||||||
|
assert series.episodeDict == {1: [3, 4]}
|
||||||
|
|
||||||
|
def test_episodeDict_preserves_unique_entries_across_seasons(
|
||||||
|
self, db_session: Session
|
||||||
|
):
|
||||||
|
"""Dedup must be per-(season, episode_number), not
|
||||||
|
per-episode_number alone — same ep number in different
|
||||||
|
seasons is legitimate and must be preserved."""
|
||||||
|
series = self._make_series(db_session, "multi-season")
|
||||||
|
|
||||||
|
for season, ep in [(1, 1), (1, 2), (2, 1), (2, 2)]:
|
||||||
|
db_session.add(
|
||||||
|
Episode(
|
||||||
|
series_id=series.id,
|
||||||
|
season=season,
|
||||||
|
episode_number=ep,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
db_session.commit()
|
||||||
|
|
||||||
|
result = series.episodeDict
|
||||||
|
|
||||||
|
# Order is not guaranteed across SQLAlchemy relationships;
|
||||||
|
# compare as sets.
|
||||||
|
assert set(result.keys()) == {1, 2}
|
||||||
|
assert set(result[1]) == {1, 2}
|
||||||
|
assert set(result[2]) == {1, 2}
|
||||||
|
|
||||||
|
|
||||||
class TestEpisode:
|
class TestEpisode:
|
||||||
"""Test cases for Episode model."""
|
"""Test cases for Episode model."""
|
||||||
|
|
||||||
|
|||||||
@@ -194,12 +194,20 @@ class TestSerieScannerSingleSeries:
|
|||||||
def test_scan_single_series_existing_entry(
|
def test_scan_single_series_existing_entry(
|
||||||
self, temp_directory, mock_loader, sample_serie
|
self, temp_directory, mock_loader, sample_serie
|
||||||
):
|
):
|
||||||
"""Test scan_single_series updates existing entry in keyDict."""
|
"""Test scan_single_series replaces the existing entry's
|
||||||
|
``episodeDict`` with the new scan's missing-episode list.
|
||||||
|
|
||||||
|
Note: the previous implementation ``extend````ed the
|
||||||
|
existing list with the new one, which accumulated
|
||||||
|
duplicates across rescans. The fix is to replace, not
|
||||||
|
extend — see ``test_serie_scanner_scan_dedup.py`` for the
|
||||||
|
regression tests for that specific bug.
|
||||||
|
"""
|
||||||
scanner = SerieScanner(temp_directory, mock_loader)
|
scanner = SerieScanner(temp_directory, mock_loader)
|
||||||
|
|
||||||
# Pre-populate keyDict
|
# Pre-populate keyDict
|
||||||
scanner.keyDict[sample_serie.key] = sample_serie
|
scanner.keyDict[sample_serie.key] = sample_serie
|
||||||
# Use deepcopy because episodeDict is modified in-place
|
# Use deepcopy because episodeDict is mutated by the scanner.
|
||||||
import copy
|
import copy
|
||||||
old_episode_dict = copy.deepcopy(sample_serie.episodeDict)
|
old_episode_dict = copy.deepcopy(sample_serie.episodeDict)
|
||||||
|
|
||||||
@@ -213,10 +221,15 @@ class TestSerieScannerSingleSeries:
|
|||||||
folder=sample_serie.folder
|
folder=sample_serie.folder
|
||||||
)
|
)
|
||||||
|
|
||||||
# Verify existing entry was updated - episodeDict is merged (not replaced)
|
# The cached episodeDict is REPLACED with the latest
|
||||||
# Old episodes [2, 3, 4] + new episodes [10, 11, 12] = merged result
|
# scan's missing-episode list — not merged. Old entries
|
||||||
assert scanner.keyDict[sample_serie.key].episodeDict != old_episode_dict
|
# ([2, 3, 4]) are dropped because the latest scan
|
||||||
assert scanner.keyDict[sample_serie.key].episodeDict == {1: [2, 3, 4, 10, 11, 12]}
|
# reports only [10, 11, 12] as still missing.
|
||||||
|
new_episode_dict = scanner.keyDict[
|
||||||
|
sample_serie.key
|
||||||
|
].episodeDict
|
||||||
|
assert new_episode_dict != old_episode_dict
|
||||||
|
assert new_episode_dict == {1: [10, 11, 12]}
|
||||||
|
|
||||||
def test_scan_single_series_empty_key_raises_error(
|
def test_scan_single_series_empty_key_raises_error(
|
||||||
self, temp_directory, mock_loader
|
self, temp_directory, mock_loader
|
||||||
|
|||||||
@@ -101,12 +101,20 @@ class TestSyncEpisodesToDb:
|
|||||||
"""Test _sync_episodes_to_db method."""
|
"""Test _sync_episodes_to_db method."""
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_preserves_downloaded_episodes(self):
|
async def test_deletes_downloaded_episodes_when_no_longer_missing(self):
|
||||||
"""Verify downloaded episodes are not removed even when no longer missing."""
|
"""Downloaded episodes are deleted once the rescan confirms the
|
||||||
|
file is on disk and they are no longer missing. The DB stays
|
||||||
|
in sync with the filesystem: a row that is not in the scanner's
|
||||||
|
missing set has no further purpose and is removed.
|
||||||
|
"""
|
||||||
mock_session = AsyncMock()
|
mock_session = AsyncMock()
|
||||||
|
|
||||||
# S01E1 was downloaded (file exists), S01E2 was missing but file now exists
|
# S01E1 was downloaded (file exists) and the scanner confirms
|
||||||
# Both are no longer in episode_dict
|
# it is no longer missing; S01E2 was previously marked as
|
||||||
|
# downloaded and is also no longer missing. Both should be
|
||||||
|
# deleted — there is no notion of "preserving download history"
|
||||||
|
# in the DB: the rescan's filesystem view is the source of
|
||||||
|
# truth, and a row whose episode is no longer missing is dead.
|
||||||
existing_eps = [
|
existing_eps = [
|
||||||
MagicMock(id=1, season=1, episode_number=1, is_downloaded=True),
|
MagicMock(id=1, season=1, episode_number=1, is_downloaded=True),
|
||||||
MagicMock(id=2, season=1, episode_number=2, is_downloaded=True),
|
MagicMock(id=2, season=1, episode_number=2, is_downloaded=True),
|
||||||
@@ -126,8 +134,16 @@ class TestSyncEpisodesToDb:
|
|||||||
mock_session, 1, {} # No episodes missing
|
mock_session, 1, {} # No episodes missing
|
||||||
)
|
)
|
||||||
|
|
||||||
# Neither should be deleted since both are downloaded
|
# Both downloaded rows should be deleted; the scanner
|
||||||
mock_delete.assert_not_called()
|
# found the files on disk and they're not in the
|
||||||
|
# missing set.
|
||||||
|
assert mock_delete.call_count == 2
|
||||||
|
deleted_calls = [
|
||||||
|
(c.args[1], c.args[2], c.args[3])
|
||||||
|
for c in mock_delete.call_args_list
|
||||||
|
]
|
||||||
|
assert (1, 1, 1) in deleted_calls
|
||||||
|
assert (1, 1, 2) in deleted_calls
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_removes_missing_episodes_when_no_longer_missing(self):
|
async def test_removes_missing_episodes_when_no_longer_missing(self):
|
||||||
|
|||||||
179
tests/unit/test_serie_scanner_scan_dedup.py
Normal file
179
tests/unit/test_serie_scanner_scan_dedup.py
Normal file
@@ -0,0 +1,179 @@
|
|||||||
|
"""Regression test for ``SerieScanner.scan_single_series``.
|
||||||
|
|
||||||
|
The previous implementation ``extend````ed the in-memory
|
||||||
|
``episodeDict`` for series already present in the scanner's
|
||||||
|
``keyDict`` — every rescan of the same series appended the new
|
||||||
|
missing-episode list on top of the existing one, so the dict grew
|
||||||
|
with duplicates across rescans. Those duplicates then propagated
|
||||||
|
through ``_update_series_in_db`` into the ``episodes`` table.
|
||||||
|
|
||||||
|
This test exercises the real ``scan_single_series`` method (not a
|
||||||
|
mock) to assert that two rescans of the same series produce a
|
||||||
|
canonical (deduplicated) ``episodeDict``, not an accumulated one.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from pathlib import Path
|
||||||
|
from types import SimpleNamespace
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from src.server.SerieScanner import SerieScanner
|
||||||
|
|
||||||
|
|
||||||
|
class _StubLoader:
|
||||||
|
"""A minimal loader that returns whatever ``missing_episodes``
|
||||||
|
the test wants. Replaces the real loader on a
|
||||||
|
``SerieScanner`` instance via ``monkeypatch.setattr``."""
|
||||||
|
|
||||||
|
def __init__(self, missing_per_call: list[dict]):
|
||||||
|
self._missing_per_call = list(missing_per_call)
|
||||||
|
self._call_index = 0
|
||||||
|
|
||||||
|
def get_season_episode_count(self, key):
|
||||||
|
# The real loader returns the total episode count per
|
||||||
|
# season; ``scan_single_series`` doesn't actually use it
|
||||||
|
# (it calls ``__get_missing_episodes_and_season`` directly),
|
||||||
|
# but the stub keeps the call site safe.
|
||||||
|
max_seen = 0
|
||||||
|
for m in self._missing_per_call:
|
||||||
|
for season, eps in m.items():
|
||||||
|
if eps:
|
||||||
|
max_seen = max(max_seen, max(eps))
|
||||||
|
return {1: max_seen or 1}
|
||||||
|
|
||||||
|
def is_language(self, season, ep, key):
|
||||||
|
return True
|
||||||
|
|
||||||
|
def next_missing(self) -> dict:
|
||||||
|
if self._call_index >= len(self._missing_per_call):
|
||||||
|
return {}
|
||||||
|
result = self._missing_per_call[self._call_index]
|
||||||
|
self._call_index += 1
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def _make_scanner(
|
||||||
|
tmp_path: Path,
|
||||||
|
missing_per_call: list[dict],
|
||||||
|
) -> tuple[SerieScanner, _StubLoader]:
|
||||||
|
"""Build a real ``SerieScanner`` with the private
|
||||||
|
``__get_missing_episodes_and_season`` method replaced by a
|
||||||
|
stub that returns the next ``missing_episodes`` dict on each
|
||||||
|
call. Everything else (events, directory) is stubbed so the
|
||||||
|
test runs without filesystem or scheduler setup."""
|
||||||
|
scanner = SerieScanner.__new__(SerieScanner)
|
||||||
|
loader = _StubLoader(missing_per_call)
|
||||||
|
scanner.loader = loader # type: ignore[assignment]
|
||||||
|
scanner.keyDict = {}
|
||||||
|
# ``self.directory`` is the attribute ``scan_single_series``
|
||||||
|
# reads at line 734. ``scan_single_series`` checks
|
||||||
|
# ``os.path.isdir(folder_path)`` and, if the folder does not
|
||||||
|
# exist, treats the scan as "no MP4 files on disk". Use a path
|
||||||
|
# whose subdirectories do not exist so the scan takes the
|
||||||
|
# empty-mp4-files branch without us having to populate any
|
||||||
|
# filesystem state.
|
||||||
|
scanner.directory = str(tmp_path)
|
||||||
|
scanner.directory_to_search = tmp_path
|
||||||
|
scanner.events = SimpleNamespace(
|
||||||
|
on_progress=lambda *a, **k: None,
|
||||||
|
on_completion=lambda *a, **k: None,
|
||||||
|
on_error=lambda *a, **k: None,
|
||||||
|
)
|
||||||
|
|
||||||
|
def fake_get_missing_episodes_and_season(key, mp4_files):
|
||||||
|
return loader.next_missing(), "aniworld.to"
|
||||||
|
|
||||||
|
scanner._SerieScanner__get_missing_episodes_and_season = ( # type: ignore[attr-defined]
|
||||||
|
fake_get_missing_episodes_and_season
|
||||||
|
)
|
||||||
|
return scanner, loader
|
||||||
|
|
||||||
|
|
||||||
|
def test_scan_single_series_replaces_not_extends(tmp_path):
|
||||||
|
"""Two rescans of the same series with the same missing
|
||||||
|
episodes must produce a canonical (deduplicated) episodeDict —
|
||||||
|
not a list with duplicates from accumulation."""
|
||||||
|
canonical = {1: [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12]}
|
||||||
|
scanner, _ = _make_scanner(
|
||||||
|
tmp_path,
|
||||||
|
# Same missing-episode list reported on each scan.
|
||||||
|
missing_per_call=[canonical, canonical],
|
||||||
|
)
|
||||||
|
|
||||||
|
# First scan: key not in keyDict, so the else branch fires
|
||||||
|
# and the cache is set to the missing-episode list.
|
||||||
|
scanner.keyDict.clear()
|
||||||
|
result_first = scanner.scan_single_series(
|
||||||
|
key="erased", folder="Erased"
|
||||||
|
)
|
||||||
|
assert result_first == canonical
|
||||||
|
cached = scanner.keyDict["erased"].episodeDict
|
||||||
|
assert cached == canonical
|
||||||
|
|
||||||
|
# Second scan: key IS in keyDict, so the if branch fires.
|
||||||
|
# Before the fix, this would extend the cached list with
|
||||||
|
# [1..12] again, producing {1: [1..12, 1..12]}. After the
|
||||||
|
# fix, the cache is replaced, not extended.
|
||||||
|
result_second = scanner.scan_single_series(
|
||||||
|
key="erased", folder="Erased"
|
||||||
|
)
|
||||||
|
assert result_second == canonical
|
||||||
|
cached_second = scanner.keyDict["erased"].episodeDict
|
||||||
|
assert cached_second == canonical, (
|
||||||
|
"second scan extended the dict instead of replacing it: "
|
||||||
|
f"{cached_second}"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Defensive: the dict has no duplicates within any season.
|
||||||
|
for season, eps in cached_second.items():
|
||||||
|
assert len(eps) == len(set(eps)), (
|
||||||
|
f"season {season} has duplicate episode numbers: {eps}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_scan_single_series_resets_when_becomes_complete(tmp_path):
|
||||||
|
"""When a rescan finds the series is complete (no missing
|
||||||
|
episodes), the episodeDict must be empty — extending the
|
||||||
|
previous dict would leave stale entries behind."""
|
||||||
|
scanner, _ = _make_scanner(
|
||||||
|
tmp_path,
|
||||||
|
# First scan: missing 1..12. Second scan: nothing missing.
|
||||||
|
missing_per_call=[
|
||||||
|
{1: [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12]},
|
||||||
|
{},
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
scanner.keyDict.clear()
|
||||||
|
scanner.scan_single_series(key="complete-me", folder="Complete Me")
|
||||||
|
assert scanner.keyDict["complete-me"].episodeDict == {
|
||||||
|
1: [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12]
|
||||||
|
}
|
||||||
|
|
||||||
|
scanner.scan_single_series(key="complete-me", folder="Complete Me")
|
||||||
|
assert scanner.keyDict["complete-me"].episodeDict == {}, (
|
||||||
|
"rescan with no missing episodes must reset the dict, "
|
||||||
|
f"got: {scanner.keyDict['complete-me'].episodeDict}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_scan_single_series_dedupes_within_a_single_call(tmp_path):
|
||||||
|
"""If the loader itself returns duplicate episode numbers in
|
||||||
|
``missing_episodes`` (a buggy upstream loader), the scanner
|
||||||
|
must still produce a canonical dict — defense in depth on top
|
||||||
|
of the loader and the read-boundary dedup in
|
||||||
|
``AnimeSeries.episodeDict``."""
|
||||||
|
scanner, _ = _make_scanner(
|
||||||
|
tmp_path,
|
||||||
|
# The loader returns the same episode numbers multiple
|
||||||
|
# times within a single call.
|
||||||
|
missing_per_call=[{1: [1, 1, 2, 2, 3, 3, 3, 4]}],
|
||||||
|
)
|
||||||
|
|
||||||
|
scanner.keyDict.clear()
|
||||||
|
scanner.scan_single_series(key="buggy", folder="Buggy")
|
||||||
|
cached = scanner.keyDict["buggy"].episodeDict
|
||||||
|
assert cached == {1: [1, 2, 3, 4]}, (
|
||||||
|
f"single-scan dedup failed: {cached}"
|
||||||
|
)
|
||||||
Reference in New Issue
Block a user