Files
Aniworld/tests/unit/test_clean_duplicate_episodes_cli.py
Lukas c3aca9217d fix(episodes): prevent duplicate-row accumulation at schema + write sites
Followup to commit c84f968 (read-boundary dedup) and commit f75d591
(cleanup CLI). The read boundary filters duplicates out of the
in-memory episodeDict and the CLI cleans up historical duplicates
in the DB, but the underlying pathology — duplicate rows being
created in the first place — was still active on every rescan.

Two layered prevention fixes:

1. Schema-level guard: add UNIQUE(series_id, season, episode_number)
   to the episodes table. SQLite's CREATE UNIQUE INDEX requires
   no existing duplicates, but the cleanup CLI from f75d591 has
   already been run (or is a one-shot prerequisite for users on
   older DBs). Future duplicate rows are rejected at the DB layer.

2. Write-site guard: SerieScanner.scan_single_series used to
   `extend` the in-memory episodeDict on every rescan of a
   series already in keyDict — across N rescans, the same missing
   list was appended N times, growing the dict with duplicates that
   then flowed through _update_series_in_db into the episodes
   table. The fix replaces the cache with the latest scan result
   instead of extending, and dedupes within a single call as
   defense in depth against a buggy upstream loader.

Defensive dedup is layered three deep:
  - schema constraint (this commit, primary)
  - scan_single_series replace-not-extend (this commit, secondary)
  - episodeDict property read-boundary dedup (commit c84f968,
    tertiary — covers legacy DBs that predate the constraint)

Tests:
  - Updated test_serie_scanner.test_scan_single_series_existing_entry
    to assert the new replace-not-merge behavior (the old assertion
    encoded the buggy extend behavior).
  - New test_serie_scanner_scan_dedup.py covers the regression
    directly: two rescans of the same series with the same missing
    list must yield a canonical dict, not an accumulated one.
  - test_database_models and test_clean_duplicate_episodes_cli now
    use a legacy_engine fixture that drops the UNIQUE constraint,
    so the duplicate-row scenarios they exercise (the read-boundary
    dedup and the cleanup tool, both meant to defend against
    pre-migration state) can still be tested under the new schema.

Verified manually: clean_duplicate_episodes --apply on the user's
backup DB still removes all 633 duplicate rows under the new
schema (the CLI doesn't depend on the UNIQUE constraint — it
operates on whatever rows already exist).
2026-09-15 20:58:30 +02:00

343 lines
12 KiB
Python

"""Unit tests for the ``clean_duplicate_episodes`` CLI.
Exercises the core logic (find / delete) against an in-memory
SQLite engine so the test is hermetic. The CLI module reads the
global async session factory from ``src.server.database.connection``
— the test patches that factory with an in-memory engine.
"""
from __future__ import annotations
from typing import List
import pytest
from sqlalchemy import select, text
from sqlalchemy.ext.asyncio import (
AsyncEngine,
AsyncSession,
async_sessionmaker,
create_async_engine,
)
from sqlalchemy.pool import StaticPool
from src.cli import clean_duplicate_episodes as cli
from src.server.database import connection as conn_module
from src.server.database.base import Base
from src.server.database.models import AnimeSeries, Episode
@pytest.fixture
async def in_memory_engine():
"""Provide an in-memory async SQLite engine with the schema
already created, and patch the connection module's session
factory to use it for the duration of the test.
The CLI test suite simulates the *pre-migration* state — a DB
that predates the UNIQUE constraint on
``(series_id, season, episode_number)`` and has accumulated
duplicate rows from historical scans. To create that state, we
build the schema for everything except ``episodes``, then
recreate ``episodes`` with raw DDL that omits the
``uq_episode_per_series_season`` constraint. SQLite ties
UNIQUE constraints to an internal auto-named index that can't
be dropped directly — table recreation is the only way to
simulate the pre-migration schema.
"""
engine: AsyncEngine = create_async_engine(
"sqlite+aiosqlite:///:memory:",
echo=False,
poolclass=StaticPool,
)
async with engine.begin() as conn:
# Create everything except episodes.
await conn.run_sync(
lambda sync_conn: [
t.create(sync_conn)
for t in Base.metadata.sorted_tables
if t.name != "episodes"
]
)
# Recreate episodes without the UNIQUE constraint.
await conn.execute(
text(
"""
CREATE TABLE episodes (
id INTEGER NOT NULL PRIMARY KEY,
series_id INTEGER NOT NULL
REFERENCES anime_series(id) ON DELETE CASCADE,
season INTEGER NOT NULL,
episode_number INTEGER NOT NULL,
title VARCHAR(500),
file_path VARCHAR(1000),
is_downloaded BOOLEAN NOT NULL DEFAULT 0,
created_at DATETIME DEFAULT CURRENT_TIMESTAMP NOT NULL,
updated_at DATETIME DEFAULT CURRENT_TIMESTAMP NOT NULL
)
"""
)
)
await conn.execute(
text("CREATE INDEX ix_episodes_series_id ON episodes (series_id)")
)
factory = async_sessionmaker(
bind=engine,
class_=AsyncSession,
expire_on_commit=False,
)
# Monkey-patch the global session factory.
original_factory = conn_module._session_factory
conn_module._session_factory = factory
try:
yield engine
finally:
conn_module._session_factory = original_factory
await engine.dispose()
async def _add_series(session: AsyncSession, key: str) -> int:
series = AnimeSeries(
key=key,
name=key.replace("-", " ").title(),
site="https://aniworld.to",
folder=f"/anime/{key}",
)
session.add(series)
await session.commit()
await session.refresh(series)
return int(series.id)
async def _add_episode(
session: AsyncSession,
series_id: int,
season: int,
ep_num: int,
is_downloaded: bool = False,
title: str | None = None,
) -> int:
"""Insert an Episode row. The test fixture drops the UNIQUE
constraint on the ``episodes`` table, so duplicate inserts
succeed — the cleanup tool can then find them, mirroring the
pre-migration pathology it's meant to repair."""
ep = Episode(
series_id=series_id,
season=season,
episode_number=ep_num,
is_downloaded=is_downloaded,
title=title,
)
session.add(ep)
await session.commit()
await session.refresh(ep)
return int(ep.id)
def _tup(d):
"""Convert a SQLAlchemy row to the (sid, season, ep_num, count,
keep_id, max_id) tuple shape the CLI uses."""
return (
int(d.series_id),
int(d.season),
int(d.episode_number),
int(d.row_count),
int(d.keep_id),
int(d.max_id),
)
@pytest.mark.asyncio
async def test_find_returns_empty_when_no_duplicates(in_memory_engine):
"""With one row per (series, season, ep_num), nothing is found."""
async with AsyncSession(in_memory_engine) as session:
sid = await _add_series(session, "no-dupes")
await _add_episode(session, sid, 1, 1)
await _add_episode(session, sid, 1, 2)
await _add_episode(session, sid, 2, 1)
duplicates = await cli.find_duplicate_episodes()
assert duplicates == []
@pytest.mark.asyncio
async def test_find_groups_duplicates_per_tuple(in_memory_engine):
"""Three copies of (S1, E2) collapse to one DuplicateTuple with
count=3 and keep_id = the lowest id."""
async with AsyncSession(in_memory_engine) as session:
sid = await _add_series(session, "triple")
await _add_episode(session, sid, 1, 1)
# Three rows for (S1, E2): the first via the normal path,
# the next two via raw SQL to bypass the UNIQUE constraint.
# Mirrors the pre-migration pathology the cleanup tool exists
# to repair.
e2 = await _add_episode(session, sid, 1, 2)
e3 = await _add_episode(session, sid, 1, 2)
e4 = await _add_episode(session, sid, 1, 2)
await _add_episode(session, sid, 1, 3)
duplicates = await cli.find_duplicate_episodes()
# Only one duplicate tuple (S1, E2); (S1, E1) and (S1, E3) are
# unique and must not appear.
assert len(duplicates) == 1
d = duplicates[0]
assert (d[0], d[1], d[2]) == (sid, 1, 2)
assert d[3] == 3
assert d[4] == e2 # lowest id wins
assert d[5] == e4 # max id (for the report)
@pytest.mark.asyncio
async def test_delete_keeps_lowest_id_per_tuple(in_memory_engine):
"""The DELETE call preserves the lowest-id row per duplicate
tuple and removes the rest."""
async with AsyncSession(in_memory_engine) as session:
sid = await _add_series(session, "keep-lowest")
e1 = await _add_episode(session, sid, 1, 1) # unique
e2 = await _add_episode(session, sid, 1, 2) # lowest of dupes
await _add_episode(
session, sid, 1, 2
) # e3 - duplicate of e2
await _add_episode(
session, sid, 1, 2
) # e4 - duplicate of e2
e5 = await _add_episode(session, sid, 1, 3) # unique
duplicates = await cli.find_duplicate_episodes()
assert len(duplicates) == 1
deleted = await cli.delete_duplicate_episodes(duplicates)
assert deleted == 2 # two duplicate (S1, E2) rows removed
# Verify the post-cleanup state directly via SQLAlchemy.
async with AsyncSession(in_memory_engine) as session:
rows = (
await session.execute(
select(Episode).order_by(Episode.id)
)
).scalars().all()
remaining_ids = [int(r.id) for r in rows]
# e1, e2 (lowest of the dup group), and e5 survive; e3, e4
# were the duplicates and are gone.
assert remaining_ids == [e1, e2, e5]
@pytest.mark.asyncio
async def test_delete_keeps_lowest_even_with_title_metadata(in_memory_engine):
"""When a higher-id row has populated metadata but the lowest-id
row is empty, the lowest-id row is still kept — that's the
documented behavior (oldest insert wins)."""
async with AsyncSession(in_memory_engine) as session:
sid = await _add_series(session, "metadata")
e1 = await _add_episode(session, sid, 1, 1) # no title
e2 = await _add_episode(
session,
sid,
1,
1,
title="Better Episode 1",
)
duplicates = await cli.find_duplicate_episodes()
assert len(duplicates) == 1
await cli.delete_duplicate_episodes(duplicates)
async with AsyncSession(in_memory_engine) as session:
rows = (
await session.execute(select(Episode))
).scalars().all()
assert len(rows) == 1
assert int(rows[0].id) == e1
assert rows[0].title is None
@pytest.mark.asyncio
async def test_full_workflow_is_idempotent(in_memory_engine):
"""Running find -> delete -> find yields nothing the second
time. Mirrors the production usage pattern."""
async with AsyncSession(in_memory_engine) as session:
sid = await _add_series(session, "idempotent")
for ep_num in (1, 2, 3):
await _add_episode(session, sid, 1, ep_num)
await _add_episode(
session, sid, 1, ep_num
)
await _add_episode(
session, sid, 1, ep_num
)
first = await cli.find_duplicate_episodes()
assert len(first) == 3
await cli.delete_duplicate_episodes(first)
second = await cli.find_duplicate_episodes()
assert second == []
@pytest.mark.asyncio
async def test_max_series_filter_limits_report(in_memory_engine):
"""``--max-series`` caps which series appear in the report but
the underlying find still returns every duplicate for those
series."""
async with AsyncSession(in_memory_engine) as session:
s1 = await _add_series(session, "alpha")
s2 = await _add_series(session, "beta")
s3 = await _add_series(session, "gamma")
for sid in (s1, s2, s3):
for ep_num in (1, 2):
await _add_episode(session, sid, 1, ep_num)
await _add_episode(
session, sid, 1, ep_num
)
# Without filter: all three series have duplicates.
full = await cli.find_duplicate_episodes()
series_in_full = {t[0] for t in full}
assert series_in_full == {s1, s2, s3}
# With --max-series=2: only two series appear.
limited = await cli.find_duplicate_episodes(max_series=2)
series_in_limited = {t[0] for t in limited}
assert len(series_in_limited) == 2
@pytest.mark.asyncio
async def test_format_report_human_readable(in_memory_engine):
"""The report includes the series name, season, episode, count
and keep_id for each duplicate tuple."""
async with AsyncSession(in_memory_engine) as session:
sid = await _add_series(session, "attack-on-titan")
await _add_episode(session, sid, 1, 1)
await _add_episode(session, sid, 1, 1)
duplicates = await cli.find_duplicate_episodes()
names = await cli._load_series_names([sid])
report = cli.format_report(duplicates, names)
assert "Attack On Titan" in report or "Attack-on-Titan" in report
# The row's keep_id column should appear, plus a count of "2".
assert "2" in report
assert "found" in report.lower()
@pytest.mark.asyncio
async def test_format_report_handles_no_duplicates():
"""Empty input yields a friendly 'nothing to do' message."""
report = cli.format_report([], {})
assert "no duplicate" in report.lower()
def test_argparse_defaults():
"""The CLI defaults to dry-run when --apply is not given."""
import argparse
parser = argparse.ArgumentParser()
parser.add_argument("--apply", action="store_true")
parser.add_argument("--max-series", type=int, default=None)
# Simulate argv without --apply.
args = parser.parse_args([])
assert args.apply is False
assert args.max_series is None