Files
dl-organizer/tests/test_enrichment.py
T

423 lines
13 KiB
Python
Raw Normal View History

"""Unit tests for enrichment pipeline."""
2026-02-13 13:36:39 +08:00
import threading
import time
import pytest
from vlm.config import Config
from vlm.enrichment import _build_display_title, _build_providers, enrich_identities_data
from vlm.providers.base import ProviderResult
from vlm.providers.tmdb import TMDBAuthError
class DummyProvider:
name = "dummy"
def __init__(self):
self.calls = 0
self.last_request_count = 0
def enrich(self, *, title: str, media_type: str, year=None):
self.calls += 1
self.last_request_count = 1
return ProviderResult(
provider="dummy",
canonical_id=f"dummy:{title}",
title_zh="测试中文名",
title_en="Test English Title",
translation_source="dummy",
reputation_score=8.2,
reputation_votes=100,
reputation_source="dummy",
)
def test_enrich_incremental_cache_hit(tmp_path, monkeypatch):
"""Second enrichment run should hit cache when fingerprint is unchanged."""
config = Config(
library_root=tmp_path,
enrichment_cache_db=tmp_path / "cache.db",
enrichment_providers=["dummy"],
translation_fallback_machine=False,
)
provider = DummyProvider()
monkeypatch.setattr(
"vlm.enrichment._build_providers",
lambda _config, request_timeout, retries: [provider],
)
identities = {
"metadata": {},
"movies": [
{
"path": "/library/movie/The.Matrix.1999.mkv",
"filename": "The.Matrix.1999.mkv",
"category": "movie",
"title": "The Matrix",
"year": 1999,
"confidence": 0.9,
"needs_review": False,
}
],
"series": [],
"anime": [],
"other": [],
}
_, stats_first = enrich_identities_data(identities, config)
assert stats_first["enriched"] == 1
assert stats_first["cache_hits"] == 0
assert stats_first["api_calls"] == 1
assert provider.calls == 1
_, stats_second = enrich_identities_data(identities, config)
assert stats_second["cache_hits"] == 1
assert stats_second["api_calls"] == 0
assert provider.calls == 1
movie = identities["movies"][0]
assert movie["title_zh"] == "测试中文名"
assert movie["title_en"] == "Test English Title"
assert movie["display_title"] == "测试中文名 Test English Title"
def test_enrich_flags_low_reputation_for_review(tmp_path, monkeypatch):
"""Low reputation should set needs_review when vote count is high enough."""
class LowScoreProvider(DummyProvider):
def enrich(self, *, title: str, media_type: str, year=None):
self.calls += 1
self.last_request_count = 1
return ProviderResult(
provider="dummy",
canonical_id=f"dummy:{title}",
title_zh="低分作品",
title_en="Low Score",
translation_source="dummy",
reputation_score=4.5,
reputation_votes=500,
reputation_source="dummy",
)
config = Config(
library_root=tmp_path,
enrichment_cache_db=tmp_path / "cache.db",
enrichment_providers=["dummy"],
translation_fallback_machine=False,
reputation_low_score_threshold=6.0,
reputation_min_votes=50,
)
provider = LowScoreProvider()
monkeypatch.setattr(
"vlm.enrichment._build_providers",
lambda _config, request_timeout, retries: [provider],
)
identities = {
"metadata": {},
"movies": [
{
"path": "/library/movie/Unknown.2020.mkv",
"filename": "Unknown.2020.mkv",
"category": "movie",
"title": "Unknown",
"year": 2020,
"confidence": 0.9,
"needs_review": False,
}
],
"series": [],
"anime": [],
"other": [],
}
enrich_identities_data(identities, config)
assert identities["movies"][0]["needs_review"] is True
def test_enrich_refresh_all_bypasses_cache(tmp_path, monkeypatch):
"""refresh_all should not use cache and should invoke provider again."""
config = Config(
library_root=tmp_path,
enrichment_cache_db=tmp_path / "cache.db",
enrichment_providers=["dummy"],
translation_fallback_machine=False,
)
provider = DummyProvider()
monkeypatch.setattr(
"vlm.enrichment._build_providers",
lambda _config, request_timeout, retries: [provider],
)
identities = {
"metadata": {},
"movies": [
{
"path": "/library/movie/Test.2020.mkv",
"filename": "Test.2020.mkv",
"category": "movie",
"title": "Test",
"year": 2020,
"confidence": 0.9,
"needs_review": False,
}
],
"series": [],
"anime": [],
"other": [],
}
enrich_identities_data(identities, config, refresh_mode="incremental")
assert provider.calls == 1
_, stats = enrich_identities_data(identities, config, refresh_mode="refresh_all")
assert provider.calls == 2
assert stats["cache_hits"] == 0
def test_build_providers_rejects_unknown_provider(tmp_path):
"""Unknown providers should fail fast with a clear error."""
config = Config(
library_root=tmp_path,
enrichment_providers=["tmdb", "tmdb_typo"],
)
with pytest.raises(ValueError, match="Unsupported enrichment providers"):
_build_providers(config, request_timeout=3, retries=1)
def test_refresh_all_clears_stale_enrichment_fields(tmp_path, monkeypatch):
"""refresh_all with no new match should clear stale enrichment data."""
class FlakyProvider:
name = "dummy"
def __init__(self):
self.calls = 0
self.last_request_count = 0
def enrich(self, *, title: str, media_type: str, year=None):
self.calls += 1
self.last_request_count = 1
if self.calls == 1:
return ProviderResult(
provider="dummy",
canonical_id="dummy:test",
title_zh="测试",
title_en="Test",
translation_source="dummy",
reputation_score=8.8,
reputation_votes=120,
reputation_source="dummy",
)
return None
config = Config(
library_root=tmp_path,
enrichment_cache_db=tmp_path / "cache.db",
enrichment_providers=["dummy"],
translation_fallback_machine=False,
)
provider = FlakyProvider()
monkeypatch.setattr(
"vlm.enrichment._build_providers",
lambda _config, request_timeout, retries: [provider],
)
identities = {
"metadata": {},
"movies": [
{
"path": "/library/movie/Test.2020.mkv",
"filename": "Test.2020.mkv",
"category": "movie",
"title": "Test",
"year": 2020,
"confidence": 0.9,
"needs_review": False,
}
],
"series": [],
"anime": [],
"other": [],
}
enrich_identities_data(identities, config, refresh_mode="incremental")
movie = identities["movies"][0]
assert movie["canonical_id"] == "dummy:test"
assert movie["title_zh"] == "测试"
assert movie["title_en"] == "Test"
enrich_identities_data(identities, config, refresh_mode="refresh_all")
assert movie["canonical_id"] is None
assert movie["title_zh"] is None
assert movie["title_en"] is None
assert movie["translation_source"] is None
assert movie["reputation_score"] is None
assert movie["reputation_votes"] is None
assert movie["reputation_source"] is None
assert movie["enrichment_confidence"] == 0.0
def test_build_display_title_deduplicates_fallback_title(tmp_path):
"""Fallback display title should not duplicate identical names."""
config = Config(library_root=tmp_path)
record = {"title": "Interstellar"}
payload = {"title_zh": None, "title_en": None}
assert _build_display_title(record, payload, config) == "Interstellar"
def test_enrich_without_provider_keys_does_not_count_api_calls(tmp_path):
"""Missing provider keys should not inflate API call metrics."""
config = Config(
library_root=tmp_path,
enrichment_cache_db=tmp_path / "cache.db",
enrichment_providers=["tmdb"],
translation_fallback_machine=True,
tmdb_api_key=None,
openai_api_key=None,
)
identities = {
"metadata": {},
"movies": [
{
"path": "/library/movie/Test.2020.mkv",
"filename": "Test.2020.mkv",
"category": "movie",
"title": "Test",
"year": 2020,
"confidence": 0.9,
"needs_review": False,
}
],
"series": [],
"anime": [],
"other": [],
}
_, stats = enrich_identities_data(identities, config, refresh_mode="refresh_all")
assert stats["api_calls"] == 0
assert stats["enriched"] == 0
assert stats["skip_reasons"] == {"no_key": 1}
def test_enrich_auth_failure_stops_immediately(tmp_path, monkeypatch):
"""Provider authentication errors should stop the run immediately."""
class AuthFailProvider:
name = "tmdb"
last_request_count = 1
def enrich(self, *, title: str, media_type: str, year=None):
raise TMDBAuthError("TMDB authentication failed (401/403)")
config = Config(
library_root=tmp_path,
enrichment_cache_db=tmp_path / "cache.db",
enrichment_providers=["tmdb"],
tmdb_api_key="fake-key",
translation_fallback_machine=False,
)
monkeypatch.setattr(
"vlm.enrichment._build_providers",
lambda _config, request_timeout, retries: [AuthFailProvider()],
)
identities = {
"metadata": {},
"movies": [
{
"path": "/library/movie/Test.2020.mkv",
"filename": "Test.2020.mkv",
"category": "movie",
"title": "Test",
"year": 2020,
"confidence": 0.9,
"needs_review": False,
}
],
"series": [],
"anime": [],
"other": [],
}
with pytest.raises(RuntimeError, match="authentication failed"):
enrich_identities_data(identities, config, refresh_mode="refresh_all")
2026-02-13 13:36:39 +08:00
def test_enrich_respects_max_concurrency_for_uncached_records(tmp_path, monkeypatch):
"""Multiple uncached records should be enriched by multiple worker threads."""
config = Config(
library_root=tmp_path,
enrichment_cache_db=tmp_path / "cache.db",
enrichment_providers=["dummy"],
translation_fallback_machine=False,
enrichment_max_concurrency=4,
)
identities = {
"metadata": {},
"movies": [
{
"path": f"/library/movie/Test.{i}.mkv",
"filename": f"Test.{i}.mkv",
"category": "movie",
"title": f"Test {i}",
"year": 2020,
"confidence": 0.9,
"needs_review": False,
}
for i in range(8)
],
"series": [],
"anime": [],
"other": [],
}
thread_ids: set[int] = set()
lock = threading.Lock()
def _fake_enrich(record, media_type, config_obj, request_timeout, retries):
time.sleep(0.01)
with lock:
thread_ids.add(threading.get_ident())
title = record.get("title", "X")
payload = {
"canonical_id": f"id:{title}",
"title_zh": title,
"title_en": title,
"translation_source": "dummy",
"reputation_score": None,
"reputation_votes": None,
"reputation_source": None,
"provider_metadata": {},
"review_status": "pending",
"enrichment_confidence": 0.8,
"needs_review": False,
"display_title": title,
"enriched": True,
}
return payload, 1, [], ""
monkeypatch.setattr(
"vlm.enrichment._enrich_record_with_fresh_providers",
_fake_enrich,
)
_, stats = enrich_identities_data(identities, config, refresh_mode="refresh_all")
assert stats["enriched"] == 8
assert stats["cache_hits"] == 0
assert len(thread_ids) > 1