Files
dl-organizer/tests/test_analysis_properties.py
6f0df5a774 release: v0.2.0 repository hygiene, CI, and docs sync
Stop tracking personal workflow artifacts at repo root, add CI and MIT
license, align README and agent skills with artifacts/ defaults, and
enable Ruff in dev/CI so releases are verifiable without local-only runs.

Co-Authored-By: Claude Sonnet 4.5 <noreply@anthropic.com>
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-01 15:32:03 +08:00

517 lines
18 KiB
Python

"""Property-based tests for the analysis engine.
Tests universal correctness properties using Hypothesis with minimum 100 iterations.
Each test validates a specific property from the design document.
"""
from datetime import datetime, timezone
from pathlib import Path
from hypothesis import given, settings
from hypothesis import strategies as st
from vlm.analysis import analyze_series_completeness, detect_duplicates
from vlm.models import DuplicateGroup, MovieIdentity, SeasonCompleteness, SeriesIdentity, VideoFile
from vlm.reports import (
generate_completeness_report,
generate_duplicate_report,
generate_summary_report,
)
# Custom strategies for generating test data
@st.composite
def series_identity_strategy(draw, title=None, season=None):
"""Generate a SeriesIdentity with optional fixed title and season."""
if title is None:
title = draw(st.text(min_size=1, max_size=50, alphabet=st.characters(
whitelist_categories=('Lu', 'Ll', 'Nd'), whitelist_characters=' '
)))
if season is None:
season = draw(st.integers(min_value=1, max_value=20))
# Generate 1-3 episode numbers
episode_count = draw(st.integers(min_value=1, max_value=3))
episodes = draw(st.lists(
st.integers(min_value=1, max_value=50),
min_size=episode_count,
max_size=episode_count,
unique=True
))
confidence = draw(st.floats(min_value=0.5, max_value=1.0))
needs_review = False
original_filename = f"{title.replace(' ', '.')}.S{season:02d}E{episodes[0]:02d}.mkv"
return SeriesIdentity(title, season, sorted(episodes), confidence, needs_review, original_filename)
@st.composite
def movie_identity_strategy(draw, title=None, year=None):
"""Generate a MovieIdentity with optional fixed title and year."""
if title is None:
title = draw(st.text(min_size=1, max_size=50, alphabet=st.characters(
whitelist_categories=('Lu', 'Ll', 'Nd'), whitelist_characters=' '
)))
if year is None:
year = draw(st.integers(min_value=1900, max_value=2030))
confidence = draw(st.floats(min_value=0.5, max_value=1.0))
needs_review = False
original_filename = f"{title.replace(' ', '.')}.{year}.mkv"
return MovieIdentity(title, year, confidence, needs_review, original_filename)
@st.composite
def video_file_strategy(draw, filename=None, category="movie"):
"""Generate a VideoFile with optional fixed filename."""
if filename is None:
filename = draw(st.text(min_size=5, max_size=50, alphabet=st.characters(
whitelist_categories=('Lu', 'Ll', 'Nd'), whitelist_characters='.-_'
))) + ".mkv"
path = Path(f"/{category}/{filename}")
size_bytes = draw(st.integers(min_value=1000000, max_value=10000000000))
modified_timestamp = datetime.now(timezone.utc)
# Optional metadata
has_metadata = draw(st.booleans())
if has_metadata:
resolution = draw(st.sampled_from(["1920x1080", "1280x720", "3840x2160", "720x480"]))
codec = draw(st.sampled_from(["h264", "h265", "vp9", "av1"]))
duration_seconds = draw(st.floats(min_value=300, max_value=10800))
bitrate_kbps = draw(st.integers(min_value=500, max_value=20000))
return VideoFile(path, filename, size_bytes, modified_timestamp, category,
resolution, codec, duration_seconds, bitrate_kbps)
else:
return VideoFile(path, filename, size_bytes, modified_timestamp, category)
# Property 10: Gap detection
# Feature: video-library-manager, Property 10: Gap detection
@settings(max_examples=100)
@given(
title=st.text(min_size=1, max_size=30, alphabet=st.characters(
whitelist_categories=('Lu', 'Ll'), whitelist_characters=' '
)),
season=st.integers(min_value=1, max_value=10),
# Generate a list of episode numbers with guaranteed gaps
episodes_data=st.lists(
st.integers(min_value=1, max_value=30),
min_size=3,
max_size=15,
unique=True
)
)
def test_property_10_gap_detection(title, season, episodes_data):
"""Property 10: For any set of episodes within season, analysis SHALL detect
missing episode numbers in range [min, max].
Validates: Requirements 4.1, 4.2
"""
# Sort episodes and ensure there's at least one gap
sorted_episodes = sorted(episodes_data)
# Create episodes, intentionally removing one to create a gap
if len(sorted_episodes) >= 3:
# Remove a middle episode to guarantee a gap
gap_index = len(sorted_episodes) // 2
removed_episode = sorted_episodes[gap_index]
episodes_with_gap = sorted_episodes[:gap_index] + sorted_episodes[gap_index + 1:]
# Create SeriesIdentity objects
episode_identities = [
SeriesIdentity(title, season, [ep], 0.9, False, f"{title}.S{season:02d}E{ep:02d}.mkv")
for ep in episodes_with_gap
]
# Analyze completeness
result = analyze_series_completeness(episode_identities)
# Should detect the gap
if len(result) > 0:
assert result[0].series_title == title
assert result[0].season == season
# The missing episode should be in the detected gaps
min_ep = min(episodes_with_gap)
max_ep = max(episodes_with_gap)
expected_missing = set(range(min_ep, max_ep + 1)) - set(episodes_with_gap)
assert set(result[0].episodes_missing) == expected_missing
assert removed_episode in result[0].episodes_missing
# Property 11: Multi-season independence
# Feature: video-library-manager, Property 11: Multi-season independence
@settings(max_examples=100)
@given(
title=st.text(min_size=1, max_size=30, alphabet=st.characters(
whitelist_categories=('Lu', 'Ll'), whitelist_characters=' '
)),
season1_episodes=st.lists(st.integers(min_value=1, max_value=20), min_size=2, max_size=10, unique=True),
season2_episodes=st.lists(st.integers(min_value=1, max_value=20), min_size=2, max_size=10, unique=True),
)
def test_property_11_multi_season_independence(title, season1_episodes, season2_episodes):
"""Property 11: For any series with multiple seasons, gap detection of one
season SHALL not affect others.
Validates: Requirements 4.4
"""
# Create episodes for season 1 with a gap
s1_sorted = sorted(season1_episodes)
if len(s1_sorted) >= 3:
gap_index = len(s1_sorted) // 2
s1_with_gap = s1_sorted[:gap_index] + s1_sorted[gap_index + 1:]
s1_missing = s1_sorted[gap_index]
else:
s1_with_gap = s1_sorted
s1_missing = None
# Create episodes for season 2 (complete, no gaps)
s2_sorted = sorted(season2_episodes)
s2_complete = list(range(min(s2_sorted), max(s2_sorted) + 1))
# Create SeriesIdentity objects
episode_identities = []
for ep in s1_with_gap:
episode_identities.append(
SeriesIdentity(title, 1, [ep], 0.9, False, f"{title}.S01E{ep:02d}.mkv")
)
for ep in s2_complete:
episode_identities.append(
SeriesIdentity(title, 2, [ep], 0.9, False, f"{title}.S02E{ep:02d}.mkv")
)
# Analyze completeness
result = analyze_series_completeness(episode_identities)
# Season 2 should not appear in results (it's complete)
season2_results = [r for r in result if r.season == 2]
assert len(season2_results) == 0
# Season 1 should appear if there's a gap
if s1_missing is not None:
season1_results = [r for r in result if r.season == 1]
if len(season1_results) > 0:
assert s1_missing in season1_results[0].episodes_missing
# Property 12: Duplicate detection for movies
# Feature: video-library-manager, Property 12: Duplicate detection for movies
@settings(max_examples=100)
@given(
title=st.text(min_size=1, max_size=30, alphabet=st.characters(
whitelist_categories=('Lu', 'Ll'), whitelist_characters=' '
)),
year=st.integers(min_value=1900, max_value=2030),
duplicate_count=st.integers(min_value=2, max_value=5)
)
def test_property_12_duplicate_detection_movies(title, year, duplicate_count):
"""Property 12: For any set of movies with identical normalized titles and years,
all SHALL be grouped as duplicates.
Validates: Requirements 5.1
"""
# Create multiple movie identities with same title and year
identities = []
files = []
for i in range(duplicate_count):
filename = f"{title.replace(' ', '.')}.{year}.{i}.mkv"
identities.append(MovieIdentity(title, year, 0.9, False, filename))
files.append(VideoFile(
Path(f"/movies/{filename}"),
filename,
1000000000 + i * 100000000,
datetime.now(timezone.utc),
"movie"
))
# Detect duplicates
result = detect_duplicates(list(zip(identities, files)))
# Should find exactly one duplicate group
assert len(result) == 1
# The group should contain all files
assert len(result[0].files) == duplicate_count
# Identity should match
assert result[0].identity.title == title
assert result[0].identity.year == year
# Property 13: Duplicate detection for series
# Feature: video-library-manager, Property 13: Duplicate detection for series
@settings(max_examples=100)
@given(
title=st.text(min_size=1, max_size=30, alphabet=st.characters(
whitelist_categories=('Lu', 'Ll'), whitelist_characters=' '
)),
season=st.integers(min_value=1, max_value=10),
episode=st.integers(min_value=1, max_value=30),
duplicate_count=st.integers(min_value=2, max_value=5)
)
def test_property_13_duplicate_detection_series(title, season, episode, duplicate_count):
"""Property 13: For any set of series files with identical normalized titles,
seasons, and episodes, all SHALL be grouped as duplicates.
Validates: Requirements 5.2
"""
# Create multiple series identities with same title, season, and episode
identities = []
files = []
for i in range(duplicate_count):
filename = f"{title.replace(' ', '.')}.S{season:02d}E{episode:02d}.{i}.mkv"
identities.append(SeriesIdentity(title, season, [episode], 0.9, False, filename))
files.append(VideoFile(
Path(f"/series/{filename}"),
filename,
1000000000 + i * 100000000,
datetime.now(timezone.utc),
"series"
))
# Detect duplicates
result = detect_duplicates(list(zip(identities, files)))
# Should find exactly one duplicate group
assert len(result) == 1
# The group should contain all files
assert len(result[0].files) == duplicate_count
# Identity should match
assert result[0].identity.title == title
assert result[0].identity.season == season
assert episode in result[0].identity.episodes
# Property 14: Duplicate quality comparison
# Feature: video-library-manager, Property 14: Duplicate quality comparison
@settings(max_examples=100)
@given(
title=st.text(min_size=1, max_size=30, alphabet=st.characters(
whitelist_categories=('Lu', 'Ll'), whitelist_characters=' '
)),
year=st.integers(min_value=1900, max_value=2030),
file_count=st.integers(min_value=2, max_value=4)
)
def test_property_14_duplicate_quality_comparison(title, year, file_count):
"""Property 14: For any duplicate group, comparison data SHALL include
available metadata for each file.
Validates: Requirements 5.3
"""
# Create movie identities and files with varying metadata
identities = []
files = []
for i in range(file_count):
filename = f"{title.replace(' ', '.')}.{year}.{i}.mkv"
identities.append(MovieIdentity(title, year, 0.9, False, filename))
# Some files have full metadata, some don't
if i % 2 == 0:
files.append(VideoFile(
Path(f"/movies/{filename}"),
filename,
1000000000 + i * 100000000,
datetime.now(timezone.utc),
"movie",
resolution="1920x1080",
codec="h264",
duration_seconds=7200.0,
bitrate_kbps=5000
))
else:
files.append(VideoFile(
Path(f"/movies/{filename}"),
filename,
1000000000 + i * 100000000,
datetime.now(timezone.utc),
"movie"
))
# Detect duplicates
result = detect_duplicates(list(zip(identities, files)))
# Should have quality comparison data
assert len(result) == 1
assert len(result[0].quality_comparison) == file_count
# Each comparison entry should have at least filename and size
for comparison in result[0].quality_comparison:
assert 'filename' in comparison
assert 'size_bytes' in comparison
assert 'path' in comparison
# Files with metadata should have those fields
if comparison['filename'].endswith('.0.mkv') or comparison['filename'].endswith('.2.mkv'):
assert 'resolution' in comparison
assert 'codec' in comparison
assert 'duration_seconds' in comparison
assert 'bitrate_kbps' in comparison
# Property 42: Completeness report
# Feature: video-library-manager, Property 42: Completeness report
@settings(max_examples=100)
@given(
series_count=st.integers(min_value=1, max_value=5),
format=st.sampled_from(["text", "json"])
)
def test_property_42_completeness_report(series_count, format):
"""Property 42: For any set of analyzed series, completeness report SHALL
include all series with detected gaps.
Validates: Requirements 11.2
"""
# Create series with gaps
analysis_results = []
for i in range(series_count):
title = f"Series {i}"
season = 1
episodes_found = [1, 2, 4, 5] # Gap at episode 3
episodes_missing = [3]
analysis_results.append(SeasonCompleteness(
series_title=title,
season=season,
episodes_found=episodes_found,
episodes_missing=episodes_missing
))
# Generate report
library_root = Path("/test/library")
report = generate_completeness_report(analysis_results, format, library_root)
# Report should include all series
for i in range(series_count):
assert f"Series {i}" in report
# Report should include metadata
assert str(library_root) in report
# Property 43: Duplicate report grouping
# Feature: video-library-manager, Property 43: Duplicate report grouping
@settings(max_examples=100)
@given(
duplicate_count=st.integers(min_value=1, max_value=5),
format=st.sampled_from(["text", "json"])
)
def test_property_43_duplicate_report_grouping(duplicate_count, format):
"""Property 43: For any set of detected duplicates, duplicate report SHALL
group files by identity with comparison data.
Validates: Requirements 11.3
"""
# Create duplicate groups
duplicate_groups = []
for i in range(duplicate_count):
title = f"Movie {i}"
year = 2020 + i
# Create 2 files for each duplicate group
files = []
quality_comparison = []
for j in range(2):
filename = f"{title.replace(' ', '.')}.{year}.{j}.mkv"
file = VideoFile(
Path(f"/movies/{filename}"),
filename,
1000000000 + j * 500000000,
datetime.now(timezone.utc),
"movie",
resolution="1920x1080" if j == 0 else "1280x720",
codec="h264"
)
files.append(file)
quality_comparison.append({
'filename': filename,
'path': str(file.path),
'size_bytes': file.size_bytes,
'resolution': file.resolution,
'codec': file.codec
})
identity = MovieIdentity(title, year, 0.9, False, files[0].filename)
duplicate_groups.append(DuplicateGroup(identity, files, quality_comparison))
# Generate report
library_root = Path("/test/library")
report = generate_duplicate_report(duplicate_groups, format, library_root)
# Report should include all duplicate groups
for i in range(duplicate_count):
assert f"Movie {i}" in report
# Report should include comparison data (file sizes, resolutions)
assert "1920x1080" in report or "resolution" in report.lower()
# Report should include metadata
assert str(library_root) in report
# Property 44: Summary report accuracy
# Feature: video-library-manager, Property 44: Summary report accuracy
@settings(max_examples=100)
@given(
file_count=st.integers(min_value=1, max_value=20),
categories=st.lists(
st.sampled_from(["movie", "series", "anime", "other"]),
min_size=1,
max_size=4
)
)
def test_property_44_summary_report_accuracy(file_count, categories):
"""Property 44: For any scanned library, summary report SHALL contain
accurate counts and sizes.
Validates: Requirements 11.4
"""
# Create video files
files = []
total_size = 0
category_counts = {}
for i in range(file_count):
category = categories[i % len(categories)]
size = 1000000000 + i * 100000000
filename = f"file_{i}.mkv"
files.append(VideoFile(
Path(f"/{category}/{filename}"),
filename,
size,
datetime.now(timezone.utc),
category
))
total_size += size
category_counts[category] = category_counts.get(category, 0) + 1
# Generate summary report
library_root = Path("/test/library")
report = generate_summary_report(files, library_root)
# Report should include total file count
assert f"Total Files: {file_count}" in report
# Report should include category breakdown
for category, count in category_counts.items():
assert category.capitalize() in report
assert f"Files: {count}" in report
# Report should include metadata
assert str(library_root) in report