Initial commit: Video Library Manager
- Add core VLM modules (scanner, parser, planner, executor, analysis) - Add CLI with quarantine, reports, rollback, and state management - Add comprehensive test suite - Add project configuration and documentation - Add .gitignore for Python project
This commit is contained in:
@@ -0,0 +1,267 @@
|
||||
"""Identity parser for extracting movie and series information from filenames.
|
||||
|
||||
This module provides functionality to parse video filenames and extract
|
||||
logical identities such as movie titles/years and series titles/seasons/episodes.
|
||||
"""
|
||||
|
||||
import re
|
||||
from typing import Optional
|
||||
|
||||
from vlm.models import MovieIdentity, SeriesIdentity
|
||||
|
||||
|
||||
# Quality tags to remove from titles
|
||||
QUALITY_TAGS = [
|
||||
r'\b1080p\b', r'\b720p\b', r'\b480p\b', r'\b2160p\b',
|
||||
r'\b4K\b', r'\bUHD\b', r'\bHD\b',
|
||||
r'\bBluRay\b', r'\bBlu-Ray\b', r'\bBRRip\b', r'\bBDRip\b',
|
||||
r'\bWEB-DL\b', r'\bWEBRip\b', r'\bWEB\b',
|
||||
r'\bHDTV\b', r'\bHDRip\b',
|
||||
r'\bDVDRip\b', r'\bDVD\b',
|
||||
r'\bx264\b', r'\bx265\b', r'\bh264\b', r'\bh265\b', r'\bHEVC\b',
|
||||
r'\bAAC\b', r'\bAC3\b', r'\bDTS\b',
|
||||
r'\b10bit\b', r'\b8bit\b',
|
||||
]
|
||||
|
||||
# Release group patterns (in brackets, but NOT years in parentheses)
|
||||
RELEASE_GROUP_PATTERNS = [
|
||||
r'\[[\w\s\-\.]+\]', # [RARBG], [YTS], etc.
|
||||
]
|
||||
|
||||
|
||||
def remove_quality_tags(text: str) -> str:
|
||||
"""Remove quality indicators from text.
|
||||
|
||||
Args:
|
||||
text: Input text containing potential quality tags
|
||||
|
||||
Returns:
|
||||
Text with quality tags removed
|
||||
"""
|
||||
result = text
|
||||
for pattern in QUALITY_TAGS:
|
||||
result = re.sub(pattern, '', result, flags=re.IGNORECASE)
|
||||
return result
|
||||
|
||||
|
||||
def remove_release_groups(text: str) -> str:
|
||||
"""Remove release group tags from text.
|
||||
|
||||
Args:
|
||||
text: Input text containing potential release group tags
|
||||
|
||||
Returns:
|
||||
Text with release group tags removed
|
||||
"""
|
||||
result = text
|
||||
for pattern in RELEASE_GROUP_PATTERNS:
|
||||
result = re.sub(pattern, '', result)
|
||||
return result
|
||||
|
||||
|
||||
def normalize_title(title: str) -> str:
|
||||
"""Normalize a title by cleaning whitespace and standardizing capitalization.
|
||||
|
||||
Args:
|
||||
title: Raw title string
|
||||
|
||||
Returns:
|
||||
Normalized title with proper capitalization and spacing
|
||||
"""
|
||||
# Replace dots and underscores with spaces
|
||||
title = title.replace('.', ' ').replace('_', ' ')
|
||||
|
||||
# Remove extra whitespace
|
||||
title = ' '.join(title.split())
|
||||
|
||||
# Apply title case
|
||||
title = title.title()
|
||||
|
||||
return title.strip()
|
||||
|
||||
|
||||
def parse_movie(filename: str) -> MovieIdentity:
|
||||
"""Parse a movie filename to extract title and year.
|
||||
|
||||
Supports patterns:
|
||||
- Title (Year)
|
||||
- Title.Year
|
||||
- Title - Year
|
||||
- Title Year
|
||||
|
||||
Args:
|
||||
filename: Movie filename to parse
|
||||
|
||||
Returns:
|
||||
MovieIdentity with extracted information
|
||||
"""
|
||||
# Remove file extension
|
||||
name_without_ext = filename
|
||||
for ext in ['.mp4', '.mkv', '.avi', '.mov', '.wmv', '.flv', '.webm', '.m4v']:
|
||||
if name_without_ext.lower().endswith(ext):
|
||||
name_without_ext = name_without_ext[:-len(ext)]
|
||||
break
|
||||
|
||||
# Try different patterns in order of confidence BEFORE cleaning
|
||||
# This preserves the year in parentheses
|
||||
patterns = [
|
||||
# Pattern: Title (Year) - High confidence
|
||||
(r'^(.+?)\s*\((\d{4})\)', 0.9),
|
||||
# Pattern: Title.Year or Title-Year - High confidence
|
||||
(r'^(.+?)[\.\-](\d{4})', 0.9),
|
||||
# Pattern: Title - Year - Medium confidence
|
||||
(r'^(.+?)\s+-\s+(\d{4})', 0.7),
|
||||
# Pattern: Title Year (4 digits at end) - Medium confidence
|
||||
(r'^(.+?)\s+(\d{4})(?:\s|$)', 0.7),
|
||||
]
|
||||
|
||||
for pattern, confidence in patterns:
|
||||
match = re.search(pattern, name_without_ext)
|
||||
if match:
|
||||
title = match.group(1)
|
||||
year = int(match.group(2))
|
||||
|
||||
# Now clean the title
|
||||
title = remove_quality_tags(title)
|
||||
title = remove_release_groups(title)
|
||||
title = normalize_title(title)
|
||||
|
||||
return MovieIdentity(
|
||||
title=title,
|
||||
year=year,
|
||||
confidence=confidence,
|
||||
needs_review=False,
|
||||
original_filename=filename
|
||||
)
|
||||
|
||||
# No year found - clean and extract title, flag for review
|
||||
cleaned = remove_quality_tags(name_without_ext)
|
||||
cleaned = remove_release_groups(cleaned)
|
||||
title = normalize_title(cleaned)
|
||||
|
||||
return MovieIdentity(
|
||||
title=title,
|
||||
year=None,
|
||||
confidence=0.3,
|
||||
needs_review=True,
|
||||
original_filename=filename
|
||||
)
|
||||
|
||||
|
||||
def parse_series(filename: str) -> SeriesIdentity:
|
||||
"""Parse a series filename to extract title, season, and episode numbers.
|
||||
|
||||
Supports patterns:
|
||||
- SXXEYY (e.g., S01E01)
|
||||
- SXXeYY (e.g., S01e01)
|
||||
- SeasonXEpisodeY (e.g., Season1Episode1)
|
||||
- XXxYY (e.g., 1x01)
|
||||
- Multi-episode: S01E01-E02, S01E01E02, etc.
|
||||
|
||||
Args:
|
||||
filename: Series filename to parse
|
||||
|
||||
Returns:
|
||||
SeriesIdentity with extracted information
|
||||
"""
|
||||
# Remove file extension
|
||||
name_without_ext = filename
|
||||
for ext in ['.mp4', '.mkv', '.avi', '.mov', '.wmv', '.flv', '.webm', '.m4v']:
|
||||
if name_without_ext.lower().endswith(ext):
|
||||
name_without_ext = name_without_ext[:-len(ext)]
|
||||
break
|
||||
|
||||
# Try different patterns in order of confidence
|
||||
patterns = [
|
||||
# Pattern: SXXEYY or SXXeYY - High confidence
|
||||
# Also handles multi-episode: S01E01-E02, S01E01E02E03, etc.
|
||||
(r'[Ss](\d{1,2})[Ee](\d{1,2})', 0.9),
|
||||
# Pattern: XXxYY - High confidence
|
||||
(r'(\d{1,2})x(\d{1,2})', 0.9),
|
||||
# Pattern: Season X Episode Y - Medium confidence
|
||||
(r'[Ss]eason\s*(\d{1,2})\s*[Ee]pisode\s*(\d{1,2})', 0.7),
|
||||
]
|
||||
|
||||
season = None
|
||||
episodes = []
|
||||
confidence = 0.0
|
||||
title_part = name_without_ext
|
||||
|
||||
for pattern, conf in patterns:
|
||||
match = re.search(pattern, name_without_ext, re.IGNORECASE)
|
||||
if match:
|
||||
season = int(match.group(1))
|
||||
episodes = [int(match.group(2))]
|
||||
confidence = conf
|
||||
|
||||
# Extract title (everything before the match)
|
||||
title_part = name_without_ext[:match.start()]
|
||||
|
||||
# Handle multi-episode files for SXXEYY pattern
|
||||
if pattern.startswith(r'[Ss]'):
|
||||
# Find the full episode section (from S01E01 onwards)
|
||||
remaining = name_without_ext[match.start():]
|
||||
|
||||
# Look for all episode numbers: E01, -E02, E03, etc.
|
||||
all_episode_matches = re.findall(r'[Ee](\d{1,2})', remaining)
|
||||
if all_episode_matches:
|
||||
episodes = [int(ep) for ep in all_episode_matches]
|
||||
|
||||
break
|
||||
|
||||
# Clean the title
|
||||
if title_part:
|
||||
title_part = remove_quality_tags(title_part)
|
||||
title_part = remove_release_groups(title_part)
|
||||
title_part = normalize_title(title_part)
|
||||
else:
|
||||
# If no title part found, use the whole filename cleaned
|
||||
title_part = remove_quality_tags(name_without_ext)
|
||||
title_part = remove_release_groups(title_part)
|
||||
title_part = normalize_title(title_part)
|
||||
|
||||
# Determine if review is needed
|
||||
needs_review = season is None or len(episodes) == 0
|
||||
|
||||
# If no pattern matched, set low confidence
|
||||
if season is None:
|
||||
confidence = 0.3
|
||||
|
||||
return SeriesIdentity(
|
||||
title=title_part,
|
||||
season=season,
|
||||
episodes=episodes,
|
||||
confidence=confidence,
|
||||
needs_review=needs_review,
|
||||
original_filename=filename
|
||||
)
|
||||
|
||||
|
||||
def group_episodes(episodes: list[SeriesIdentity]) -> dict[tuple[str, int], list[SeriesIdentity]]:
|
||||
"""Group parsed episodes by normalized series title and season number.
|
||||
|
||||
Episodes are grouped by (normalized_title, season) tuple. Episodes with
|
||||
season=None are excluded from grouping as they need manual review.
|
||||
|
||||
Args:
|
||||
episodes: List of parsed series identities
|
||||
|
||||
Returns:
|
||||
Dictionary mapping (title, season) tuples to lists of SeriesIdentity objects
|
||||
"""
|
||||
groups: dict[tuple[str, int], list[SeriesIdentity]] = {}
|
||||
|
||||
for episode in episodes:
|
||||
# Skip episodes without season (they need manual review)
|
||||
if episode.season is None:
|
||||
continue
|
||||
|
||||
# Create grouping key from normalized title and season
|
||||
key = (episode.title, episode.season)
|
||||
|
||||
# Add episode to the appropriate group
|
||||
if key not in groups:
|
||||
groups[key] = []
|
||||
groups[key].append(episode)
|
||||
|
||||
return groups
|
||||
Reference in New Issue
Block a user