5 Commits
10 changed files with 157 additions and 39 deletions
+6 -1
View File
@@ -647,7 +647,12 @@ except (LookupError, ValueError, AttributeError) as e:
1. **Read Before Modify**: Always read files before suggesting modifications
2. **Follow Existing Patterns**: Understand established architecture before changes
3. **Test Everything**: Run `uv run pytest` after all changes
3. **Run CI Checks After Every Code Change**: After any code modification, run the same checks as GitHub CI:
```bash
uv run pytest # all tests must pass
uv run mypy src/ --ignore-missing-imports # no type errors
```
Do NOT consider work complete until both commands succeed.
4. **Simplicity First**: Avoid over-engineering solutions
5. **Document Changes**: Update relevant documentation
+3 -3
View File
@@ -28,7 +28,7 @@ powershell -c "irm https://astral.sh/uv/install.sh | iex"
uv tool install https://github.com/shadoll/moma/releases/latest/download/moma-latest.tar.gz
# Specific version
uv tool install https://github.com/shadoll/moma/releases/download/v0.9.3/moma-0.9.3-py3-none-any.whl
uv tool install https://github.com/shadoll/moma/releases/download/v0.9.5/moma-0.9.5-py3-none-any.whl
# From PyPI (when published)
uv tool install moma
@@ -40,7 +40,7 @@ uv tool install moma
uv tool install --force https://github.com/shadoll/moma/releases/latest/download/moma-latest.tar.gz
# Upgrade to a newer specific version
uv tool install --force https://github.com/shadoll/moma/releases/download/v0.9.3/moma-0.9.3-py3-none-any.whl
uv tool install --force https://github.com/shadoll/moma/releases/download/v0.9.5/moma-0.9.5-py3-none-any.whl
```
#### Usage
@@ -56,7 +56,7 @@ moma /path/to/directory # Scan specific directory
pip install https://github.com/shadoll/moma/releases/latest/download/moma-latest.tar.gz
# Specific version
pip install https://github.com/shadoll/moma/releases/download/v0.9.3/moma-0.9.3-py3-none-any.whl
pip install https://github.com/shadoll/moma/releases/download/v0.9.5/moma-0.9.5-py3-none-any.whl
```
### Method 3: Development Installation
+1 -1
View File
@@ -1,6 +1,6 @@
[project]
name = "moma"
version = "0.9.3"
version = "0.9.5"
description = "Terminal-based media file renamer and metadata viewer"
readme = "README.md"
requires-python = ">=3.12"
+3 -3
View File
@@ -211,9 +211,9 @@ class MomaApp(App):
icons = {
'mkv': '🎥', # Video camera for MKV
'mk3d': '🕹️', # Clapper board for 3D
'mp4': '🎥', # Video camera
'mov': '🎥', # Video camera
'webm': '🎥', # Video camera
'mp4': '🌐', # Web
'mov': '📽️', # Video camera
'webm': '🌐', # Web
'avi': '💿', # Film frames for AVI
'wmv': '📀', # Video camera
'm4v': '📹', # Video camera
+51 -24
View File
@@ -64,19 +64,26 @@ class FilenameExtractor:
if dot_match:
year_pos = dot_match.start()
else:
# Last resort: any 4-digit number
any_match = re.search(r'\b(\d{4})\b', self.file_name)
if any_match:
year = int(any_match.group(1))
# Basic sanity check using constants
if is_valid_year(year):
year_pos = any_match.start() # Cut before the year for plain years
# Try year between mixed separators (like .1967_ or _1967.)
sep_match = re.search(r'(?<=[.\-_\s])(\d{4})(?=[.\-_\s])', self.file_name)
if sep_match:
year_val = int(sep_match.group(1))
if is_valid_year(year_val):
year_pos = sep_match.start(1)
else:
# Last resort: any 4-digit number
any_match = re.search(r'\b(\d{4})\b', self.file_name)
if any_match:
year = int(any_match.group(1))
# Basic sanity check using constants
if is_valid_year(year):
year_pos = any_match.start() # Cut before the year for plain years
# Find source position
source = self.extract_source()
if source:
for alias in SOURCE_DICT[source]:
match = re.search(r'\b' + re.escape(alias) + r'\b', self.file_name, re.IGNORECASE)
match = re.search(r'(?<![a-zA-Z])' + re.escape(alias) + r'(?![a-zA-Z])', self.file_name, re.IGNORECASE)
if match:
source_pos = match.start()
break
@@ -108,26 +115,23 @@ class FilenameExtractor:
# Remove bracketed prefixes like [01.1], [1], etc.
title = re.sub(r'^\s*\[[^\]]+\]\s*', '', title)
# Remove order number prefixes like 01., 1., 1.1 followed by space/underscore
# Only remove if the number is multi-digit or has decimal (to avoid removing single digit titles)
match = re.match(r'^\s*(\d+(?:\.\d+)?)\.(?=\s|_)', title)
if match:
order = match.group(1)
if len(order) > 1 or '.' in order:
title = re.sub(r'^\s*(\d+(?:\.\d+)?)\.(?=\s|_)', '', title)
# Remove order like 1.9 where 1 is order, 9 is title
# Remove order prefix (order followed by dot or space)
order = self.extract_order()
if order:
match = re.match(r'^' + re.escape(order) + r'\.(.+)', title)
match = re.match(r'^' + re.escape(order) + r'[.\s]+(.+)', title)
if match:
title = match.group(1)
# Clean up any remaining leading separators
title = title.lstrip('_ \t')
# Clean up title: remove leading/trailing brackets and dots
title = title.strip('[](). ')
# Clean up title: remove leading/trailing brackets and orphaned dots
title = title.strip('[]. ')
# Only strip unmatched leading/trailing parens
if title.endswith(')') and title.count('(') < title.count(')'):
title = title.rstrip(')')
if title.startswith('(') and title.count('(') > title.count(')'):
title = title.lstrip('(')
# Replace dots with spaces if they appear to be word separators
# Only replace dots that are surrounded by letters/digits (not at edges)
@@ -150,7 +154,14 @@ class FilenameExtractor:
dot_match = re.search(r'\.(\d{4})\.', self.file_name)
if dot_match:
return dot_match.group(1)
# Try year between mixed separators (like .1967_ or _1967.)
sep_match = re.search(r'(?<=[.\-_\s])(\d{4})(?=[.\-_\s])', self.file_name)
if sep_match:
year = int(sep_match.group(1))
if is_valid_year(year):
return str(year)
# Last resort: any 4-digit number (but this is less reliable)
any_match = re.search(r'\b(\d{4})\b', self.file_name)
if any_match:
@@ -212,6 +223,13 @@ class FilenameExtractor:
return frame_class
# Fallback to height-based if not in constants
return self._get_frame_class_from_height(height)
# Check for bare resolution numbers inside brackets (e.g., [720,ukr,eng])
bare_match = re.search(r'[\[,](\d{3,4})(?=[,\]])', normalized_name, re.IGNORECASE)
if bare_match:
bare_fc = self._get_frame_class_from_height(int(bare_match.group(1)))
if bare_fc:
return bare_fc
# If no specific resolution found, check for non-standard quality indicators
for indicator in NON_STANDARD_QUALITY_INDICATORS:
@@ -334,8 +352,17 @@ class FilenameExtractor:
# Remove bracketed content first
text_without_brackets = re.sub(r'\[([^\]]+)\]', '', self.file_name)
# Split on dots, spaces, and underscores
parts = re.split(r'[.\s_]+', text_without_brackets)
# Find start of metadata section (after title) to avoid title words being
# misdetected as language codes (e.g. "War" from "The.War.Wagon")
metadata_start = 0
year_m = (re.search(r'\(\d{4}\)', text_without_brackets) or
re.search(r'\.\d{4}\.', text_without_brackets) or
re.search(r'(?<=[.\-_\s])\d{4}(?=[.\-_\s])', text_without_brackets))
if year_m:
metadata_start = year_m.start()
# Split on dots, spaces, and underscores (only in the post-title portion)
parts = re.split(r'[.\s_]+', text_without_brackets[metadata_start:])
for part in parts:
part = part.strip()
+8 -2
View File
@@ -195,7 +195,7 @@ class MediaInfoExtractor:
resolution = self.extract_resolution()
if not resolution:
return None
height, width = resolution
width, height = resolution
logger.debug(
f"[{self.file_path.name}] Frame class detection - Resolution: {width}x{height}"
@@ -329,7 +329,10 @@ class MediaInfoExtractor:
return None
langs = []
for a in tracks:
lang_code = getattr(a, "language", "und") or "und"
lang_code = getattr(a, "language", None)
# Skip tracks with no language tag or 'und' (undetermined)
if not lang_code or lang_code.lower() in ("und", "undefined"):
continue
try:
# Try to get the 3-letter code
lang_obj = langcodes.Language.get(lang_code.lower())
@@ -340,6 +343,9 @@ class MediaInfoExtractor:
logger.debug(f"Invalid language code '{lang_code}': {e}")
langs.append(lang_code.lower()[:3])
if not langs:
return None # No meaningful language info — let Filename extractor try
lang_counts = Counter(langs)
audio_langs = [
f"{count}{lang}" if count > 1 else lang
@@ -2,6 +2,24 @@
"description": "Comprehensive test dataset for filename metadata extraction",
"version": "2.0",
"test_cases": [
{
"filename": "The.War.Wagon.1967_BDRip Ukr_Eng[Hurtom].mkv",
"expected": {
"order": null,
"title": "The War Wagon",
"year": "1967",
"source": "BDRip",
"frame_class": null,
"hdr": null,
"movie_db": null,
"special_info": null,
"audio_langs": "ukr,eng",
"extension": "mkv"
},
"testname": "edge-multi-lang-001",
"category": "edge_cases",
"description": "Multiple languages without brackets"
},
{
"filename": "Le Jaguar.(1996).[1080i,3ukr,fra].mkv",
"expected": {
@@ -799,7 +817,7 @@
"filename": "Movie.Title (2020) BDRip [1080p,ukr,eng].mkv",
"expected": {
"order": null,
"title": "Movie.Title",
"title": "Movie Title",
"year": "2020",
"source": "BDRip",
"frame_class": "1080p",
@@ -810,7 +828,7 @@
"extension": "mkv"
},
"category": "edge_cases",
"description": "Title with dots"
"description": "Title with dot separator (dot replaced with space by extractor)"
},
{
"testname": "edge-no-brackets-001",
+63 -1
View File
@@ -15,6 +15,22 @@ def load_test_filenames():
return []
def load_test_cases():
"""Load full test cases (testname, filename, expected) from dataset"""
dataset_file = Path(__file__).parent / "datasets" / "filenames" / "filename_patterns.json"
if dataset_file.exists():
with open(dataset_file, 'r', encoding='utf-8') as f:
data = json.load(f)
return [
pytest.param(
case['filename'],
case['expected'],
id=case.get('testname', case['filename'])
)
for case in data['test_cases']
]
return []
@pytest.mark.parametrize("filename", load_test_filenames())
def test_extract_title(filename):
"""Test title extraction from filename"""
@@ -128,4 +144,50 @@ def test_extract_audio_tracks(filename):
assert isinstance(audio_tracks, list)
for track in audio_tracks:
assert isinstance(track, dict)
assert 'language' in track
assert 'language' in track
# ---------------------------------------------------------------------------
# Dataset-based value-checking tests (check against filename_patterns.json)
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("filename,expected", load_test_cases())
def test_expected_title(filename, expected):
"""Test that extracted title matches the expected value from dataset."""
extractor = FilenameExtractor(Path(filename), use_cache=False)
assert extractor.extract_title() == expected['title']
@pytest.mark.parametrize("filename,expected", load_test_cases())
def test_expected_year(filename, expected):
"""Test that extracted year matches the expected value from dataset."""
extractor = FilenameExtractor(Path(filename), use_cache=False)
assert extractor.extract_year() == expected['year']
@pytest.mark.parametrize("filename,expected", load_test_cases())
def test_expected_source(filename, expected):
"""Test that extracted source matches the expected value from dataset."""
extractor = FilenameExtractor(Path(filename), use_cache=False)
assert extractor.extract_source() == expected['source']
@pytest.mark.parametrize("filename,expected", load_test_cases())
def test_expected_frame_class(filename, expected):
"""Test that extracted frame_class matches the expected value from dataset."""
extractor = FilenameExtractor(Path(filename), use_cache=False)
assert extractor.extract_frame_class() == expected['frame_class']
@pytest.mark.parametrize("filename,expected", load_test_cases())
def test_expected_audio_langs(filename, expected):
"""Test that extracted audio_langs matches the expected value from dataset."""
extractor = FilenameExtractor(Path(filename), use_cache=False)
assert extractor.extract_audio_langs() == expected['audio_langs']
@pytest.mark.parametrize("filename,expected", load_test_cases())
def test_expected_movie_db(filename, expected):
"""Test that extracted movie_db matches the expected value from dataset."""
extractor = FilenameExtractor(Path(filename), use_cache=False)
assert extractor.extract_movie_db() == expected['movie_db']
+1 -1
View File
@@ -39,7 +39,7 @@ def test_frame_class_detection(test_case):
extractor.video_tracks = [mock_track]
extractor._get_tracks.return_value = [mock_track] # satisfies @requires_tracks_type decorator
extractor._get_track.return_value = mock_track
extractor.extract_resolution.return_value = (height, width)
extractor.extract_resolution.return_value = (width, height)
extractor.extract_interlaced.return_value = interlaced
# Test the method
Generated
+1 -1
View File
@@ -208,7 +208,7 @@ wheels = [
[[package]]
name = "moma"
version = "0.9.3"
version = "0.9.5"
source = { editable = "." }
dependencies = [
{ name = "langcodes" },