OpenTranscode/tests/test_title_clean.py

87 lines
2.9 KiB
Python

"""Title-artifact scrubbing tests — v4.10.0.
``clean_title`` removes the SOURCE encode's claims from a filename stem
so outputs are not misdescribed: after transcoding to (typically)
AV1/Opus/MKV — or Theora/OGV, or any other profile — stale tags like
`x264`, `h.265`, `webm`, `xvid` or `dts` are wrong, as are this app's
own `_archived` / `_<w>x<h>` suffixes when a file is re-encoded.
All cases mirror real release-name shapes. No PySide6 needed — the
module is pure.
"""
from __future__ import annotations
import pytest
from conftest import load_program_module
_m = load_program_module()
TITLE_ARTIFACT_TOKENS = _m.TITLE_ARTIFACT_TOKENS
clean_title = _m.clean_title
# (input stem, expected output) — table-driven, one row per behavior.
SCRUB_CASES = [
# Multi-tag scene release: codecs die, content tags survive
("Movie.x264.1080p.WEBRip.x265-GRP", "Movie.1080p.WEBRip.GRP"),
("Show.S01E05.720p.hdtv.x264-RLSGRP", "Show.S01E05.720p.hdtv.RLSGRP"),
# Mixed video+audio tags
("Film.h264.aac.5.1", "Film.5.1"),
("xvid_classic.dts", "classic"),
("Thing.S02E04.hevc.vorbis.webm", "Thing.S02E04"),
# Bracketed tags and the residue they leave
("Old.Movie[DivX].2001", "Old.Movie.2001"),
# Container-only artifacts
("clip.webm", "clip"),
("svt-av1.test.theora.flac", "test"),
# Dash-joined residue collapses to a single separator
("Movie.x265-GRP", "Movie.GRP"),
# This app's own suffixes never stack across re-encodes
("vid_1920x1080_archived", "vid"),
("already_archived", "already"),
("clip_1280x720", "clip"),
# Case-insensitive matching
("MOVIE.X264", "MOVIE"),
("Movie.H.265.He-AAC", "Movie"),
]
# Stems that must pass through byte-for-byte.
PASSTHROUGH_CASES = [
# Tokens inside larger words are NOT codec claims
"MP4Box.and.Aviator.H264file",
"Totally Clean Name",
"Show.S01E12.1080p.BluRay",
"documentary_2026",
]
@pytest.mark.parametrize("stem,expected", SCRUB_CASES)
def test_scrub_cases(stem, expected):
assert clean_title(stem) == expected
@pytest.mark.parametrize("stem", PASSTHROUGH_CASES)
def test_passthrough_cases(stem):
assert clean_title(stem) == stem
def test_scrub_to_empty_falls_back_to_original():
# A file literally named after a codec still needs a valid output
# name — the original stem wins over an empty scrub.
assert clean_title("x264") == "x264"
@pytest.mark.parametrize(
"tag", ["svt-av1", "he-aac", "dtshd", "mpeg2", "mpeg4", "eac3",
"m2ts", "vc-1", "h.264", "h.265", "svtav1", "truehd"]
)
def test_overlapping_tags_scrub_completely(tag):
# Longest-first alternation: a tag that contains a shorter token
# (`svt-av1` ⊃ `av1`, `he-aac` ⊃ `aac`, `dtshd` ⊃ `dts`) must be
# consumed whole, leaving no partial residue behind.
assert clean_title(f"clip.{tag}.name") == "clip.name"
def test_token_table_has_no_duplicates():
assert len(TITLE_ARTIFACT_TOKENS) == len(set(TITLE_ARTIFACT_TOKENS))