2022-01-03 10:56:15 +03:00
|
|
|
#!/usr/bin/env python3
|
|
|
|
|
|
|
|
|
|
import argparse
|
|
|
|
|
import collections
|
2022-02-09 18:08:06 +03:00
|
|
|
import enum
|
2022-01-03 10:56:15 +03:00
|
|
|
import logging
|
|
|
|
|
import os
|
|
|
|
|
import os.path
|
2022-02-09 18:08:06 +03:00
|
|
|
import pprint
|
2022-01-03 10:56:15 +03:00
|
|
|
import re
|
|
|
|
|
import sys
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
PROCESSED_FILETYPES = (
|
|
|
|
|
"mkv",
|
|
|
|
|
"avi",
|
|
|
|
|
"ts",
|
|
|
|
|
)
|
|
|
|
|
SEPARATORS = r"[() .!,_\[\]]"
|
|
|
|
|
SEPARATORS_HYPHEN = r"[\-" + SEPARATORS[1:]
|
|
|
|
|
LANGUAGES = r"(rus|eng|ukr|jap|ita|chi|kor|ger|fre|spa|pol)"
|
|
|
|
|
PATTERNS = (
|
2026-07-14 21:20:26 -07:00
|
|
|
("season", r"s\d{1,2}"),
|
|
|
|
|
("episode", r"(s\d{1,2})?e\d{1,2}"),
|
2022-01-03 10:56:15 +03:00
|
|
|
("year", r"(19|20)\d{2}"),
|
|
|
|
|
("edition", r"((theatrical|director'*s|extended|un)[-.]?cut"
|
|
|
|
|
r"|imax[-.]edition"
|
|
|
|
|
r"|noir[-.]edition"
|
2022-01-12 23:24:10 +03:00
|
|
|
r"|black[-.]chrome[-.]edition"
|
2022-01-09 22:09:02 +03:00
|
|
|
r"|extended[-.]edition"
|
2022-01-16 09:11:00 +03:00
|
|
|
r"|hq[-.]edition"
|
2022-01-03 10:56:15 +03:00
|
|
|
r"|theatrical)"),
|
|
|
|
|
("restrictions", r"(unrated)"),
|
|
|
|
|
("resolution", r"[0-9]{3,4}[pi]"),
|
|
|
|
|
("quality", r"((blu[-.]?ray|bd)[-.]?remux"
|
|
|
|
|
r"|(blu[-.]?ray|bd|uhd|hd(dvd|tv)?|web([-.]?dl)?|dvd)[-.]?rip"
|
|
|
|
|
r"|web[-.]?dl|blu[-.]?ray|hdtv|hddvd|dvd(9)?|f-hd|uhd|remastered"
|
|
|
|
|
r"|amzn)"),
|
|
|
|
|
("codec", r"([hx]\.?26[45]|(mpeg4-)?avc|hevc(10)?|xvid|divx)"),
|
|
|
|
|
("hdr", r"(hdr(10)?|10bit)"),
|
|
|
|
|
("audio", r"%s?(dts(-es)?|ac3|flac|dd5\.1|aac2\.0|dub-line)" % LANGUAGES),
|
|
|
|
|
("subtitles", r"%s?sub" % LANGUAGES),
|
|
|
|
|
("language", r"(\d{1,2}x)?%s" % LANGUAGES),
|
2022-01-16 09:11:00 +03:00
|
|
|
("file_extension", r"mkv|avi"),
|
2022-01-03 10:56:15 +03:00
|
|
|
("unknown", r".*")
|
|
|
|
|
)
|
2026-07-14 21:20:26 -07:00
|
|
|
VALID_CHUNK_TYPES = frozenset({k for k, _ in PATTERNS} | {"name", "episode_name"})
|
2022-01-03 10:56:15 +03:00
|
|
|
|
2022-02-09 18:08:06 +03:00
|
|
|
|
|
|
|
|
# noinspection PyInterpreter
|
|
|
|
|
class EnumAction(argparse.Action):
|
|
|
|
|
"""
|
|
|
|
|
Argparse action for handling Enums
|
|
|
|
|
"""
|
|
|
|
|
def __init__(self, **kwargs):
|
|
|
|
|
# Pop off the type value
|
|
|
|
|
enum_type = kwargs.pop("type", None)
|
|
|
|
|
|
|
|
|
|
# Ensure an Enum subclass is provided
|
|
|
|
|
if enum_type is None:
|
|
|
|
|
raise ValueError("type must be assigned an Enum when using EnumAction")
|
|
|
|
|
if not issubclass(enum_type, enum.Enum):
|
|
|
|
|
raise TypeError("type must be an Enum when using EnumAction")
|
|
|
|
|
|
|
|
|
|
# Generate choices from the Enum
|
|
|
|
|
kwargs.setdefault("choices", tuple(e.value for e in enum_type))
|
|
|
|
|
|
|
|
|
|
super(EnumAction, self).__init__(**kwargs)
|
|
|
|
|
|
|
|
|
|
self._enum = enum_type
|
|
|
|
|
|
|
|
|
|
def __call__(self, parser, namespace, values, option_string=None):
|
|
|
|
|
# Convert value back into an Enum
|
|
|
|
|
value = self._enum(values)
|
|
|
|
|
setattr(namespace, self.dest, value)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
class CliAction(enum.Enum):
|
|
|
|
|
parse = "parse"
|
|
|
|
|
rename = "rename"
|
|
|
|
|
|
|
|
|
|
|
2022-01-03 10:56:15 +03:00
|
|
|
_lg = logging.getLogger("spqr.movie-renamer")
|
|
|
|
|
|
|
|
|
|
|
2026-07-14 21:20:26 -07:00
|
|
|
def _parse_set_arg(value):
|
|
|
|
|
""" Parse --set TYPE=VALUE into (chunk_type, chunk_value). """
|
|
|
|
|
chunk_type, sep, chunk_value = value.partition("=")
|
|
|
|
|
if not sep:
|
|
|
|
|
raise argparse.ArgumentTypeError("expected TYPE=VALUE, got %r" % value)
|
|
|
|
|
if chunk_type not in VALID_CHUNK_TYPES:
|
|
|
|
|
raise argparse.ArgumentTypeError(
|
|
|
|
|
"unknown chunk type %r, expected one of: %s"
|
|
|
|
|
% (chunk_type, ", ".join(sorted(VALID_CHUNK_TYPES)))
|
|
|
|
|
)
|
|
|
|
|
return chunk_type, chunk_value
|
|
|
|
|
|
|
|
|
|
|
2022-01-03 10:56:15 +03:00
|
|
|
def main():
|
|
|
|
|
parser = argparse.ArgumentParser(description="Rename media files.")
|
2022-02-09 18:08:06 +03:00
|
|
|
parser.add_argument("action", type=CliAction, action=EnumAction, metavar="ACTION",
|
|
|
|
|
help="what to do with media file/directory (%(choices)s)")
|
|
|
|
|
parser.add_argument("target", type=str, metavar="TARGET",
|
2022-01-03 10:56:15 +03:00
|
|
|
help="path to the media file/directory")
|
2026-07-14 21:20:26 -07:00
|
|
|
parser.add_argument("--set", dest="overrides", type=_parse_set_arg, action="append",
|
|
|
|
|
default=[], metavar="TYPE=VALUE",
|
|
|
|
|
help="override a parsed chunk, e.g. --set name=Foo.Bar (repeatable)")
|
|
|
|
|
parser.add_argument("--season", type=int, default=None, metavar="N",
|
|
|
|
|
help="season number; auto-numbers episodes (E01, E02, ...) "
|
|
|
|
|
"across media files found under TARGET")
|
2022-01-03 10:56:15 +03:00
|
|
|
parser.add_argument("-v", "--verbose", action="store_true", default=False,
|
|
|
|
|
help="verbose output")
|
|
|
|
|
args = parser.parse_args()
|
|
|
|
|
|
|
|
|
|
loglevel = logging.DEBUG if args.verbose else logging.INFO
|
|
|
|
|
logging.basicConfig(level=loglevel)
|
|
|
|
|
|
2026-07-14 21:20:26 -07:00
|
|
|
process_path(args.action, args.target, overrides=dict(args.overrides), season=args.season)
|
2022-01-03 10:56:15 +03:00
|
|
|
|
|
|
|
|
return 0
|
|
|
|
|
|
|
|
|
|
|
2026-07-14 21:20:26 -07:00
|
|
|
def _is_media_file(path):
|
|
|
|
|
if os.path.isdir(path):
|
|
|
|
|
return False
|
|
|
|
|
ext = os.path.splitext(path)[1][1:]
|
|
|
|
|
return ext.lower() in PROCESSED_FILETYPES
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def process_path(action: CliAction, path, overrides=None, season=None, episode=None):
|
|
|
|
|
overrides = overrides or {}
|
|
|
|
|
|
2022-02-09 18:08:06 +03:00
|
|
|
# process only files
|
|
|
|
|
if os.path.isdir(path):
|
2026-07-14 21:20:26 -07:00
|
|
|
episode_num = 0
|
|
|
|
|
for child_name in sorted(os.listdir(path)):
|
|
|
|
|
child_path = os.path.join(path, child_name)
|
|
|
|
|
child_episode = None
|
|
|
|
|
if season is not None and _is_media_file(child_path):
|
|
|
|
|
episode_num += 1
|
|
|
|
|
child_episode = episode_num
|
|
|
|
|
process_path(action, child_path, overrides=overrides, season=season,
|
|
|
|
|
episode=child_episode)
|
|
|
|
|
return
|
|
|
|
|
|
|
|
|
|
if not _is_media_file(path):
|
|
|
|
|
_lg.debug("Extension is not supported: %s", path)
|
|
|
|
|
return
|
|
|
|
|
|
|
|
|
|
file_overrides = overrides
|
|
|
|
|
if season is not None:
|
|
|
|
|
file_overrides = dict(overrides)
|
|
|
|
|
file_overrides["season"] = "S%02d" % season
|
|
|
|
|
file_overrides["episode"] = "E%02d" % (episode if episode is not None else 1)
|
2022-02-09 18:08:06 +03:00
|
|
|
|
|
|
|
|
# split filepath to dir path, title, and extension
|
|
|
|
|
dir_path, fname = os.path.split(path)
|
|
|
|
|
title, ext = os.path.splitext(fname)
|
|
|
|
|
ext = ext[1:]
|
|
|
|
|
|
|
|
|
|
parsed_title = parse_title(title)
|
2026-07-14 21:20:26 -07:00
|
|
|
if "name" in file_overrides and "name" not in parsed_title:
|
|
|
|
|
# nothing else in the title was recognized, so the whole thing
|
|
|
|
|
# landed in "unknown" instead of "name" -- the override replaces it
|
|
|
|
|
parsed_title.pop("unknown", None)
|
|
|
|
|
for chunk_type, chunk_value in file_overrides.items():
|
|
|
|
|
parsed_title[chunk_type] = [chunk_value]
|
|
|
|
|
|
2022-02-09 18:08:06 +03:00
|
|
|
if action == CliAction.parse:
|
|
|
|
|
print_parsed_title(title, parsed_title)
|
|
|
|
|
return
|
|
|
|
|
|
|
|
|
|
if action == CliAction.rename:
|
|
|
|
|
pretty_title = generate_pretty_name(parsed_title)
|
|
|
|
|
pretty_title += ".%s" % ext
|
|
|
|
|
if pretty_title != fname:
|
|
|
|
|
_lg.warning("%s -> %s", fname, pretty_title)
|
2026-07-14 21:20:26 -07:00
|
|
|
os.rename(path, os.path.join(dir_path, pretty_title))
|
2022-02-09 18:08:06 +03:00
|
|
|
return
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def print_parsed_title(title, parsed):
|
|
|
|
|
print(title)
|
|
|
|
|
pprint.pprint(parsed, indent=4)
|
2022-01-03 10:56:15 +03:00
|
|
|
|
|
|
|
|
|
2022-02-08 23:45:55 +03:00
|
|
|
def generate_pretty_name(parsed_title):
|
|
|
|
|
""" Create file name from parsed chunks. """
|
|
|
|
|
chunk_order = [k for k, _ in PATTERNS]
|
|
|
|
|
chunk_order = ["name"] + chunk_order
|
|
|
|
|
ep_idx = chunk_order.index("episode") + 1
|
|
|
|
|
chunk_order = chunk_order[:ep_idx] + ["episode_name"] + chunk_order[ep_idx:]
|
|
|
|
|
|
|
|
|
|
result = []
|
2026-07-14 21:20:26 -07:00
|
|
|
pending_season = None
|
2022-02-08 23:45:55 +03:00
|
|
|
for chunk_type in chunk_order:
|
2026-07-14 21:20:26 -07:00
|
|
|
chunk_values = parsed_title.get(chunk_type, [])
|
|
|
|
|
if not chunk_values:
|
|
|
|
|
continue
|
|
|
|
|
chunk_str = ".".join(chunk_values)
|
|
|
|
|
if chunk_type == "season":
|
|
|
|
|
pending_season = chunk_str
|
2022-02-08 23:45:55 +03:00
|
|
|
continue
|
2026-07-14 21:20:26 -07:00
|
|
|
if chunk_type == "episode" and pending_season is not None:
|
|
|
|
|
chunk_str = pending_season + chunk_str
|
|
|
|
|
pending_season = None
|
|
|
|
|
result.append(chunk_str)
|
|
|
|
|
if pending_season is not None:
|
|
|
|
|
result.append(pending_season)
|
|
|
|
|
return ".".join(result)
|
2022-02-08 23:45:55 +03:00
|
|
|
|
|
|
|
|
|
2022-01-16 09:11:00 +03:00
|
|
|
def _get_parsed_title_dict(chunk_list, chunk_map):
|
2022-02-08 23:45:55 +03:00
|
|
|
""" Get {chunk_type: [chunk_value_1, ..., chunk_value_n]} dictionary. """
|
2022-01-16 09:11:00 +03:00
|
|
|
p_title = collections.defaultdict(list)
|
|
|
|
|
for idx, chunk in enumerate(chunk_list):
|
|
|
|
|
chunk_type = chunk_map[idx]
|
|
|
|
|
p_title[chunk_type].append(chunk)
|
|
|
|
|
return p_title
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _guess_combined(chunk_values, chunk_map):
|
2022-02-08 23:45:55 +03:00
|
|
|
""" Try to combine unknown chunks in pairs and parse them. """
|
2022-01-16 09:11:00 +03:00
|
|
|
is_changed = False
|
|
|
|
|
p_title = _get_parsed_title_dict(chunk_values, chunk_map)
|
|
|
|
|
if len(p_title["unknown"]) < 2:
|
|
|
|
|
return is_changed, chunk_values, chunk_map
|
|
|
|
|
|
|
|
|
|
# i - begin of slice, j - end of slice
|
|
|
|
|
i = 0
|
|
|
|
|
# process up to second-to-last element
|
|
|
|
|
while i < len(chunk_map) - 1:
|
|
|
|
|
# we need slice with at least two elements
|
|
|
|
|
j = i + 2
|
|
|
|
|
# we need only unknown elements
|
|
|
|
|
while set(chunk_map[i:j]) == {"unknown"} and j <= len(chunk_map):
|
|
|
|
|
# create combined chunk
|
|
|
|
|
cmb_chunk = ".".join(chunk_values[i:j])
|
|
|
|
|
cmb_chunk_type = guess_part(cmb_chunk)
|
|
|
|
|
|
|
|
|
|
# add new combined chunk in lists
|
|
|
|
|
# first subelement gets new chunk, rest - None
|
|
|
|
|
# (will be removed later)
|
|
|
|
|
if cmb_chunk_type != "unknown":
|
|
|
|
|
is_changed = True
|
|
|
|
|
chunk_values[i] = cmb_chunk
|
|
|
|
|
chunk_map[i] = cmb_chunk_type
|
|
|
|
|
for idx in range(i+1, j):
|
|
|
|
|
chunk_values[idx] = None
|
|
|
|
|
chunk_map[idx] = None
|
|
|
|
|
# to start checking next chunks right after the end of slice
|
|
|
|
|
i = idx
|
|
|
|
|
break
|
|
|
|
|
# try add more elements to combined chunk
|
|
|
|
|
else:
|
|
|
|
|
j += 1
|
|
|
|
|
|
|
|
|
|
# start checking next value
|
|
|
|
|
i += 1
|
|
|
|
|
|
|
|
|
|
# clean up from None values
|
|
|
|
|
chunk_values = list(filter(None, chunk_values))
|
|
|
|
|
chunk_map = list(filter(None, chunk_map))
|
|
|
|
|
|
|
|
|
|
return is_changed, chunk_values, chunk_map
|
|
|
|
|
|
|
|
|
|
|
2022-01-03 10:56:15 +03:00
|
|
|
def parse_title(title):
|
|
|
|
|
""" Split media title to components. """
|
|
|
|
|
|
2026-07-14 21:20:26 -07:00
|
|
|
# split title by separators
|
2022-01-16 09:11:00 +03:00
|
|
|
chunk_values = filter(None, re.split(SEPARATORS, title))
|
2022-01-03 10:56:15 +03:00
|
|
|
|
2022-01-16 09:16:28 +03:00
|
|
|
# remove non-word chunks (like single hyphens), but leave ampersands (&)
|
|
|
|
|
chunk_values = list(filter(lambda ch: re.search(r"(\w|&)+", ch), chunk_values))
|
2022-01-09 22:09:02 +03:00
|
|
|
|
2022-02-08 23:45:55 +03:00
|
|
|
chunk_map = [] # list of chunk_types
|
2022-01-03 10:56:15 +03:00
|
|
|
# parse each chunk
|
2022-01-16 09:11:00 +03:00
|
|
|
for ch_value in chunk_values:
|
|
|
|
|
chunk_map.append(guess_part(ch_value))
|
|
|
|
|
|
|
|
|
|
_, chunk_values, chunk_map = _guess_combined(chunk_values, chunk_map)
|
|
|
|
|
|
2022-02-08 23:45:55 +03:00
|
|
|
# try to parse unknown chunks, replacing all hyphens in them with dots
|
2022-01-16 09:11:00 +03:00
|
|
|
p_title = _get_parsed_title_dict(chunk_values, chunk_map)
|
|
|
|
|
is_changed = False
|
|
|
|
|
if p_title.get("unknown"):
|
|
|
|
|
spl_ch_values = []
|
|
|
|
|
spl_ch_map = []
|
|
|
|
|
for idx, ch_value in enumerate(chunk_values):
|
|
|
|
|
ch_type = chunk_map[idx]
|
|
|
|
|
if ch_type == "unknown" and "-" in ch_value:
|
|
|
|
|
spl_values = ch_value.split("-")
|
|
|
|
|
for spl_val in spl_values:
|
|
|
|
|
if not spl_val:
|
|
|
|
|
continue
|
|
|
|
|
spl_type = guess_part(spl_val)
|
|
|
|
|
if spl_type != "unknown":
|
|
|
|
|
is_changed = True
|
|
|
|
|
spl_ch_values.append(spl_val)
|
|
|
|
|
spl_ch_map.append(spl_type)
|
|
|
|
|
else:
|
|
|
|
|
spl_ch_values.append(ch_value)
|
|
|
|
|
spl_ch_map.append(ch_type)
|
|
|
|
|
|
|
|
|
|
is_combined, spl_ch_values, spl_ch_map = _guess_combined(spl_ch_values, spl_ch_map)
|
|
|
|
|
if is_changed or is_combined:
|
|
|
|
|
chunk_values = spl_ch_values
|
|
|
|
|
chunk_map = spl_ch_map
|
|
|
|
|
|
|
|
|
|
# parse name and episode name
|
|
|
|
|
# only if there is something except unknown chunks
|
|
|
|
|
p_title = _get_parsed_title_dict(chunk_values, chunk_map)
|
|
|
|
|
if len(p_title["unknown"]) != len(chunk_values):
|
|
|
|
|
idx = 0
|
|
|
|
|
while idx < len(chunk_map) and chunk_map[idx] == "unknown":
|
|
|
|
|
chunk_map[idx] = "name"
|
|
|
|
|
idx += 1
|
2026-07-14 21:20:26 -07:00
|
|
|
# if season/episode number is found, next unknown chunks are episode name
|
|
|
|
|
if p_title.get("season") or p_title.get("episode"):
|
|
|
|
|
if p_title.get("episode"):
|
|
|
|
|
idx = chunk_map.index("episode") + 1
|
|
|
|
|
else:
|
|
|
|
|
idx = chunk_map.index("season") + 1
|
2022-01-16 09:11:00 +03:00
|
|
|
while idx < len(chunk_map) and chunk_map[idx] == "unknown":
|
|
|
|
|
chunk_map[idx] = "episode_name"
|
|
|
|
|
idx += 1
|
|
|
|
|
|
|
|
|
|
# at last, strip hyphens from unknown chunks
|
|
|
|
|
# only if there is something except unknown chunks
|
|
|
|
|
p_title = _get_parsed_title_dict(chunk_values, chunk_map)
|
|
|
|
|
if len(p_title["unknown"]) != len(chunk_values):
|
|
|
|
|
for idx, chunk_type in enumerate(chunk_map):
|
|
|
|
|
if chunk_type != "unknown":
|
2022-01-09 22:09:02 +03:00
|
|
|
continue
|
2022-01-16 09:11:00 +03:00
|
|
|
chunk_value = chunk_values[idx]
|
|
|
|
|
if chunk_value[0] != "-" and chunk_value[-1] != "-":
|
2022-01-09 22:09:02 +03:00
|
|
|
continue
|
2022-01-16 09:11:00 +03:00
|
|
|
chunk_values[idx] = chunk_value.strip("-")
|
2022-01-09 22:09:02 +03:00
|
|
|
|
2026-07-14 21:20:26 -07:00
|
|
|
# split glued SxxExx episode tokens (e.g. "S04E06") into separate
|
|
|
|
|
# season and episode chunks
|
2022-01-16 09:11:00 +03:00
|
|
|
p_title = _get_parsed_title_dict(chunk_values, chunk_map)
|
2026-07-14 21:20:26 -07:00
|
|
|
if p_title.get("episode"):
|
|
|
|
|
bare_episodes = []
|
|
|
|
|
for ep_value in p_title["episode"]:
|
|
|
|
|
if ep_value[0].lower() != "s":
|
|
|
|
|
bare_episodes.append(ep_value)
|
|
|
|
|
continue
|
|
|
|
|
season, episode = _split_combined_episode(ep_value)
|
|
|
|
|
p_title["season"].append(season)
|
|
|
|
|
bare_episodes.append(episode)
|
|
|
|
|
p_title["episode"] = bare_episodes
|
|
|
|
|
|
2022-01-09 22:09:02 +03:00
|
|
|
return dict(p_title)
|
2022-01-03 10:56:15 +03:00
|
|
|
|
|
|
|
|
|
2026-07-14 21:20:26 -07:00
|
|
|
def _split_combined_episode(chunk_value):
|
|
|
|
|
""" Split a glued SxxExx token like "S04E06" into ("S04", "E06"). """
|
|
|
|
|
match = re.match(r"(s\d{1,2})(e\d{1,2})$", chunk_value, flags=re.I)
|
|
|
|
|
return match.group(1), match.group(2)
|
|
|
|
|
|
|
|
|
|
|
2022-02-08 23:45:55 +03:00
|
|
|
def guess_part(chunk_value):
|
|
|
|
|
""" Return chunk type for given chunk value. """
|
|
|
|
|
for chunk_type, pattern in PATTERNS:
|
2022-01-03 10:56:15 +03:00
|
|
|
full_match_pat = r"^" + pattern + r"$"
|
2022-02-08 23:45:55 +03:00
|
|
|
if re.match(full_match_pat, chunk_value, flags=re.I):
|
|
|
|
|
return chunk_type
|
2022-01-03 10:56:15 +03:00
|
|
|
raise RuntimeError("unhandled pattern type")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
if __name__ == "__main__":
|
|
|
|
|
sys.exit(main())
|