Files
playground/file_renamer.py

314 lines
9.4 KiB
Python
Executable File

#!/usr/bin/env python3
import argparse
import dataclasses
import datetime
import json
import logging
import re
import subprocess
import typing
from pathlib import Path
script_path = Path(__file__)
db_filename = script_path.with_name(f"{script_path.stem}.known.txt")
def save_actors(actors: typing.Iterable[str]):
existing = load_actors()
all_actors = sorted({*existing, *actors})
with db_filename.open("w", encoding="utf-8", newline="\n") as f:
f.writelines(all_actors)
def load_actors() -> set[str]:
try:
with db_filename.open(encoding="utf-8", newline="\n") as f:
return {line.strip() for line in f}
except FileNotFoundError:
return set()
re_date_only = re.compile(r"\b(\d{4})\b") # 2024
re_date_us = re.compile(r"\b(\d{2})\D(\d{2})\D(\d{2})\b") # 12/31/21
re_date_iso = re.compile(r"\b(\d{4})\D?(\d{2})\D?(\d{2})\b") # 2020-12-31
re_date_iso_short = re.compile(r"\b(\d{2})\D?(\d{2})\D?(\d{2})\b") # 20-12-31"'
re_date_iso_rev = re.compile(r"\b(\d{2})\D?(\d{2})\D?(\d{4})\b") # 31-12-2020
re_date_iso_rev_short = re.compile(r"\b(\d{2})\D(\d{2})\D(\d{2})\b") # 31.12.21
class Pattern:
def __init__(self, pattern: re.Pattern):
self.re = pattern if isinstance(pattern, re.Pattern) else re.compile(pattern)
def __eq__(self, other):
if isinstance(other, str):
return self.re.search(other)
raise ValueError
@dataclasses.dataclass
class ParsedDate:
date: datetime.date
start: int
end: int
def parse_date(filename: str) -> ParsedDate | None:
def from_date_only():
match = re_date_only.search(filename)
y = int(match.group(1))
return ParsedDate(datetime.date(y, 1, 1), start=match.start(), end=match.end())
def from_us_format():
match = re_date_us.search(filename)
m, d, y = map(int, match.groups())
return ParsedDate(
datetime.date(y + 2000, m, d), start=match.start(), end=match.end()
)
def from_iso():
match = re_date_iso.search(filename)
y, m, d = map(int, match.groups())
return ParsedDate(datetime.date(y, m, d), start=match.start(), end=match.end())
def from_iso_short():
match = re_date_iso_short.search(filename)
y, m, d = map(int, match.groups())
return ParsedDate(
datetime.date(y + 2000, m, d), start=match.start(), end=match.end()
)
def from_iso_reversed():
match = re_date_iso_rev.search(filename)
d, m, y = map(int, match.groups())
return ParsedDate(datetime.date(y, m, d), start=match.start(), end=match.end())
def from_iso_reversed_short():
match = re_date_iso_rev_short.search(filename)
d, m, y = map(int, match.groups())
return ParsedDate(
datetime.date(y + 2000, m, d), start=match.start(), end=match.end()
)
candidates = []
for fn in [
from_date_only,
from_us_format,
from_iso,
from_iso_short,
from_iso_reversed,
from_iso_reversed_short,
]:
try:
candidates.append(fn())
except (ValueError, AttributeError):
pass
today = datetime.date.today()
future_threshold = today + datetime.timedelta(days=60)
past_threshold = datetime.date(2010, 1, 1)
candidates = [
it for it in candidates if past_threshold < it.date < future_threshold
]
if not candidates:
return
return candidates[0]
@dataclasses.dataclass
class Release:
actors: list[str]
studio: str | None = None
title: str | None = None
released_at: datetime.date | None = None
def to_filename(self):
parts = []
if self.actors:
parts.append(", ".join(self.actors))
if self.studio:
parts.append(f"@{self.studio}")
if self.title:
parts.append(self.title)
if self.released_at:
parts.append(self.released_at.isoformat())
return " -- ".join(parts)
def trash(path: Path):
filename = f'"{path}"'
cmd = [
"osascript",
"-e",
f'tell app "Finder" to move (POSIX file {filename}) to trash',
]
subprocess.run(cmd).check_returncode()
def parse_release(
filename: str, known_actors: set[str] | None = None
) -> Release | None:
if not known_actors:
known_actors = set()
if any(filename.lower().endswith(ext) for ext in [".mp4", ".mkv"]):
filename = re.sub(r"\.\w{3,4}$", "", filename)
def from_own():
remaining = re.sub(r"\s+\[[^]]+]$", "", filename)
remaining = re.sub(r"\s+\[([^]]+|\d+\w)(,\s*[^]]+)?]", "", remaining)
remaining = re.sub(r"\s*--\s*", " -- ", remaining)
match remaining.split(" -- "):
case [actors, studio, title, date] if studio.startswith("@"):
return Release(
actors=sorted(actors.split(", ")),
studio=studio.removeprefix("@"),
title=title,
released_at=datetime.date.fromisoformat(date),
)
case [actors, studio, date] if studio.startswith("@") and (
parsed := parse_date(date)
):
return Release(
actors=sorted(actors.split(", ")),
studio=studio.removeprefix("@"),
title=None,
released_at=parsed.date,
)
case [actors, studio, title] if studio.startswith("@"):
return Release(
actors=sorted(actors.split(", ")),
studio=studio.removeprefix("@"),
title=title,
released_at=None,
)
case [actors, title, date] if parsed := parse_date(date):
return Release(
actors=sorted(actors.split(", ")),
studio=None,
title=title,
released_at=parsed.date,
)
case [actors, studio] if studio.startswith("@"):
return Release(
actors=sorted(actors.split(", ")),
studio=studio,
)
case [actors, title]:
return Release(
actors=sorted(actors.split(", ")),
title=title,
)
def from_prt():
# assert '.PRT' in filename
remaining = re.sub(r"\.(720p|1080p|HEVC|x265|PRT|XXX)", " ", filename)
studio = remaining[: remaining.index(".")]
remaining = remaining[len(studio) :]
parsed_date = parse_date(remaining)
if parsed_date:
remaining = remaining[parsed_date.end :]
remaining = re.sub(r"[. ]+", " ", remaining).strip()
actors = []
title = None
# PornFidelity.E1109.Uma.Jolie
match remaining.split(" "):
case [episode, *names] if episode.startswith("E") and episode[1:].isdigit():
title = episode
actors = [" ".join(names)]
case [a_first, a_last, "And", b_first, b_last]:
actors = [f"{a_first} {a_last}", f"{b_first} {b_last}"]
case [a_first, a_last, "And", b_first, b_last, *rest]:
actors = [f"{a_first} {a_last}", f"{b_first} {b_last}"]
title = " ".join(rest)
case [first, last]:
actors = [f"{first} {last}"]
case [first, last, *rest]:
actors = [f"{first} {last}"]
title = " ".join(rest)
case [first]:
actors = [first]
return Release(
studio=studio,
actors=sorted(actors),
title=title,
released_at=parsed_date.date if parsed_date else None,
)
for fn in [
from_own,
from_prt,
]:
try:
if res := fn():
return res
except (ValueError, AttributeError, AssertionError):
pass
def parse_args():
arger = argparse.ArgumentParser()
arger.add_argument(
"filenames", nargs="+", type=lambda v: Path(v).resolve(), help="Filenames"
)
return arger.parse_args()
@dataclasses.dataclass
class MoveOp:
source: Path
target: Path
def to_dict(self) -> dict:
return {
"source": str(self.source),
"target": str(self.target),
}
def write_undo_script(save_dir: Path, ops: list[MoveOp]):
now = datetime.datetime.now().isoformat().replace(":", "")
save_path = save_dir / f"undo_{now}.json"
with save_path.open("w", encoding="utf-8") as f:
for it in ops:
f.write(json.dumps(it.to_dict()) + "\n")
def main():
actors = load_actors()
args = parse_args()
paths: list[Path] = args.filenames
jobs = []
for it in paths:
if parsed := parse_release(it.name, known_actors=actors):
logging.debug(f"filename={it} release={parsed}")
suggested = parsed.to_filename()
target = it.with_stem(suggested)
if target.is_file():
logging.error("target already exists")
continue
jobs.append(MoveOp(source=it, target=target))
if not jobs:
return
# write_undo_script(paths[0].parent, jobs)
for it in jobs:
it.target.hardlink_to(it.source)
for it in jobs:
trash(it.source)
if __name__ == "__main__":
logging.basicConfig(level=logging.DEBUG)
main()