1693 lines
59 KiB
Python
1693 lines
59 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Photo collection metadata cleanup.
|
|
|
|
Drag files or directories onto this script, or run:
|
|
python3 photo_metadata.py path/to/file/or/directory [...]
|
|
|
|
The script reads metadata with exiftool, builds a full preview, then edits files in
|
|
place only after confirmation.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import math
|
|
import os
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import sys
|
|
import xml.etree.ElementTree as ET
|
|
from collections import defaultdict
|
|
from dataclasses import dataclass, field
|
|
from datetime import datetime, timedelta, timezone
|
|
from pathlib import Path
|
|
from typing import Iterable
|
|
|
|
from tools.console import pause_if_interactive, prompt_input
|
|
from tools.exiftool import require_exiftool, run_exiftool_json, run_exiftool_write
|
|
from tools.filenames import (
|
|
format_filename_stem,
|
|
group_stem_from_name,
|
|
naive_wall_time,
|
|
normalized_extension,
|
|
)
|
|
from tools.filesystem import (
|
|
cleanup_roots_from_args,
|
|
collect_files,
|
|
determine_working_dir,
|
|
display_path_from_working_dir,
|
|
remove_empty_dirs,
|
|
unique_existing_target,
|
|
)
|
|
|
|
|
|
IMAGE_EXTS = {".jpg", ".jpeg", ".heic", ".arw"}
|
|
VIDEO_EXTS = {".mp4", ".mov", ".mts"}
|
|
SUPPORTED_EXTS = IMAGE_EXTS | VIDEO_EXTS
|
|
|
|
IMAGE_TIME_TAGS = (
|
|
"EXIF:DateTimeOriginal",
|
|
"EXIF:CreateDate",
|
|
"EXIF:DateTime",
|
|
"EXIF:ModifyDate",
|
|
"XMP:DateTimeOriginal",
|
|
"XMP:CreateDate",
|
|
)
|
|
IMAGE_OFFSET_TAGS = (
|
|
"EXIF:OffsetTimeOriginal",
|
|
"EXIF:OffsetTimeDigitized",
|
|
"EXIF:OffsetTime",
|
|
)
|
|
VIDEO_TIME_TAGS = (
|
|
"QuickTime:MediaCreateDate",
|
|
"QuickTime:TrackCreateDate",
|
|
"QuickTime:CreateDate",
|
|
"QuickTime:MediaModifyDate",
|
|
"QuickTime:TrackModifyDate",
|
|
"QuickTime:ModifyDate",
|
|
"XMP:DateTimeOriginal",
|
|
"XMP:CreateDate",
|
|
)
|
|
VIDEO_BEGIN_TIME_TAGS = (
|
|
"H264:DateTimeOriginal",
|
|
)
|
|
VIDEO_DURATION_TAGS = (
|
|
"QuickTime:Duration",
|
|
"Composite:Duration",
|
|
"File:Duration",
|
|
"M2TS:Duration",
|
|
)
|
|
SEQUENCE_TAGS = (
|
|
"MakerNotes:SequenceNumber",
|
|
"MakerNotes:SequenceNumberOriginal",
|
|
"MakerNotes:ImageNumber",
|
|
"EXIF:ImageNumber",
|
|
"SequenceNumber",
|
|
)
|
|
|
|
DATETIME_RE = re.compile(
|
|
r"(?P<Y>\d{4})[:\-]?(?P<M>\d{2})[:\-]?(?P<D>\d{2})"
|
|
r"(?:[ T_])?"
|
|
r"(?P<h>\d{2}):?(?P<m>\d{2}):?(?P<s>\d{2})"
|
|
r"(?:[.,](?P<sub>\d+))?"
|
|
r"(?:\s*(?P<tz>Z|[+-]\d{2}:?\d{2}))?"
|
|
)
|
|
TIME_ONLY_RE = re.compile(
|
|
r"^(?P<h>\d{1,2})(?::?(?P<m>\d{2}))(?::?(?P<s>\d{2}))?$"
|
|
)
|
|
TZ_RE = re.compile(r"^(?P<sign>[+-])(?P<h>\d{1,2})(?::?(?P<m>\d{2}))?$")
|
|
OFFSET_RE = re.compile(r"^([+-])(?:(\d+):)?(\d{1,2})(?::(\d{2}))?$")
|
|
TOKEN_SHIFT_RE = re.compile(r"(\d+|[a-zA-Z]+)")
|
|
ORDERED_TIMESTAMP_STEM_RE = re.compile(r"^(?P<timestamp>\d{8}_\d{6})-[1-9]\d*$")
|
|
TIMESTAMP_RANGE_STEM_RE = re.compile(
|
|
r"^(?P<start>\d{8}_\d{6})-(?P<end>\d{8}_\d{6})(?P<suffix>_.*)?$"
|
|
)
|
|
CAMERA_RANGE_STEM_RE = re.compile(
|
|
r"^(?P<prefix>[A-Za-z]*)(?P<first>\d+)-(?P=prefix)(?P<last>\d+)(?P<suffix>_.*)$"
|
|
)
|
|
|
|
|
|
@dataclass
|
|
class XmlSidecar:
|
|
path: Path
|
|
creation: datetime | None = None
|
|
duration_seconds: float | None = None
|
|
timezone_value: timezone | None = None
|
|
device: str | None = None
|
|
|
|
|
|
@dataclass
|
|
class TagWrite:
|
|
label: str
|
|
arg: str
|
|
current_value: str | None
|
|
new_value: str
|
|
will_write: bool
|
|
|
|
|
|
@dataclass
|
|
class MediaRecord:
|
|
path: Path
|
|
metadata: dict
|
|
kind: str
|
|
original_time: datetime | None = None
|
|
adjusted_time: datetime | None = None
|
|
timezone_value: timezone | None = None
|
|
subsec: int | None = None
|
|
inferred_subsec: int | None = None
|
|
sequence: int | None = None
|
|
camera_key: str = ""
|
|
duration_seconds: float | None = None
|
|
video_time_is_beginning: bool = False
|
|
parsed_from_filename: bool = False
|
|
sidecar: XmlSidecar | None = None
|
|
new_stem: str | None = None
|
|
final_name: str | None = None
|
|
group_id: int | None = None
|
|
group_name: str | None = None
|
|
target_path: Path | None = None
|
|
target_sidecar: Path | None = None
|
|
rename_skip_reason: str | None = None
|
|
write_plan: list[TagWrite] = field(default_factory=list)
|
|
warnings: list[str] = field(default_factory=list)
|
|
|
|
@property
|
|
def is_image(self) -> bool:
|
|
return self.kind == "image"
|
|
|
|
@property
|
|
def is_video(self) -> bool:
|
|
return self.kind == "video"
|
|
|
|
|
|
@dataclass
|
|
class PtoPlan:
|
|
path: Path
|
|
target_path: Path
|
|
referenced_records: list[MediaRecord]
|
|
updated_text: str
|
|
replacements: list[tuple[str, str]]
|
|
|
|
|
|
@dataclass
|
|
class PtoReference:
|
|
path: Path
|
|
original_text: str
|
|
records: list[MediaRecord]
|
|
|
|
|
|
@dataclass
|
|
class UserChoices:
|
|
working_dir: Path
|
|
time_offset: timedelta | None
|
|
explicit_timezone: timezone | None
|
|
authoritative_timezone: timezone | None
|
|
write_missing_photo_timezones: bool
|
|
video_filename_timezone: timezone | None
|
|
artist_action: str
|
|
artist_value: str | None
|
|
rename_files: bool
|
|
group_photos: bool
|
|
group_min_size: int
|
|
move_to_working_dir: bool
|
|
process_pto_files: bool
|
|
apply_filename_timestamps: bool
|
|
|
|
|
|
def ask_yes_no(prompt: str, default: bool) -> bool:
|
|
suffix = "Y/n" if default else "y/N"
|
|
while True:
|
|
answer = prompt_input(f"{prompt} ({suffix}) ").strip().lower()
|
|
if not answer:
|
|
return default
|
|
if answer in {"y", "yes"}:
|
|
return True
|
|
if answer in {"n", "no"}:
|
|
return False
|
|
print("Please answer y or n.")
|
|
|
|
|
|
def ask_group_min_size() -> int:
|
|
while True:
|
|
answer = prompt_input(
|
|
"Create automatic burst/HDR groups? Enter=yes, n=no, or minimum group size [2]: "
|
|
).strip().lower()
|
|
if not answer or answer in {"y", "yes"}:
|
|
return 2
|
|
if answer in {"n", "no"}:
|
|
return 0
|
|
if answer.isdigit():
|
|
value = int(answer)
|
|
if value <= 0:
|
|
return 0
|
|
return max(value, 2)
|
|
print("Please answer y, n, or an integer minimum group size.")
|
|
|
|
|
|
def parse_timezone_offset(value: str) -> timezone:
|
|
match = TZ_RE.match(value.strip())
|
|
if not match:
|
|
raise ValueError("timezone must look like +09:00 or -05:30")
|
|
sign = 1 if match.group("sign") == "+" else -1
|
|
hours = int(match.group("h"))
|
|
minutes = int(match.group("m") or "0")
|
|
if minutes not in {0, 15, 30, 45}:
|
|
raise ValueError("timezone minutes must be 00, 15, 30, or 45")
|
|
return timezone(sign * timedelta(hours=hours, minutes=minutes))
|
|
|
|
|
|
def timezone_to_string(tz: timezone) -> str:
|
|
offset = tz.utcoffset(None)
|
|
if offset is None:
|
|
return "+00:00"
|
|
total_minutes = int(offset.total_seconds() // 60)
|
|
sign = "+" if total_minutes >= 0 else "-"
|
|
total_minutes = abs(total_minutes)
|
|
return f"{sign}{total_minutes // 60:02d}:{total_minutes % 60:02d}"
|
|
|
|
|
|
def parse_datetime(value: str, default_date: datetime | None = None) -> datetime:
|
|
text = value.strip()
|
|
match = DATETIME_RE.search(text)
|
|
if match:
|
|
tzinfo = None
|
|
tz_text = match.group("tz")
|
|
if tz_text == "Z":
|
|
tzinfo = timezone.utc
|
|
elif tz_text:
|
|
tzinfo = parse_timezone_offset(tz_text)
|
|
return datetime(
|
|
int(match.group("Y")),
|
|
int(match.group("M")),
|
|
int(match.group("D")),
|
|
int(match.group("h")),
|
|
int(match.group("m")),
|
|
int(match.group("s")),
|
|
tzinfo=tzinfo,
|
|
)
|
|
|
|
match = TIME_ONLY_RE.match(text)
|
|
if match and default_date is not None:
|
|
return default_date.replace(
|
|
hour=int(match.group("h")),
|
|
minute=int(match.group("m")),
|
|
second=int(match.group("s") or "0"),
|
|
microsecond=0,
|
|
)
|
|
|
|
raise ValueError("timestamp must include YYYYMMDD_HHMMSS or a time with reference date")
|
|
|
|
|
|
def parse_exif_datetime(value: object, assume_utc: bool = False) -> datetime | None:
|
|
if value is None:
|
|
return None
|
|
try:
|
|
dt = parse_datetime(str(value))
|
|
except ValueError:
|
|
return None
|
|
if assume_utc and dt.tzinfo is None:
|
|
dt = dt.replace(tzinfo=timezone.utc)
|
|
return dt
|
|
|
|
|
|
def parse_subsec(value: object) -> int | None:
|
|
if value is None:
|
|
return None
|
|
match = re.search(r"\d+", str(value))
|
|
if not match:
|
|
return None
|
|
digits = match.group(0)[:3].ljust(3, "0")
|
|
return int(digits)
|
|
|
|
|
|
def parse_sequence(value: object) -> int | None:
|
|
if value is None:
|
|
return None
|
|
match = re.search(r"\d+", str(value))
|
|
if not match:
|
|
return None
|
|
return int(match.group(0))
|
|
|
|
|
|
def parse_duration_seconds(value: object) -> float | None:
|
|
if value is None:
|
|
return None
|
|
text = str(value).strip()
|
|
|
|
number_match = re.match(r"^([0-9]+(?:\.[0-9]+)?)\s*(?:s|sec|seconds)?$", text, re.I)
|
|
if number_match:
|
|
return float(number_match.group(1))
|
|
|
|
hms_match = re.match(
|
|
r"^(?:(?P<h>\d+):)?(?P<m>\d{1,2}):(?P<s>\d{1,2}(?:\.\d+)?)$", text
|
|
)
|
|
if hms_match:
|
|
hours = int(hms_match.group("h") or "0")
|
|
minutes = int(hms_match.group("m"))
|
|
seconds = float(hms_match.group("s"))
|
|
return hours * 3600 + minutes * 60 + seconds
|
|
|
|
embedded = re.search(r"([0-9]+(?:\.[0-9]+)?)\s*s", text, re.I)
|
|
if embedded:
|
|
return float(embedded.group(1))
|
|
|
|
return None
|
|
|
|
|
|
def round_half_up(value: float) -> int:
|
|
return int(math.floor(value + 0.5))
|
|
|
|
|
|
def parse_timeshift(value: str) -> timedelta:
|
|
text = value.strip().replace(" ", "")
|
|
if not text or text[0] not in "+-":
|
|
raise ValueError("time shift must start with + or -")
|
|
sign = 1 if text[0] == "+" else -1
|
|
body = text[1:]
|
|
|
|
if re.fullmatch(r"\d+(?::\d{2}){0,2}", body):
|
|
parts = [int(part) for part in body.split(":")]
|
|
parts = [0] * (3 - len(parts)) + parts
|
|
return sign * timedelta(hours=parts[0], minutes=parts[1], seconds=parts[2])
|
|
|
|
units = {
|
|
"h": 3600,
|
|
"hr": 3600,
|
|
"hrs": 3600,
|
|
"hour": 3600,
|
|
"hours": 3600,
|
|
"m": 60,
|
|
"min": 60,
|
|
"mins": 60,
|
|
"minute": 60,
|
|
"minutes": 60,
|
|
"s": 1,
|
|
"sec": 1,
|
|
"secs": 1,
|
|
"second": 1,
|
|
"seconds": 1,
|
|
}
|
|
tokens = TOKEN_SHIFT_RE.findall(body)
|
|
if not tokens or len(tokens) % 2 != 0:
|
|
raise ValueError("time shift units must look like +1h30m or -2 minutes")
|
|
seconds = 0
|
|
for amount, unit in zip(tokens[::2], tokens[1::2]):
|
|
if not amount.isdigit() or unit.lower() not in units:
|
|
raise ValueError("invalid time shift unit")
|
|
seconds += int(amount) * units[unit.lower()]
|
|
return sign * timedelta(seconds=seconds)
|
|
|
|
|
|
def collect_media_files(args: list[str]) -> list[Path]:
|
|
return collect_files(args, allowed_exts=SUPPORTED_EXTS, print_missing=True)
|
|
|
|
|
|
def collect_pto_files(args: list[str]) -> list[Path]:
|
|
return collect_files(args, allowed_exts={".pto"})
|
|
|
|
|
|
def first_tag(metadata: dict, tags: Iterable[str]) -> object | None:
|
|
for tag in tags:
|
|
if tag in metadata:
|
|
return metadata[tag]
|
|
return None
|
|
|
|
|
|
def tag_value(metadata: dict, tag: str) -> str | None:
|
|
value = metadata.get(tag)
|
|
if value is None:
|
|
return None
|
|
return str(value)
|
|
|
|
|
|
def same_text_value(current: str | None, new_value: str) -> bool:
|
|
return (current or "") == new_value
|
|
|
|
|
|
def same_datetime_value(current: str | None, new_value: str) -> bool:
|
|
if current is None:
|
|
return False
|
|
current_dt = parse_exif_datetime(current)
|
|
new_dt = parse_exif_datetime(new_value)
|
|
if current_dt is None or new_dt is None:
|
|
return current == new_value
|
|
return naive_wall_time(current_dt) == naive_wall_time(new_dt)
|
|
|
|
|
|
def same_subsec_value(current: str | None, new_value: str) -> bool:
|
|
if current is None:
|
|
return False
|
|
current_subsec = parse_subsec(current)
|
|
if current_subsec is None:
|
|
return current == new_value
|
|
return f"{current_subsec:03d}" == new_value
|
|
|
|
|
|
def add_tag_write(
|
|
writes: list[TagWrite],
|
|
label: str,
|
|
arg_name: str,
|
|
current_value: str | None,
|
|
new_value: str,
|
|
same_value=same_text_value,
|
|
) -> None:
|
|
writes.append(
|
|
TagWrite(
|
|
label=label,
|
|
arg=f"-{arg_name}={new_value}",
|
|
current_value=current_value,
|
|
new_value=new_value,
|
|
will_write=not same_value(current_value, new_value),
|
|
)
|
|
)
|
|
|
|
|
|
def find_sidecar(video_path: Path) -> XmlSidecar | None:
|
|
expected = f"{video_path.stem}M01.XML"
|
|
for child in video_path.parent.iterdir():
|
|
if child.is_file() and child.name.lower() == expected.lower():
|
|
return parse_xml_sidecar(child)
|
|
return None
|
|
|
|
|
|
def parse_xml_sidecar(path: Path) -> XmlSidecar:
|
|
sidecar = XmlSidecar(path=path)
|
|
try:
|
|
root = ET.parse(path).getroot()
|
|
except ET.ParseError as exc:
|
|
sidecar.device = f"unreadable XML: {exc}"
|
|
return sidecar
|
|
|
|
duration_value: float | None = None
|
|
fps_value: float | None = None
|
|
|
|
def strip_ns(name: str) -> str:
|
|
return name.rsplit("}", 1)[-1]
|
|
|
|
for element in root.iter():
|
|
name = strip_ns(element.tag)
|
|
if name == "CreationDate":
|
|
value = element.attrib.get("value")
|
|
if value:
|
|
sidecar.creation = parse_exif_datetime(value)
|
|
if sidecar.creation and sidecar.creation.tzinfo:
|
|
sidecar.timezone_value = sidecar.creation.tzinfo
|
|
elif name == "Duration":
|
|
value = element.attrib.get("value")
|
|
if value:
|
|
duration_value = parse_duration_seconds(value)
|
|
elif name == "VideoFrame":
|
|
fps_text = element.attrib.get("captureFps") or element.attrib.get("formatFps")
|
|
if fps_text:
|
|
fps_match = re.search(r"\d+(?:\.\d+)?", fps_text)
|
|
if fps_match:
|
|
fps_value = float(fps_match.group(0))
|
|
elif name == "Device":
|
|
manufacturer = element.attrib.get("manufacturer")
|
|
model = element.attrib.get("modelName")
|
|
sidecar.device = " ".join(part for part in (manufacturer, model) if part)
|
|
|
|
if duration_value is not None and fps_value:
|
|
sidecar.duration_seconds = duration_value / fps_value
|
|
else:
|
|
sidecar.duration_seconds = duration_value
|
|
|
|
return sidecar
|
|
|
|
|
|
def build_records(paths: list[Path], metadata_items: list[dict]) -> list[MediaRecord]:
|
|
by_source = {
|
|
Path(item.get("SourceFile", "")).resolve(): item
|
|
for item in metadata_items
|
|
if item.get("SourceFile")
|
|
}
|
|
records: list[MediaRecord] = []
|
|
|
|
for path in paths:
|
|
metadata = by_source.get(path.resolve(), {})
|
|
ext = path.suffix.lower()
|
|
kind = "video" if ext in VIDEO_EXTS else "image"
|
|
record = MediaRecord(path=path, metadata=metadata, kind=kind)
|
|
|
|
if record.is_image:
|
|
record.original_time = parse_exif_datetime(first_tag(metadata, IMAGE_TIME_TAGS))
|
|
record.adjusted_time = record.original_time
|
|
offset_value = first_tag(metadata, IMAGE_OFFSET_TAGS)
|
|
if offset_value:
|
|
try:
|
|
record.timezone_value = parse_timezone_offset(str(offset_value))
|
|
except ValueError:
|
|
record.warnings.append(f"Invalid timezone offset metadata: {offset_value}")
|
|
record.subsec = parse_subsec(
|
|
first_tag(
|
|
metadata,
|
|
(
|
|
"EXIF:SubSecTimeOriginal",
|
|
"EXIF:SubSecTimeDigitized",
|
|
"EXIF:SubSecTime",
|
|
),
|
|
)
|
|
)
|
|
else:
|
|
begin_time_value = first_tag(metadata, VIDEO_BEGIN_TIME_TAGS)
|
|
if ext == ".mts" and begin_time_value is not None:
|
|
record.original_time = parse_exif_datetime(begin_time_value, assume_utc=True)
|
|
record.video_time_is_beginning = True
|
|
else:
|
|
record.original_time = parse_exif_datetime(
|
|
first_tag(metadata, VIDEO_TIME_TAGS), assume_utc=True
|
|
)
|
|
record.adjusted_time = record.original_time
|
|
record.timezone_value = timezone.utc
|
|
record.duration_seconds = parse_duration_seconds(first_tag(metadata, VIDEO_DURATION_TAGS))
|
|
record.sidecar = find_sidecar(path)
|
|
if record.sidecar and record.duration_seconds is not None:
|
|
xml_duration = record.sidecar.duration_seconds
|
|
if xml_duration is not None and abs(xml_duration - record.duration_seconds) > 1:
|
|
record.warnings.append(
|
|
f"XML duration {xml_duration:g}s differs from metadata duration "
|
|
f"{record.duration_seconds:g}s"
|
|
)
|
|
|
|
record.sequence = parse_sequence(first_tag(metadata, SEQUENCE_TAGS))
|
|
record.camera_key = camera_key(record)
|
|
records.append(record)
|
|
|
|
return records
|
|
|
|
|
|
def camera_key(record: MediaRecord) -> str:
|
|
metadata = record.metadata
|
|
make = first_tag(metadata, ("EXIF:Make", "QuickTime:Make", "MakerNotes:Make")) or ""
|
|
if not make:
|
|
make = first_tag(metadata, ("H264:Make",)) or ""
|
|
model = first_tag(metadata, ("EXIF:Model", "QuickTime:Model", "MakerNotes:Model")) or ""
|
|
if not model:
|
|
model = first_tag(metadata, ("H264:Model",)) or ""
|
|
serial = first_tag(
|
|
metadata,
|
|
(
|
|
"EXIF:SerialNumber",
|
|
"MakerNotes:SerialNumber",
|
|
"MakerNotes:InternalSerialNumber",
|
|
"Composite:SerialNumber",
|
|
),
|
|
)
|
|
if make or model or serial:
|
|
return "|".join(str(part).strip() for part in (make, model, serial) if part)
|
|
return f"dir:{record.path.parent.resolve()}"
|
|
|
|
|
|
def parse_filename_timestamp(record: MediaRecord) -> datetime | None:
|
|
try:
|
|
return parse_datetime(record.path.stem)
|
|
except ValueError:
|
|
return None
|
|
|
|
|
|
def resolve_reference_time(records: list[MediaRecord], value: str) -> datetime:
|
|
text = value.strip()
|
|
lower = text.lower()
|
|
for record in records:
|
|
candidates = {
|
|
str(record.path).lower(),
|
|
record.path.name.lower(),
|
|
record.path.stem.lower(),
|
|
}
|
|
if lower in candidates:
|
|
if record.adjusted_time is None:
|
|
raise ValueError(f"{value} has no readable timestamp")
|
|
return naive_wall_time(record.adjusted_time)
|
|
return naive_wall_time(parse_datetime(text))
|
|
|
|
|
|
def infer_reference_target(source_time: datetime, target_text: str) -> datetime:
|
|
try:
|
|
return naive_wall_time(parse_datetime(target_text))
|
|
except ValueError:
|
|
pass
|
|
|
|
base = naive_wall_time(source_time)
|
|
target_same_date = parse_datetime(target_text, default_date=base)
|
|
candidates = [
|
|
target_same_date - timedelta(days=1),
|
|
target_same_date,
|
|
target_same_date + timedelta(days=1),
|
|
]
|
|
|
|
def score(candidate: datetime) -> tuple[int, float]:
|
|
delta_hours = abs((candidate - base).total_seconds()) / 3600
|
|
same_date_penalty = 0 if candidate.date() == base.date() else 1
|
|
return (0 if delta_hours <= 12 else 1, delta_hours + same_date_penalty)
|
|
|
|
return min(candidates, key=score)
|
|
|
|
|
|
def prompt_time_offset(records: list[MediaRecord]) -> timedelta | None:
|
|
while True:
|
|
mode = prompt_input("Time correction: Enter=none, o=offset, r=reference photo/clock: ").strip().lower()
|
|
if mode == "":
|
|
return None
|
|
if mode in {"o", "offset"}:
|
|
while True:
|
|
text = prompt_input("Enter time shift (+1:30, -02:00:00, +1h 2m): ").strip()
|
|
try:
|
|
return parse_timeshift(text)
|
|
except ValueError as exc:
|
|
print(exc)
|
|
if mode in {"r", "ref", "reference"}:
|
|
while True:
|
|
source_text = prompt_input("Reference source timestamp, filename, or path: ").strip()
|
|
target_text = prompt_input("Correct reference-clock time: ").strip()
|
|
try:
|
|
source_time = resolve_reference_time(records, source_text)
|
|
target_time = infer_reference_target(source_time, target_text)
|
|
offset = target_time - source_time
|
|
print(f"Computed time shift: {format_timedelta(offset)}")
|
|
return offset
|
|
except ValueError as exc:
|
|
print(exc)
|
|
print("Please choose Enter, o, or r.")
|
|
|
|
|
|
def format_timedelta(value: timedelta) -> str:
|
|
total_seconds = int(value.total_seconds())
|
|
sign = "+" if total_seconds >= 0 else "-"
|
|
total_seconds = abs(total_seconds)
|
|
hours = total_seconds // 3600
|
|
minutes = (total_seconds % 3600) // 60
|
|
seconds = total_seconds % 60
|
|
return f"{sign}{hours:02d}:{minutes:02d}:{seconds:02d}"
|
|
|
|
|
|
def prompt_timezone(records: list[MediaRecord], rename_files: bool) -> tuple[timezone | None, timezone | None, bool, timezone | None]:
|
|
while True:
|
|
text = prompt_input("Enter timezone offset (+09:00), or blank to keep/infer: ").strip()
|
|
if not text:
|
|
explicit = None
|
|
break
|
|
try:
|
|
explicit = parse_timezone_offset(text)
|
|
break
|
|
except ValueError as exc:
|
|
print(exc)
|
|
|
|
photo_zones = {
|
|
timezone_to_string(record.timezone_value): record.timezone_value
|
|
for record in records
|
|
if record.is_image and record.timezone_value is not None
|
|
}
|
|
missing_photo_zones = [
|
|
record for record in records if record.is_image and record.timezone_value is None
|
|
]
|
|
|
|
authoritative = explicit
|
|
if authoritative is None and len(photo_zones) == 1:
|
|
authoritative = next(iter(photo_zones.values()))
|
|
elif authoritative is None and len(photo_zones) > 1:
|
|
print("Photo timezone offsets conflict; no authoritative timezone inferred.")
|
|
|
|
write_missing = False
|
|
if authoritative is not None and explicit is not None:
|
|
write_missing = True
|
|
elif authoritative is not None and missing_photo_zones and photo_zones:
|
|
write_missing = ask_yes_no(
|
|
f"Write inferred timezone {timezone_to_string(authoritative)} to "
|
|
f"{len(missing_photo_zones)} photo(s) missing it?",
|
|
default=True,
|
|
)
|
|
|
|
video_tz = None
|
|
has_videos = any(record.is_video for record in records)
|
|
if rename_files and has_videos:
|
|
if authoritative is not None and explicit is None:
|
|
if ask_yes_no(
|
|
f"Use photo timezone {timezone_to_string(authoritative)} for video filenames?",
|
|
default=True,
|
|
):
|
|
video_tz = authoritative
|
|
elif authoritative is not None:
|
|
video_tz = authoritative
|
|
|
|
while video_tz is None:
|
|
text = prompt_input("Video renaming needs local timezone (+09:00): ").strip()
|
|
try:
|
|
video_tz = parse_timezone_offset(text)
|
|
except ValueError as exc:
|
|
print(exc)
|
|
|
|
return explicit, authoritative, write_missing, video_tz
|
|
|
|
|
|
def prompt_artist_action() -> tuple[str, str | None]:
|
|
while True:
|
|
text = prompt_input("Artist/author: Enter=leave unchanged, s=set, c=clear: ").strip().lower()
|
|
if text == "":
|
|
return "leave", None
|
|
if text in {"c", "clear"}:
|
|
return "clear", None
|
|
if text in {"s", "set"}:
|
|
value = prompt_input("Artist/author value: ").strip()
|
|
if value:
|
|
return "set", value
|
|
print("Use clear if you want an empty artist/author.")
|
|
else:
|
|
print("Please choose Enter, s, or c.")
|
|
|
|
|
|
def prompt_choices(records: list[MediaRecord], working_dir: Path, has_pto_files: bool) -> UserChoices:
|
|
missing_time = [record for record in records if record.original_time is None]
|
|
apply_filename_timestamps = False
|
|
if missing_time:
|
|
apply_filename_timestamps = ask_yes_no(
|
|
f"{len(missing_time)} file(s) have no readable metadata timestamp. "
|
|
"Try timestamp from filename?",
|
|
default=True,
|
|
)
|
|
if apply_filename_timestamps:
|
|
for record in missing_time:
|
|
parsed = parse_filename_timestamp(record)
|
|
if parsed is not None:
|
|
record.original_time = parsed
|
|
record.adjusted_time = parsed
|
|
record.parsed_from_filename = True
|
|
|
|
time_offset = prompt_time_offset(records)
|
|
rename_files = ask_yes_no("Rename files to timestamps?", default=True)
|
|
explicit_tz, authoritative_tz, write_missing_tz, video_tz = prompt_timezone(records, rename_files)
|
|
artist_action, artist_value = prompt_artist_action()
|
|
group_min_size = 0
|
|
if sum(1 for record in records if record.is_image and record.adjusted_time) >= 2:
|
|
group_min_size = ask_group_min_size()
|
|
group_photos = group_min_size >= 2
|
|
|
|
process_pto_files = False
|
|
if has_pto_files:
|
|
process_pto_files = ask_yes_no(
|
|
"Use panorama project files (.pto) for grouping and update their references?",
|
|
default=True,
|
|
)
|
|
|
|
move_to_working_dir = False
|
|
has_subdir_media = any(record.path.parent.resolve() != working_dir for record in records)
|
|
if has_subdir_media:
|
|
if group_photos or process_pto_files:
|
|
move_to_working_dir = True
|
|
else:
|
|
move_to_working_dir = ask_yes_no(
|
|
f"Move files from subdirectories into the working directory ({working_dir})?",
|
|
default=False,
|
|
)
|
|
|
|
return UserChoices(
|
|
working_dir=working_dir,
|
|
time_offset=time_offset,
|
|
explicit_timezone=explicit_tz,
|
|
authoritative_timezone=authoritative_tz,
|
|
write_missing_photo_timezones=write_missing_tz,
|
|
video_filename_timezone=video_tz,
|
|
artist_action=artist_action,
|
|
artist_value=artist_value,
|
|
rename_files=rename_files,
|
|
group_photos=group_photos,
|
|
group_min_size=group_min_size,
|
|
move_to_working_dir=move_to_working_dir,
|
|
process_pto_files=process_pto_files,
|
|
apply_filename_timestamps=apply_filename_timestamps,
|
|
)
|
|
|
|
|
|
def apply_time_offset(records: list[MediaRecord], offset: timedelta | None) -> None:
|
|
if offset is None:
|
|
return
|
|
for record in records:
|
|
if record.adjusted_time is not None:
|
|
record.adjusted_time = record.adjusted_time + offset
|
|
|
|
|
|
def photo_sort_key(record: MediaRecord) -> tuple:
|
|
assert record.adjusted_time is not None
|
|
return (
|
|
naive_wall_time(record.adjusted_time),
|
|
record.subsec if record.subsec is not None else -1,
|
|
record.sequence if record.sequence is not None else -1,
|
|
record.path.name.lower(),
|
|
)
|
|
|
|
|
|
def group_sort_key(record: MediaRecord) -> tuple:
|
|
timestamp = naive_wall_time(record.adjusted_time) if record.adjusted_time is not None else datetime.max
|
|
return (
|
|
timestamp,
|
|
record.subsec if record.subsec is not None else -1,
|
|
record.sequence if record.sequence is not None else -1,
|
|
str(record.path).lower(),
|
|
)
|
|
|
|
|
|
def detect_photo_groups(records: list[MediaRecord], min_size: int = 2) -> list[list[MediaRecord]]:
|
|
groups: list[list[MediaRecord]] = []
|
|
by_camera: dict[str, list[MediaRecord]] = defaultdict(list)
|
|
for record in records:
|
|
if record.is_image and record.adjusted_time is not None:
|
|
by_camera[record.camera_key].append(record)
|
|
|
|
for camera_records in by_camera.values():
|
|
groups.extend(detect_groups_for_camera(sorted(camera_records, key=photo_sort_key), min_size))
|
|
|
|
return groups
|
|
|
|
|
|
def detect_groups_for_camera(records: list[MediaRecord], min_size: int) -> list[list[MediaRecord]]:
|
|
by_second: dict[datetime, list[MediaRecord]] = defaultdict(list)
|
|
for record in records:
|
|
assert record.adjusted_time is not None
|
|
by_second[naive_wall_time(record.adjusted_time).replace(microsecond=0)].append(record)
|
|
|
|
seconds = sorted(by_second)
|
|
spans: list[list[datetime]] = []
|
|
current: list[datetime] = []
|
|
for second in seconds:
|
|
if not current or second == current[-1] + timedelta(seconds=1):
|
|
current.append(second)
|
|
else:
|
|
spans.append(current)
|
|
current = [second]
|
|
if current:
|
|
spans.append(current)
|
|
|
|
groups: list[list[MediaRecord]] = []
|
|
for span in spans:
|
|
if max(len(by_second[second]) for second in span) < 2:
|
|
continue
|
|
span_records = [record for second in span for record in by_second[second]]
|
|
if all(record.sequence is not None for record in span_records) and len(span_records) >= 2:
|
|
groups.extend(sequence_groups(span_records, min_size))
|
|
else:
|
|
groups.extend(timestamp_groups(span, by_second, min_size))
|
|
return [group for group in groups if len(group) >= min_size]
|
|
|
|
|
|
def sequence_groups(records: list[MediaRecord], min_size: int) -> list[list[MediaRecord]]:
|
|
ordered = sorted(records, key=photo_sort_key)
|
|
groups: list[list[MediaRecord]] = []
|
|
current: list[MediaRecord] = [ordered[0]]
|
|
for record in ordered[1:]:
|
|
previous = current[-1]
|
|
if record.sequence is not None and previous.sequence is not None and record.sequence > previous.sequence:
|
|
current.append(record)
|
|
else:
|
|
if len(current) >= min_size:
|
|
groups.append(current)
|
|
current = [record]
|
|
if len(current) >= min_size:
|
|
groups.append(current)
|
|
return groups
|
|
|
|
|
|
def timestamp_groups(
|
|
span: list[datetime], by_second: dict[datetime, list[MediaRecord]], min_size: int
|
|
) -> list[list[MediaRecord]]:
|
|
multi_indices = [index for index, second in enumerate(span) if len(by_second[second]) >= 2]
|
|
if not multi_indices:
|
|
return []
|
|
|
|
clusters: list[list[int]] = [[multi_indices[0]]]
|
|
for index in multi_indices[1:]:
|
|
if index - clusters[-1][-1] <= 2:
|
|
clusters[-1].append(index)
|
|
else:
|
|
clusters.append([index])
|
|
|
|
groups: list[list[MediaRecord]] = []
|
|
used_seconds: set[datetime] = set()
|
|
for cluster in clusters:
|
|
start = cluster[0]
|
|
end = cluster[-1]
|
|
if start > 0 and len(by_second[span[start - 1]]) == 1:
|
|
start -= 1
|
|
if end + 1 < len(span) and len(by_second[span[end + 1]]) == 1:
|
|
end += 1
|
|
seconds = [second for second in span[start : end + 1] if second not in used_seconds]
|
|
for second in seconds:
|
|
used_seconds.add(second)
|
|
records = [record for second in seconds for record in sorted(by_second[second], key=photo_sort_key)]
|
|
if len(records) >= min_size:
|
|
groups.append(records)
|
|
return groups
|
|
|
|
|
|
def infer_group_subseconds(groups: list[list[MediaRecord]]) -> None:
|
|
for group in groups:
|
|
by_second: dict[datetime, list[MediaRecord]] = defaultdict(list)
|
|
for record in group:
|
|
if not record.is_image or record.adjusted_time is None:
|
|
continue
|
|
second = naive_wall_time(record.adjusted_time).replace(microsecond=0)
|
|
by_second[second].append(record)
|
|
|
|
seconds = sorted(by_second)
|
|
counts = {second: len(records) for second, records in by_second.items()}
|
|
for index, second in enumerate(seconds):
|
|
records = sorted(by_second[second], key=photo_sort_key)
|
|
count = len(records)
|
|
if count <= 1:
|
|
continue
|
|
fps = count
|
|
if len(seconds) == 2:
|
|
fps = max(counts[seconds[0]], counts[seconds[1]])
|
|
elif index == 0 and len(seconds) > 1:
|
|
fps = max(count, counts[seconds[index + 1]])
|
|
elif index == len(seconds) - 1 and len(seconds) > 1:
|
|
fps = max(count, counts[seconds[index - 1]])
|
|
|
|
start_slot = 0
|
|
if index == 0 and fps > count:
|
|
start_slot = fps - count
|
|
|
|
for item_index, record in enumerate(records):
|
|
record.inferred_subsec = int((start_slot + item_index) * 1000 / fps)
|
|
|
|
|
|
def filename_range_stem(original_stem: str, base_dt: datetime) -> str | None:
|
|
match = TIMESTAMP_RANGE_STEM_RE.match(original_stem)
|
|
if not match:
|
|
return None
|
|
try:
|
|
old_start = parse_datetime(match.group("start"))
|
|
old_end = parse_datetime(match.group("end"))
|
|
except ValueError:
|
|
return None
|
|
interval = old_end - old_start
|
|
new_start = naive_wall_time(base_dt)
|
|
new_end = new_start + interval
|
|
suffix = match.group("suffix") or ""
|
|
return f"{format_filename_stem(new_start)}-{format_filename_stem(new_end)}{suffix}"
|
|
|
|
|
|
def unique_records_by_stem(records: list[MediaRecord]) -> dict[str, MediaRecord]:
|
|
buckets: dict[str, list[MediaRecord]] = defaultdict(list)
|
|
for record in records:
|
|
buckets[record.path.stem.lower()].append(record)
|
|
return {stem: bucket[0] for stem, bucket in buckets.items() if len(bucket) == 1}
|
|
|
|
|
|
def camera_range_stem(record: MediaRecord, records_by_stem: dict[str, MediaRecord]) -> str | None:
|
|
match = CAMERA_RANGE_STEM_RE.match(record.path.stem)
|
|
if not match:
|
|
return None
|
|
|
|
first_number = int(match.group("first"))
|
|
last_number = int(match.group("last"))
|
|
if last_number <= first_number:
|
|
return None
|
|
|
|
prefix = match.group("prefix")
|
|
first_record = records_by_stem.get(f"{prefix}{match.group('first')}".lower())
|
|
last_record = records_by_stem.get(f"{prefix}{match.group('last')}".lower())
|
|
first_time = first_record.adjusted_time if first_record else None
|
|
last_time = last_record.adjusted_time if last_record else None
|
|
own_time = record.adjusted_time
|
|
|
|
if own_time is not None:
|
|
start_time = own_time
|
|
end_time = last_time or own_time
|
|
else:
|
|
start_time = first_time or last_time
|
|
end_time = last_time or first_time
|
|
|
|
if start_time is None or end_time is None:
|
|
return None
|
|
|
|
suffix = match.group("suffix")
|
|
return f"{format_filename_stem(start_time)}-{format_filename_stem(end_time)}{suffix}"
|
|
|
|
|
|
def filename_stem_for_dt(record: MediaRecord, dt: datetime) -> str:
|
|
range_stem = filename_range_stem(record.path.stem, dt)
|
|
if range_stem is not None:
|
|
return range_stem
|
|
return format_filename_stem(dt)
|
|
|
|
|
|
def plan_names(records: list[MediaRecord], choices: UserChoices) -> None:
|
|
if not choices.rename_files:
|
|
for record in records:
|
|
record.final_name = record.path.name
|
|
return
|
|
|
|
records_by_stem = unique_records_by_stem(records)
|
|
for record in records:
|
|
if record.adjusted_time is None:
|
|
range_stem = camera_range_stem(record, records_by_stem)
|
|
if range_stem is not None:
|
|
record.new_stem = range_stem
|
|
else:
|
|
record.rename_skip_reason = "missing timestamp"
|
|
continue
|
|
if record.is_video:
|
|
if record.duration_seconds is None and not record.parsed_from_filename:
|
|
record.rename_skip_reason = "missing duration"
|
|
continue
|
|
if choices.video_filename_timezone is None:
|
|
record.rename_skip_reason = "missing video timezone"
|
|
continue
|
|
if record.parsed_from_filename and record.original_time is not None:
|
|
begin = record.adjusted_time
|
|
elif record.video_time_is_beginning:
|
|
begin = record.adjusted_time
|
|
else:
|
|
duration = round_half_up(record.duration_seconds or 0)
|
|
begin = record.adjusted_time - timedelta(seconds=duration)
|
|
if begin.tzinfo is None:
|
|
begin = begin.replace(tzinfo=timezone.utc)
|
|
local_begin = begin.astimezone(choices.video_filename_timezone)
|
|
record.new_stem = filename_stem_for_dt(record, local_begin)
|
|
else:
|
|
record.new_stem = camera_range_stem(record, records_by_stem) or filename_stem_for_dt(
|
|
record, record.adjusted_time
|
|
)
|
|
|
|
assign_unique_names(records)
|
|
|
|
|
|
def assign_unique_names(records: list[MediaRecord]) -> None:
|
|
buckets: dict[tuple[int | str, str], list[MediaRecord]] = defaultdict(list)
|
|
for record in records:
|
|
if record.new_stem is None:
|
|
record.final_name = record.path.name
|
|
continue
|
|
container: int | str = record.group_id if record.group_id is not None else "all"
|
|
buckets[(container, normalized_extension(record.path))].append(record)
|
|
|
|
for (_container, ext), bucket in buckets.items():
|
|
by_stem: dict[str, list[MediaRecord]] = defaultdict(list)
|
|
for record in sorted(bucket, key=lambda item: (item.new_stem or "", item.path.name.lower())):
|
|
assert record.new_stem is not None
|
|
by_stem[record.new_stem].append(record)
|
|
|
|
for stem, records_with_stem in by_stem.items():
|
|
if len(records_with_stem) == 1:
|
|
records_with_stem[0].final_name = f"{stem}{ext}"
|
|
continue
|
|
for index, record in enumerate(records_with_stem, start=1):
|
|
record.final_name = f"{stem}-{index}{ext}"
|
|
|
|
|
|
def merge_record_groups(
|
|
records: list[MediaRecord], group_candidates: list[list[MediaRecord]]
|
|
) -> list[list[MediaRecord]]:
|
|
record_indexes = {record.path.resolve(): index for index, record in enumerate(records)}
|
|
parents = list(range(len(records)))
|
|
grouped_indexes: set[int] = set()
|
|
|
|
def find(index: int) -> int:
|
|
while parents[index] != index:
|
|
parents[index] = parents[parents[index]]
|
|
index = parents[index]
|
|
return index
|
|
|
|
def union(left: int, right: int) -> None:
|
|
left_root = find(left)
|
|
right_root = find(right)
|
|
if left_root != right_root:
|
|
parents[right_root] = left_root
|
|
|
|
for candidate in group_candidates:
|
|
indexes = [
|
|
record_indexes[record.path.resolve()]
|
|
for record in candidate
|
|
if record.path.resolve() in record_indexes
|
|
]
|
|
if not indexes:
|
|
continue
|
|
grouped_indexes.update(indexes)
|
|
first = indexes[0]
|
|
for index in indexes[1:]:
|
|
union(first, index)
|
|
|
|
merged: dict[int, list[MediaRecord]] = defaultdict(list)
|
|
for index in grouped_indexes:
|
|
merged[find(index)].append(records[index])
|
|
|
|
return sorted(
|
|
[sorted(group, key=group_sort_key) for group in merged.values()],
|
|
key=lambda group: group_sort_key(group[0]),
|
|
)
|
|
|
|
|
|
def assign_groups(records: list[MediaRecord], groups: list[list[MediaRecord]]) -> None:
|
|
for record in records:
|
|
record.group_id = None
|
|
record.group_name = None
|
|
for index, group in enumerate(groups, start=1):
|
|
ordered = sorted(group, key=group_sort_key)
|
|
for record in ordered:
|
|
record.group_id = index
|
|
|
|
|
|
def unique_group_name(base_name: str, working_dir: Path, used_names: set[str]) -> str:
|
|
if (working_dir / base_name).exists():
|
|
used_names.add(base_name)
|
|
return base_name
|
|
|
|
if base_name not in used_names:
|
|
used_names.add(base_name)
|
|
return base_name
|
|
|
|
for index in range(1, 10000):
|
|
candidate = f"{base_name}-{index}"
|
|
if candidate not in used_names and not (working_dir / candidate).exists():
|
|
used_names.add(candidate)
|
|
return candidate
|
|
|
|
raise RuntimeError(f"Could not find an available group directory name for {base_name}")
|
|
|
|
|
|
def assign_group_names(groups: list[list[MediaRecord]], working_dir: Path) -> None:
|
|
used_names: set[str] = set()
|
|
for group in groups:
|
|
ordered = sorted(group, key=group_sort_key)
|
|
first = group_stem_from_name(ordered[0].final_name or ordered[0].path.name)
|
|
last = group_stem_from_name(ordered[-1].final_name or ordered[-1].path.name)
|
|
group_name = unique_group_name(f"{first}-{last}", working_dir, used_names)
|
|
for record in ordered:
|
|
record.group_name = group_name
|
|
|
|
|
|
def plan_targets(records: list[MediaRecord], choices: UserChoices) -> None:
|
|
for record in records:
|
|
if record.final_name is None:
|
|
record.final_name = record.path.name
|
|
out_dir = record.path.parent
|
|
if record.group_name:
|
|
out_dir = choices.working_dir / record.group_name
|
|
elif choices.move_to_working_dir:
|
|
out_dir = choices.working_dir
|
|
record.target_path = out_dir / record.final_name
|
|
if record.is_video and record.sidecar is not None:
|
|
target_sidecar = out_dir / f"{Path(record.final_name).stem}M01.XML"
|
|
if target_sidecar.resolve() != record.sidecar.path.resolve():
|
|
record.target_sidecar = target_sidecar
|
|
|
|
|
|
def normalize_reference(value: str) -> str:
|
|
return value.replace("\\", "/")
|
|
|
|
|
|
def pto_target_for_base(
|
|
path: Path, target_dir: Path, base_name: str, used_targets: set[Path]
|
|
) -> Path:
|
|
desired = target_dir / f"{base_name}.pto"
|
|
if desired.resolve() == path.resolve() or (not desired.exists() and desired not in used_targets):
|
|
used_targets.add(desired)
|
|
return desired
|
|
|
|
for index in range(1, 10000):
|
|
candidate = target_dir / f"{base_name}-{index}.pto"
|
|
if candidate.resolve() == path.resolve() or (
|
|
not candidate.exists() and candidate not in used_targets
|
|
):
|
|
used_targets.add(candidate)
|
|
return candidate
|
|
|
|
raise RuntimeError(f"Could not find an available .pto name for {base_name}")
|
|
|
|
|
|
def pto_reference_map(pto_path: Path, records: list[MediaRecord]) -> dict[str, MediaRecord]:
|
|
pto_dir = pto_path.parent
|
|
basename_counts: dict[str, int] = defaultdict(int)
|
|
for record in records:
|
|
basename_counts[record.path.name] += 1
|
|
|
|
mapping: dict[str, MediaRecord] = {}
|
|
for record in records:
|
|
old_abs = record.path.resolve()
|
|
candidates = {
|
|
normalize_reference(str(old_abs)),
|
|
normalize_reference(os.path.relpath(old_abs, pto_dir)),
|
|
}
|
|
if basename_counts[record.path.name] == 1:
|
|
candidates.add(record.path.name)
|
|
for candidate in candidates:
|
|
mapping[candidate] = record
|
|
return mapping
|
|
|
|
|
|
PTO_QUOTED_VALUE_RE = re.compile(r"([\"'])(.*?)(\1)")
|
|
|
|
|
|
def read_pto_text(path: Path) -> str:
|
|
try:
|
|
return path.read_text(encoding="utf-8")
|
|
except UnicodeDecodeError:
|
|
return path.read_text(encoding="utf-8", errors="replace")
|
|
|
|
|
|
def read_pto_references(pto_files: list[Path], records: list[MediaRecord]) -> list[PtoReference]:
|
|
references: list[PtoReference] = []
|
|
for pto_path in pto_files:
|
|
original_text = read_pto_text(pto_path)
|
|
mapping = pto_reference_map(pto_path, records)
|
|
if not mapping:
|
|
continue
|
|
|
|
referenced: list[MediaRecord] = []
|
|
for match in PTO_QUOTED_VALUE_RE.finditer(original_text):
|
|
record = mapping.get(normalize_reference(match.group(2)))
|
|
if record is not None and record not in referenced:
|
|
referenced.append(record)
|
|
|
|
if referenced:
|
|
references.append(PtoReference(path=pto_path, original_text=original_text, records=referenced))
|
|
return references
|
|
|
|
|
|
def replacement_for_pto_value(
|
|
value: str, mapping: dict[str, MediaRecord], pto_target_dir: Path
|
|
) -> tuple[str, MediaRecord] | None:
|
|
normalized = normalize_reference(value)
|
|
record = mapping.get(normalized)
|
|
if record is None:
|
|
return None
|
|
assert record.target_path is not None
|
|
replacement = normalize_reference(os.path.relpath(record.target_path.resolve(), pto_target_dir))
|
|
return replacement, record
|
|
|
|
|
|
def plan_pto_updates(pto_references: list[PtoReference], records: list[MediaRecord]) -> list[PtoPlan]:
|
|
plans: list[PtoPlan] = []
|
|
used_targets: set[Path] = set()
|
|
for pto_reference in pto_references:
|
|
pto_path = pto_reference.path
|
|
original_text = pto_reference.original_text
|
|
referenced = sorted(pto_reference.records, key=group_sort_key)
|
|
if not referenced:
|
|
continue
|
|
|
|
target_dir = pto_path.parent
|
|
if referenced[0].group_name:
|
|
target_dir = referenced[0].target_path.parent if referenced[0].target_path else target_dir
|
|
|
|
mapping = pto_reference_map(pto_path, records)
|
|
if not mapping:
|
|
continue
|
|
|
|
replacements: list[tuple[str, str]] = []
|
|
|
|
def replace_match(match: re.Match) -> str:
|
|
quote = match.group(1)
|
|
value = match.group(2)
|
|
replacement = replacement_for_pto_value(value, mapping, target_dir)
|
|
if replacement is None:
|
|
return match.group(0)
|
|
new_value, record = replacement
|
|
if value != new_value:
|
|
replacements.append((value, new_value))
|
|
return f"{quote}{new_value}{quote}"
|
|
|
|
updated_text = PTO_QUOTED_VALUE_RE.sub(replace_match, original_text)
|
|
first = group_stem_from_name(referenced[0].target_path.name)
|
|
last = group_stem_from_name(referenced[-1].target_path.name)
|
|
target_path = pto_target_for_base(pto_path, target_dir, f"{first}-{last}", used_targets)
|
|
if updated_text == original_text and target_path.resolve() == pto_path.resolve():
|
|
continue
|
|
plans.append(
|
|
PtoPlan(
|
|
path=pto_path,
|
|
target_path=target_path,
|
|
referenced_records=referenced,
|
|
updated_text=updated_text,
|
|
replacements=replacements,
|
|
)
|
|
)
|
|
|
|
return plans
|
|
|
|
|
|
def build_write_plan(record: MediaRecord, choices: UserChoices) -> list[TagWrite]:
|
|
writes: list[TagWrite] = []
|
|
if choices.artist_action == "set":
|
|
assert choices.artist_value is not None
|
|
add_tag_write(
|
|
writes,
|
|
"Artist",
|
|
"Artist",
|
|
tag_value(record.metadata, "EXIF:Artist") or tag_value(record.metadata, "XMP:Artist"),
|
|
choices.artist_value,
|
|
)
|
|
add_tag_write(
|
|
writes,
|
|
"Author",
|
|
"Author",
|
|
tag_value(record.metadata, "EXIF:Author") or tag_value(record.metadata, "XMP:Author"),
|
|
choices.artist_value,
|
|
)
|
|
elif choices.artist_action == "clear":
|
|
add_tag_write(
|
|
writes,
|
|
"Artist",
|
|
"Artist",
|
|
tag_value(record.metadata, "EXIF:Artist") or tag_value(record.metadata, "XMP:Artist"),
|
|
"",
|
|
)
|
|
add_tag_write(
|
|
writes,
|
|
"Author",
|
|
"Author",
|
|
tag_value(record.metadata, "EXIF:Author") or tag_value(record.metadata, "XMP:Author"),
|
|
"",
|
|
)
|
|
|
|
if record.is_image:
|
|
if choices.time_offset is not None or record.parsed_from_filename:
|
|
if record.adjusted_time is not None:
|
|
dt_text = record.adjusted_time.strftime("%Y:%m:%d %H:%M:%S")
|
|
for label, arg_name, current_tag in (
|
|
("DateTimeOriginal", "EXIF:DateTimeOriginal", "EXIF:DateTimeOriginal"),
|
|
("CreateDate", "EXIF:CreateDate", "EXIF:CreateDate"),
|
|
("ModifyDate", "EXIF:ModifyDate", "EXIF:ModifyDate"),
|
|
):
|
|
add_tag_write(
|
|
writes,
|
|
label,
|
|
arg_name,
|
|
tag_value(record.metadata, current_tag),
|
|
dt_text,
|
|
same_datetime_value,
|
|
)
|
|
|
|
should_write_tz = False
|
|
tz_to_write = choices.explicit_timezone
|
|
if tz_to_write is not None:
|
|
should_write_tz = True
|
|
elif (
|
|
choices.write_missing_photo_timezones
|
|
and record.timezone_value is None
|
|
and choices.authoritative_timezone is not None
|
|
):
|
|
tz_to_write = choices.authoritative_timezone
|
|
should_write_tz = True
|
|
if should_write_tz and tz_to_write is not None:
|
|
tz_text = timezone_to_string(tz_to_write)
|
|
for label, arg_name, current_tag in (
|
|
("OffsetTimeOriginal", "EXIF:OffsetTimeOriginal", "EXIF:OffsetTimeOriginal"),
|
|
("OffsetTimeDigitized", "EXIF:OffsetTimeDigitized", "EXIF:OffsetTimeDigitized"),
|
|
("OffsetTime", "EXIF:OffsetTime", "EXIF:OffsetTime"),
|
|
):
|
|
add_tag_write(
|
|
writes,
|
|
label,
|
|
arg_name,
|
|
tag_value(record.metadata, current_tag),
|
|
tz_text,
|
|
)
|
|
|
|
if record.inferred_subsec is not None:
|
|
subsec_text = f"{record.inferred_subsec:03d}"
|
|
for label, arg_name, current_tag in (
|
|
("SubSecTimeOriginal", "EXIF:SubSecTimeOriginal", "EXIF:SubSecTimeOriginal"),
|
|
("SubSecTimeDigitized", "EXIF:SubSecTimeDigitized", "EXIF:SubSecTimeDigitized"),
|
|
("SubSecTime", "EXIF:SubSecTime", "EXIF:SubSecTime"),
|
|
):
|
|
add_tag_write(
|
|
writes,
|
|
label,
|
|
arg_name,
|
|
tag_value(record.metadata, current_tag),
|
|
subsec_text,
|
|
same_subsec_value,
|
|
)
|
|
elif record.is_video and choices.time_offset is not None and record.adjusted_time is not None:
|
|
utc_time = record.adjusted_time
|
|
if utc_time.tzinfo is None:
|
|
utc_time = utc_time.replace(tzinfo=timezone.utc)
|
|
utc_text = utc_time.astimezone(timezone.utc).strftime("%Y:%m:%d %H:%M:%S")
|
|
for label, arg_name, current_tag in (
|
|
("QuickTime CreateDate", "QuickTime:CreateDate", "QuickTime:CreateDate"),
|
|
("QuickTime ModifyDate", "QuickTime:ModifyDate", "QuickTime:ModifyDate"),
|
|
("QuickTime TrackCreateDate", "QuickTime:TrackCreateDate", "QuickTime:TrackCreateDate"),
|
|
("QuickTime TrackModifyDate", "QuickTime:TrackModifyDate", "QuickTime:TrackModifyDate"),
|
|
("QuickTime MediaCreateDate", "QuickTime:MediaCreateDate", "QuickTime:MediaCreateDate"),
|
|
("QuickTime MediaModifyDate", "QuickTime:MediaModifyDate", "QuickTime:MediaModifyDate"),
|
|
):
|
|
add_tag_write(
|
|
writes,
|
|
label,
|
|
arg_name,
|
|
tag_value(record.metadata, current_tag),
|
|
utc_text,
|
|
same_datetime_value,
|
|
)
|
|
|
|
return writes
|
|
|
|
|
|
def build_write_args(record: MediaRecord, choices: UserChoices) -> list[str]:
|
|
if not record.write_plan:
|
|
record.write_plan = build_write_plan(record, choices)
|
|
return [write.arg for write in record.write_plan if write.will_write]
|
|
|
|
|
|
def path_display(path: Path, choices: UserChoices, use_relative: bool) -> str:
|
|
if use_relative:
|
|
return display_path_from_working_dir(path, choices.working_dir)
|
|
return str(path)
|
|
|
|
|
|
def directory_display(path: Path, choices: UserChoices) -> str:
|
|
text = path_display(path, choices, use_relative=True)
|
|
if text != "." and not text.endswith(os.sep):
|
|
return f"{text}{os.sep}"
|
|
return text
|
|
|
|
|
|
def move_rename_preview(
|
|
source: Path,
|
|
target: Path,
|
|
choices: UserChoices,
|
|
source_label: str | None = None,
|
|
) -> str | None:
|
|
is_moving = source.parent.resolve() != target.parent.resolve()
|
|
is_renaming = source.name != target.name
|
|
source_text = source_label or path_display(source, choices, use_relative=is_moving)
|
|
|
|
if is_moving and is_renaming:
|
|
target_text = path_display(target, choices, use_relative=True)
|
|
return f"move+rename {source_text} -> {target_text}"
|
|
if is_moving:
|
|
target_text = directory_display(target.parent, choices)
|
|
return f"move {source_text} -> {target_text}"
|
|
if is_renaming:
|
|
return f"rename {source_label or source.name} -> {target.name}"
|
|
return None
|
|
|
|
|
|
def print_preview(records: list[MediaRecord], choices: UserChoices, pto_plans: list[PtoPlan]) -> None:
|
|
print("\nPreview")
|
|
print(f"Working directory: {choices.working_dir}")
|
|
print(f"Move ungrouped files into working directory: {'yes' if choices.move_to_working_dir else 'no'}")
|
|
print(
|
|
"Automatic burst/HDR grouping: "
|
|
+ (f"minimum {choices.group_min_size}" if choices.group_photos else "no")
|
|
)
|
|
print(f"Panorama project files (.pto): {'group/update' if choices.process_pto_files else 'ignore'}")
|
|
print(f"Time shift: {format_timedelta(choices.time_offset) if choices.time_offset else 'none'}")
|
|
print(
|
|
"Photo timezone write: "
|
|
+ (
|
|
timezone_to_string(choices.explicit_timezone)
|
|
if choices.explicit_timezone
|
|
else (
|
|
f"missing only -> {timezone_to_string(choices.authoritative_timezone)}"
|
|
if choices.write_missing_photo_timezones and choices.authoritative_timezone
|
|
else "unchanged"
|
|
)
|
|
)
|
|
)
|
|
print(
|
|
"Video filename timezone: "
|
|
+ (timezone_to_string(choices.video_filename_timezone) if choices.video_filename_timezone else "n/a")
|
|
)
|
|
if choices.artist_action == "set":
|
|
print(f"Artist/author: set to {choices.artist_value}")
|
|
elif choices.artist_action == "clear":
|
|
print("Artist/author: clear")
|
|
else:
|
|
print("Artist/author: unchanged")
|
|
|
|
for record in sorted(records, key=lambda item: str(item.path).lower()):
|
|
target = record.target_path or record.path
|
|
operation = []
|
|
operation_text = move_rename_preview(record.path, target, choices)
|
|
if operation_text:
|
|
operation.append(operation_text)
|
|
if record.rename_skip_reason:
|
|
operation.append(f"rename skipped: {record.rename_skip_reason}")
|
|
for warning in record.warnings:
|
|
operation.append(f"warning: {warning}")
|
|
if record.target_sidecar and record.sidecar:
|
|
sidecar_source = path_display(record.sidecar.path, choices, use_relative=True)
|
|
sidecar_target = path_display(record.target_sidecar, choices, use_relative=True)
|
|
operation.append(f"sidecar move {sidecar_source} -> {sidecar_target}")
|
|
if operation:
|
|
print("; ".join(operation))
|
|
|
|
if pto_plans:
|
|
print("\nPTO updates:")
|
|
for plan in pto_plans:
|
|
source = path_display(plan.path, choices, use_relative=True)
|
|
operation = move_rename_preview(plan.path, plan.target_path, choices, source_label=source)
|
|
print(operation or source)
|
|
for old_ref, new_ref in plan.replacements:
|
|
print(f" {old_ref} -> {new_ref}")
|
|
|
|
|
|
def has_planned_media_change(record: MediaRecord, write_args: list[str]) -> bool:
|
|
if write_args:
|
|
return True
|
|
if record.target_path and record.target_path.resolve() != record.path.resolve():
|
|
return True
|
|
if record.sidecar and record.target_sidecar and record.sidecar.path.exists():
|
|
return True
|
|
return False
|
|
|
|
|
|
def apply_changes(
|
|
executable: str, records: list[MediaRecord], choices: UserChoices
|
|
) -> list[str]:
|
|
failures: list[str] = []
|
|
planned_sources = {record.path.resolve() for record in records}
|
|
|
|
for index, record in enumerate(records, start=1):
|
|
changed = False
|
|
try:
|
|
write_args = build_write_args(record, choices)
|
|
changed = has_planned_media_change(record, write_args)
|
|
run_exiftool_write(executable, record.path, write_args)
|
|
|
|
if record.target_path and record.target_path != record.path:
|
|
target = unique_existing_target(record.target_path, planned_sources)
|
|
target.parent.mkdir(parents=True, exist_ok=True)
|
|
shutil.move(str(record.path), str(target))
|
|
record.target_path = target
|
|
|
|
if record.sidecar and record.target_sidecar:
|
|
target_sidecar = unique_existing_target(record.target_sidecar, planned_sources)
|
|
target_sidecar.parent.mkdir(parents=True, exist_ok=True)
|
|
if record.sidecar.path.exists():
|
|
shutil.move(str(record.sidecar.path), str(target_sidecar))
|
|
except (OSError, subprocess.CalledProcessError, RuntimeError) as exc:
|
|
failures.append(f"{record.path}: {exc}")
|
|
action = "Processed" if changed else "Skipped"
|
|
print(f"{action} {index} of {len(records)}")
|
|
|
|
return failures
|
|
|
|
|
|
def apply_pto_changes(pto_plans: list[PtoPlan]) -> list[str]:
|
|
failures: list[str] = []
|
|
for plan in pto_plans:
|
|
try:
|
|
plan.path.write_text(plan.updated_text, encoding="utf-8")
|
|
if plan.target_path.resolve() != plan.path.resolve():
|
|
target = unique_existing_target(plan.target_path, {plan.path.resolve()})
|
|
target.parent.mkdir(parents=True, exist_ok=True)
|
|
shutil.move(str(plan.path), str(target))
|
|
except OSError as exc:
|
|
failures.append(f"{plan.path}: {exc}")
|
|
return failures
|
|
|
|
|
|
def print_removed_dirs(removed_dirs: list[Path]) -> None:
|
|
if not removed_dirs:
|
|
return
|
|
print("\nRemoved empty director" + ("y:" if len(removed_dirs) == 1 else "ies:"))
|
|
for directory in removed_dirs:
|
|
print(f" - {directory}")
|
|
|
|
|
|
def prepare_plan(
|
|
records: list[MediaRecord], choices: UserChoices, pto_references: list[PtoReference]
|
|
) -> list[list[MediaRecord]]:
|
|
apply_time_offset(records, choices.time_offset)
|
|
|
|
group_candidates: list[list[MediaRecord]] = []
|
|
if choices.group_photos:
|
|
group_candidates.extend(detect_photo_groups(records, choices.group_min_size))
|
|
if choices.process_pto_files:
|
|
group_candidates.extend(reference.records for reference in pto_references)
|
|
|
|
groups = merge_record_groups(records, group_candidates)
|
|
if groups:
|
|
assign_groups(records, groups)
|
|
infer_group_subseconds(groups)
|
|
|
|
plan_names(records, choices)
|
|
if groups:
|
|
assign_group_names(groups, choices.working_dir)
|
|
plan_targets(records, choices)
|
|
for record in records:
|
|
record.write_plan = build_write_plan(record, choices)
|
|
return groups
|
|
|
|
|
|
def run(argv: list[str]) -> int:
|
|
if len(argv) < 2:
|
|
print("Usage: python3 photo_metadata.py path/to/file/or/directory [...]")
|
|
return 2
|
|
|
|
try:
|
|
working_dir = determine_working_dir(argv[1:])
|
|
except RuntimeError as exc:
|
|
print(exc)
|
|
return 1
|
|
|
|
cleanup_roots = cleanup_roots_from_args(argv[1:])
|
|
files = collect_media_files(argv[1:])
|
|
pto_files = collect_pto_files(argv[1:])
|
|
if not files:
|
|
print("No supported media files found.")
|
|
return 1
|
|
|
|
try:
|
|
exiftool = require_exiftool()
|
|
except RuntimeError as exc:
|
|
print(exc)
|
|
return 1
|
|
|
|
print(f"Found {len(files)} supported media file(s). Reading metadata...")
|
|
try:
|
|
metadata = run_exiftool_json(exiftool, files)
|
|
except (subprocess.CalledProcessError, json.JSONDecodeError) as exc:
|
|
print(f"Failed to read metadata with exiftool: {exc}")
|
|
return 1
|
|
|
|
records = build_records(files, metadata)
|
|
try:
|
|
choices = prompt_choices(records, working_dir, has_pto_files=bool(pto_files))
|
|
except RuntimeError as exc:
|
|
print(exc)
|
|
return 1
|
|
|
|
pto_references = read_pto_references(pto_files, records) if choices.process_pto_files else []
|
|
prepare_plan(records, choices, pto_references)
|
|
pto_plans = plan_pto_updates(pto_references, records) if choices.process_pto_files else []
|
|
print_preview(records, choices, pto_plans)
|
|
|
|
try:
|
|
proceed = ask_yes_no("Proceed with these changes?", default=True)
|
|
except RuntimeError as exc:
|
|
print(exc)
|
|
return 1
|
|
|
|
if not proceed:
|
|
print("Nothing changed.")
|
|
return 0
|
|
|
|
failures = apply_changes(exiftool, records, choices)
|
|
failures.extend(apply_pto_changes(pto_plans))
|
|
removed_dirs = remove_empty_dirs(cleanup_roots, keep_dirs={working_dir.resolve()})
|
|
print_removed_dirs(removed_dirs)
|
|
|
|
if failures:
|
|
print("\nSome files failed:")
|
|
for failure in failures:
|
|
print(f" - {failure}")
|
|
return 1
|
|
|
|
print("\nDone.")
|
|
return 0
|
|
|
|
|
|
def main(argv: list[str]) -> int:
|
|
try:
|
|
return run(argv)
|
|
finally:
|
|
pause_if_interactive()
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main(sys.argv))
|