"""Pure planning for photo metadata updates, grouping, and PTO projects.""" from __future__ import annotations import os import re from collections import defaultdict from dataclasses import dataclass from datetime import datetime, timedelta, timezone from pathlib import Path from tools.filenames import format_filename_stem, group_stem_from_name, naive_wall_time, normalized_extension from tools.media_metadata import first_tag, parse_exif_datetime, round_half_up, tag_value from tools.photo_grouping import detect_photo_groups, group_sort_key, infer_group_subseconds from tools.photo_records import FILENAME_TIMESTAMP_RE, IMAGE_MODIFY_TIME_TAGS, MediaRecord, TagWrite, parse_filename_timestamp, parse_subsec from tools.timezones import timezone_to_string IMAGE_WRITE_TIME_TAGS = ( "EXIF:DateTimeOriginal", "EXIF:CreateDate", "XMP:DateTimeOriginal", "XMP:CreationDate", "XMP:CreateDate", "EXIF:ModifyDate", "XMP:ModifyDate", ) OFFSET_TAG_FOR_TIME_TAG = { "EXIF:DateTimeOriginal": "EXIF:OffsetTimeOriginal", "EXIF:CreateDate": "EXIF:OffsetTimeDigitized", "EXIF:ModifyDate": "EXIF:OffsetTime", } CAMERA_RANGE_STEM_RE = re.compile(r"^(?P[A-Za-z]*)(?P\d+)-(?P=prefix)(?P\d+)(?P_.*)$") PTO_QUOTED_VALUE_RE = re.compile(r"([\"'])(.*?)(\1)") @dataclass class UserChoices: working_dir: Path time_offset: timedelta | None fixed_timezone: timezone | None fill_timezone_gaps: bool artist_action: str artist_value: str | None rename_mode: str organize_files: bool group_min_size: int process_pto_files: bool infer_missing_timestamps: bool @property def group_photos(self) -> bool: return self.organize_files and self.group_min_size >= 2 @property def move_to_working_dir(self) -> bool: return self.organize_files @dataclass class PtoPlan: path: Path target_path: Path referenced_records: list[MediaRecord] updated_text: str replacements: list[tuple[str, str]] @dataclass class PtoReference: path: Path original_text: str records: list[MediaRecord] def apply_time_offset(records: list[MediaRecord], offset: timedelta | None) -> None: if offset is not None: for record in records: if record.adjusted_time is not None: record.adjusted_time += offset def infer_missing_timestamps(records: list[MediaRecord], enabled: bool) -> None: if not enabled: return for record in records: if record.original_time is not None: continue parsed, source = parse_filename_timestamp(record), "filename" if parsed is None and record.is_image: parsed, source = parse_exif_datetime(first_tag(record.metadata, IMAGE_MODIFY_TIME_TAGS)), "ModifyDate" if parsed is None: continue if record.is_video: if record.resolved_timezone is None or record.duration_seconds is None: continue parsed = parsed.replace(tzinfo=record.resolved_timezone).astimezone(timezone.utc) parsed += timedelta(seconds=record.duration_seconds) record.original_time = record.adjusted_time = parsed record.inferred_timestamp_source = source record.parsed_from_filename = source == "filename" def filename_timestamp_stem(stem: str, timestamp: datetime, mode: str) -> str: matches = [(match, parse_exif_datetime(match.group(0))) for match in FILENAME_TIMESTAMP_RE.finditer(stem)] matches = [(match, value) for match, value in matches if value is not None] if mode.startswith("adjust") and matches: first_time = matches[0][1] pieces, position, base = [], 0, naive_wall_time(timestamp) for match, old_time in matches: replacement = format_filename_stem(base + (old_time - first_time)).replace("_", match.group(0)[8]) pieces.extend((stem[position:match.start()], replacement)) position = match.end() return "".join((*pieces, stem[position:])) formatted = format_filename_stem(timestamp) return f"{formatted} {stem}" if mode.endswith("add") else formatted def plan_names(records: list[MediaRecord], choices: UserChoices) -> None: by_stem = _unique_records_by_stem(records) for record in records: record.new_stem = record.rename_skip_reason = None if record.adjusted_time is None: record.new_stem = _camera_range_stem(record, by_stem) record.rename_skip_reason = None if record.new_stem else "missing timestamp" elif record.is_video: if record.duration_seconds is None: record.rename_skip_reason = "missing duration" elif record.resolved_timezone is None: record.rename_skip_reason = "missing video timezone" else: begin = record.adjusted_time if record.video_time_is_beginning else record.adjusted_time - timedelta(seconds=round_half_up(record.duration_seconds)) if begin.tzinfo is None: begin = begin.replace(tzinfo=timezone.utc) record.new_stem = filename_timestamp_stem(record.path.stem, begin.astimezone(record.resolved_timezone), choices.rename_mode) else: record.new_stem = filename_timestamp_stem(record.path.stem, record.adjusted_time, choices.rename_mode) _assign_unique_names(records) def _unique_records_by_stem(records: list[MediaRecord]) -> dict[str, MediaRecord]: buckets: dict[str, list[MediaRecord]] = defaultdict(list) for record in records: buckets[record.path.stem.lower()].append(record) return {stem: bucket[0] for stem, bucket in buckets.items() if len(bucket) == 1} def _camera_range_stem(record: MediaRecord, by_stem: dict[str, MediaRecord]) -> str | None: match = CAMERA_RANGE_STEM_RE.match(record.path.stem) if not match or int(match.group("last")) <= int(match.group("first")): return None prefix = match.group("prefix") first = by_stem.get(f"{prefix}{match.group('first')}".lower()) last = by_stem.get(f"{prefix}{match.group('last')}".lower()) start = record.adjusted_time or (first.adjusted_time if first else None) or (last.adjusted_time if last else None) end = (last.adjusted_time if last else None) or (first.adjusted_time if first else None) or start return f"{format_filename_stem(start)}-{format_filename_stem(end)}{match.group('suffix')}" if start and end else None def _assign_unique_names(records: list[MediaRecord]) -> None: buckets: dict[tuple[int | str, str], list[MediaRecord]] = defaultdict(list) for record in records: if record.new_stem: buckets[(record.group_id if record.group_id is not None else str(record.path.parent.resolve()), record.new_stem.lower())].append(record) for bucket in buckets.values(): if len(bucket) == 1: bucket[0].final_name = f"{bucket[0].new_stem}{normalized_extension(bucket[0].path)}" continue for index, record in enumerate(sorted(bucket, key=group_sort_key), 1): record.final_name = f"{record.new_stem}-{index}{normalized_extension(record.path)}" def merge_record_groups(records: list[MediaRecord], candidates: list[list[MediaRecord]]) -> list[list[MediaRecord]]: indexes = {record.path.resolve(): index for index, record in enumerate(records)} parents = list(range(len(records))) grouped: set[int] = set() def find(index: int) -> int: while parents[index] != index: parents[index] = parents[parents[index]] index = parents[index] return index for candidate in candidates: members = [indexes[record.path.resolve()] for record in candidate if record.path.resolve() in indexes] if members: grouped.update(members) for index in members[1:]: parents[find(index)] = find(members[0]) merged: dict[int, list[MediaRecord]] = defaultdict(list) for index in grouped: merged[find(index)].append(records[index]) return sorted((sorted(group, key=group_sort_key) for group in merged.values()), key=lambda group: group_sort_key(group[0])) def assign_groups(records: list[MediaRecord], groups: list[list[MediaRecord]]) -> None: for record in records: record.group_id = record.group_name = None for index, group in enumerate(groups, 1): for record in group: record.group_id = index def _group_name(base: str, working_dir: Path, used: set[str]) -> str: if (working_dir / base).exists() or base not in used: used.add(base) return base for index in range(1, 10000): name = f"{base}-{index}" if name not in used and not (working_dir / name).exists(): used.add(name) return name raise RuntimeError(f"Could not find an available group directory name for {base}") def assign_group_names(groups: list[list[MediaRecord]], working_dir: Path) -> None: used: set[str] = set() for group in groups: ordered = sorted(group, key=group_sort_key) first, last = (group_stem_from_name(record.final_name or record.path.name) for record in (ordered[0], ordered[-1])) name = _group_name(f"{first}-{last}", working_dir, used) for record in group: record.group_name = name def plan_targets(records: list[MediaRecord], choices: UserChoices) -> None: for record in records: directory = choices.working_dir if choices.organize_files else record.path.parent if record.group_name: directory /= record.group_name name = record.final_name or record.path.name record.target_path = directory / name if record.sidecar: record.target_sidecar = directory / f"{Path(name).stem}M01.XML" def normalize_reference(value: str) -> str: return value.replace("\\", "/") def _pto_mapping(pto_path: Path, records: list[MediaRecord]) -> dict[str, MediaRecord]: basename_counts: dict[str, int] = defaultdict(int) for record in records: basename_counts[record.path.name] += 1 mapping: dict[str, MediaRecord] = {} for record in records: candidates = {normalize_reference(str(record.path.resolve())), normalize_reference(os.path.relpath(record.path.resolve(), pto_path.parent))} if basename_counts[record.path.name] == 1: candidates.add(record.path.name) for candidate in candidates: mapping[normalize_reference(candidate)] = record return mapping def _pto_target(path: Path, directory: Path, base: str, used: set[Path]) -> Path: for index in range(10000): target = directory / f"{base}{'' if index == 0 else f'-{index}'}.pto" if target.resolve() == path.resolve() or (not target.exists() and target not in used): used.add(target) return target raise RuntimeError(f"Could not find an available .pto name for {base}") def read_pto_references(pto_files: list[Path], records: list[MediaRecord]) -> list[PtoReference]: references = [] for path in pto_files: try: text = path.read_text(encoding="utf-8") except UnicodeDecodeError: text = path.read_text(encoding="utf-8", errors="replace") mapping = _pto_mapping(path, records) found = [] for match in PTO_QUOTED_VALUE_RE.finditer(text): record = mapping.get(normalize_reference(match.group(2))) if record is not None and record not in found: found.append(record) if found: references.append(PtoReference(path, text, found)) return references def plan_pto_updates(references: list[PtoReference], records: list[MediaRecord]) -> list[PtoPlan]: plans, used = [], set() for reference in references: referenced = sorted(reference.records, key=group_sort_key) target_dir = referenced[0].target_path.parent if referenced[0].group_name and referenced[0].target_path else reference.path.parent mapping, replacements = _pto_mapping(reference.path, records), [] def replace(match: re.Match) -> str: value = match.group(2) record = mapping.get(normalize_reference(value)) if record is None or record.target_path is None: return match.group(0) replacement = normalize_reference(os.path.relpath(record.target_path.resolve(), target_dir)) if value != replacement: replacements.append((value, replacement)) return f"{match.group(1)}{replacement}{match.group(1)}" text = PTO_QUOTED_VALUE_RE.sub(replace, reference.original_text) base = f"{group_stem_from_name(referenced[0].target_path.name)}-{group_stem_from_name(referenced[-1].target_path.name)}" target = _pto_target(reference.path, target_dir, base, used) if text != reference.original_text or target.resolve() != reference.path.resolve(): plans.append(PtoPlan(reference.path, target, referenced, text, replacements)) return plans def _same_text(current: str | None, value: str) -> bool: return (current or "") == value def _same_datetime(current: str | None, value: str) -> bool: current_time, new_time = parse_exif_datetime(current), parse_exif_datetime(value) return current == value if current_time is None or new_time is None else naive_wall_time(current_time) == naive_wall_time(new_time) def _same_subsec(current: str | None, value: str) -> bool: parsed = parse_subsec(current) return current == value if parsed is None else f"{parsed:03d}" == value def _add_write(writes: list[TagWrite], label: str, tag: str, current: str | None, value: str, same=_same_text) -> None: writes.append(TagWrite(label, f"-{tag}={value}", current, value, not same(current, value))) def build_write_plan(record: MediaRecord, choices: UserChoices) -> list[TagWrite]: writes: list[TagWrite] = [] if choices.artist_action != "leave": value = choices.artist_value or "" for label, tag in (("Artist", "Artist"), ("Author", "Author")): _add_write(writes, label, tag, tag_value(record.metadata, f"EXIF:{tag}") or tag_value(record.metadata, f"XMP:{tag}"), value) if record.is_image: capture_tags: set[str] = set() if record.inferred_timestamp_source and record.adjusted_time: capture_tags.add("EXIF:DateTimeOriginal") _add_write(writes, "DateTimeOriginal", "EXIF:DateTimeOriginal", tag_value(record.metadata, "EXIF:DateTimeOriginal"), record.adjusted_time.strftime("%Y:%m:%d %H:%M:%S"), _same_datetime) if record.original_time and record.adjusted_time: original, adjusted = naive_wall_time(record.original_time), naive_wall_time(record.adjusted_time) for tag in IMAGE_WRITE_TIME_TAGS: current = parse_exif_datetime(tag_value(record.metadata, tag)) if current is None: continue current_time = naive_wall_time(current) matches_capture = abs((current_time - original).total_seconds()) <= 1 if matches_capture: capture_tags.add(tag) target = current_time + choices.time_offset if choices.time_offset and matches_capture else current_time if tag.endswith("ModifyDate") and target < adjusted: target, capture_tags = adjusted, capture_tags | {tag} if "ModifyDate was earlier than capture time and will be clamped" not in record.warnings: record.warnings.append("ModifyDate was earlier than capture time and will be clamped") if target != current_time: _add_write(writes, tag.split(":", 1)[1], tag, tag_value(record.metadata, tag), target.strftime("%Y:%m:%d %H:%M:%S"), _same_datetime) if record.resolved_timezone and (choices.fixed_timezone or record.timezone_value is None): for time_tag, offset_tag in OFFSET_TAG_FOR_TIME_TAG.items(): if time_tag in capture_tags: _add_write(writes, offset_tag.split(":", 1)[1], offset_tag, tag_value(record.metadata, offset_tag), timezone_to_string(record.resolved_timezone)) if record.inferred_subsec is not None: for label, tag in (("SubSecTimeOriginal", "EXIF:SubSecTimeOriginal"), ("SubSecTimeDigitized", "EXIF:SubSecTimeDigitized"), ("SubSecTime", "EXIF:SubSecTime")): _add_write(writes, label, tag, tag_value(record.metadata, tag), f"{record.inferred_subsec:03d}", _same_subsec) elif record.is_video and (choices.time_offset or record.inferred_timestamp_source) and record.adjusted_time: value = record.adjusted_time.replace(tzinfo=record.adjusted_time.tzinfo or timezone.utc).astimezone(timezone.utc).strftime("%Y:%m:%d %H:%M:%S") for tag in ("QuickTime:CreateDate", "QuickTime:ModifyDate", "QuickTime:TrackCreateDate", "QuickTime:TrackModifyDate", "QuickTime:MediaCreateDate", "QuickTime:MediaModifyDate"): _add_write(writes, tag.split(":", 1)[1], tag, tag_value(record.metadata, tag), value, _same_datetime) return writes def build_write_args(record: MediaRecord, choices: UserChoices) -> list[str]: if not record.write_plan: record.write_plan = build_write_plan(record, choices) return [write.arg for write in record.write_plan if write.will_write] def prepare_plan(records: list[MediaRecord], choices: UserChoices, references: list[PtoReference]) -> list[list[MediaRecord]]: infer_missing_timestamps(records, choices.infer_missing_timestamps) apply_time_offset(records, choices.time_offset) burst_groups = detect_photo_groups(records, choices.group_min_size) if choices.group_photos else [] groups = merge_record_groups(records, [*burst_groups, *(reference.records for reference in references if choices.process_pto_files)]) if groups: assign_groups(records, groups) if choices.organize_files: infer_group_subseconds(burst_groups) plan_names(records, choices) if groups: assign_group_names(groups, choices.working_dir) plan_targets(records, choices) for record in records: record.write_plan = build_write_plan(record, choices) return groups