#!/usr/bin/env python3 """ Photo collection metadata cleanup. Drag files or directories onto this script, or run: python3 photo_metadata.py path/to/file/or/directory [...] The script reads metadata with exiftool, builds a full preview, then edits files in place only after confirmation. """ from __future__ import annotations import json import math import os import re import shutil import subprocess import sys import xml.etree.ElementTree as ET from collections import defaultdict from dataclasses import dataclass, field from datetime import datetime, timedelta, timezone from pathlib import Path from shutil import which from typing import Iterable IMAGE_EXTS = {".jpg", ".jpeg", ".heic", ".arw"} VIDEO_EXTS = {".mp4", ".mov", ".mts"} SUPPORTED_EXTS = IMAGE_EXTS | VIDEO_EXTS IMAGE_TIME_TAGS = ( "EXIF:DateTimeOriginal", "EXIF:CreateDate", "EXIF:DateTime", "EXIF:ModifyDate", "XMP:DateTimeOriginal", "XMP:CreateDate", ) IMAGE_OFFSET_TAGS = ( "EXIF:OffsetTimeOriginal", "EXIF:OffsetTimeDigitized", "EXIF:OffsetTime", ) VIDEO_TIME_TAGS = ( "QuickTime:MediaCreateDate", "QuickTime:TrackCreateDate", "QuickTime:CreateDate", "QuickTime:MediaModifyDate", "QuickTime:TrackModifyDate", "QuickTime:ModifyDate", "XMP:DateTimeOriginal", "XMP:CreateDate", ) VIDEO_BEGIN_TIME_TAGS = ( "H264:DateTimeOriginal", ) VIDEO_DURATION_TAGS = ( "QuickTime:Duration", "Composite:Duration", "File:Duration", "M2TS:Duration", ) SEQUENCE_TAGS = ( "MakerNotes:SequenceNumber", "MakerNotes:SequenceNumberOriginal", "MakerNotes:ImageNumber", "EXIF:ImageNumber", "SequenceNumber", ) DATETIME_RE = re.compile( r"(?P\d{4})[:\-]?(?P\d{2})[:\-]?(?P\d{2})" r"(?:[ T_])?" r"(?P\d{2}):?(?P\d{2}):?(?P\d{2})" r"(?:[.,](?P\d+))?" r"(?:\s*(?PZ|[+-]\d{2}:?\d{2}))?" ) TIME_ONLY_RE = re.compile( r"^(?P\d{1,2})(?::?(?P\d{2}))(?::?(?P\d{2}))?$" ) TZ_RE = re.compile(r"^(?P[+-])(?P\d{1,2})(?::?(?P\d{2}))?$") OFFSET_RE = re.compile(r"^([+-])(?:(\d+):)?(\d{1,2})(?::(\d{2}))?$") TOKEN_SHIFT_RE = re.compile(r"(\d+|[a-zA-Z]+)") @dataclass class XmlSidecar: path: Path creation: datetime | None = None duration_seconds: float | None = None timezone_value: timezone | None = None device: str | None = None @dataclass class MediaRecord: path: Path metadata: dict kind: str original_time: datetime | None = None adjusted_time: datetime | None = None timezone_value: timezone | None = None subsec: int | None = None inferred_subsec: int | None = None sequence: int | None = None camera_key: str = "" duration_seconds: float | None = None video_time_is_beginning: bool = False parsed_from_filename: bool = False sidecar: XmlSidecar | None = None new_stem: str | None = None final_name: str | None = None group_id: int | None = None group_name: str | None = None target_path: Path | None = None target_sidecar: Path | None = None rename_skip_reason: str | None = None warnings: list[str] = field(default_factory=list) @property def is_image(self) -> bool: return self.kind == "image" @property def is_video(self) -> bool: return self.kind == "video" @dataclass class UserChoices: working_dir: Path time_offset: timedelta | None explicit_timezone: timezone | None authoritative_timezone: timezone | None write_missing_photo_timezones: bool video_filename_timezone: timezone | None artist_action: str artist_value: str | None rename_files: bool group_photos: bool move_to_working_dir: bool apply_filename_timestamps: bool def ask_yes_no(prompt: str, default: bool) -> bool: suffix = "Y/n" if default else "y/N" while True: answer = prompt_input(f"{prompt} ({suffix}) ").strip().lower() if not answer: return default if answer in {"y", "yes"}: return True if answer in {"n", "no"}: return False print("Please answer y or n.") def prompt_input(prompt: str) -> str: try: return input(prompt) except EOFError as exc: raise RuntimeError("Interactive input ended before choices were complete.") from exc def parse_timezone_offset(value: str) -> timezone: match = TZ_RE.match(value.strip()) if not match: raise ValueError("timezone must look like +09:00 or -05:30") sign = 1 if match.group("sign") == "+" else -1 hours = int(match.group("h")) minutes = int(match.group("m") or "0") if minutes not in {0, 15, 30, 45}: raise ValueError("timezone minutes must be 00, 15, 30, or 45") return timezone(sign * timedelta(hours=hours, minutes=minutes)) def timezone_to_string(tz: timezone) -> str: offset = tz.utcoffset(None) if offset is None: return "+00:00" total_minutes = int(offset.total_seconds() // 60) sign = "+" if total_minutes >= 0 else "-" total_minutes = abs(total_minutes) return f"{sign}{total_minutes // 60:02d}:{total_minutes % 60:02d}" def parse_datetime(value: str, default_date: datetime | None = None) -> datetime: text = value.strip() match = DATETIME_RE.search(text) if match: tzinfo = None tz_text = match.group("tz") if tz_text == "Z": tzinfo = timezone.utc elif tz_text: tzinfo = parse_timezone_offset(tz_text) return datetime( int(match.group("Y")), int(match.group("M")), int(match.group("D")), int(match.group("h")), int(match.group("m")), int(match.group("s")), tzinfo=tzinfo, ) match = TIME_ONLY_RE.match(text) if match and default_date is not None: return default_date.replace( hour=int(match.group("h")), minute=int(match.group("m")), second=int(match.group("s") or "0"), microsecond=0, ) raise ValueError("timestamp must include YYYYMMDD_HHMMSS or a time with reference date") def parse_exif_datetime(value: object, assume_utc: bool = False) -> datetime | None: if value is None: return None try: dt = parse_datetime(str(value)) except ValueError: return None if assume_utc and dt.tzinfo is None: dt = dt.replace(tzinfo=timezone.utc) return dt def parse_subsec(value: object) -> int | None: if value is None: return None match = re.search(r"\d+", str(value)) if not match: return None digits = match.group(0)[:3].ljust(3, "0") return int(digits) def parse_sequence(value: object) -> int | None: if value is None: return None match = re.search(r"\d+", str(value)) if not match: return None return int(match.group(0)) def parse_duration_seconds(value: object) -> float | None: if value is None: return None text = str(value).strip() number_match = re.match(r"^([0-9]+(?:\.[0-9]+)?)\s*(?:s|sec|seconds)?$", text, re.I) if number_match: return float(number_match.group(1)) hms_match = re.match( r"^(?:(?P\d+):)?(?P\d{1,2}):(?P\d{1,2}(?:\.\d+)?)$", text ) if hms_match: hours = int(hms_match.group("h") or "0") minutes = int(hms_match.group("m")) seconds = float(hms_match.group("s")) return hours * 3600 + minutes * 60 + seconds embedded = re.search(r"([0-9]+(?:\.[0-9]+)?)\s*s", text, re.I) if embedded: return float(embedded.group(1)) return None def round_half_up(value: float) -> int: return int(math.floor(value + 0.5)) def parse_timeshift(value: str) -> timedelta: text = value.strip().replace(" ", "") if not text or text[0] not in "+-": raise ValueError("time shift must start with + or -") sign = 1 if text[0] == "+" else -1 body = text[1:] if re.fullmatch(r"\d+(?::\d{2}){0,2}", body): parts = [int(part) for part in body.split(":")] parts = [0] * (3 - len(parts)) + parts return sign * timedelta(hours=parts[0], minutes=parts[1], seconds=parts[2]) units = { "h": 3600, "hr": 3600, "hrs": 3600, "hour": 3600, "hours": 3600, "m": 60, "min": 60, "mins": 60, "minute": 60, "minutes": 60, "s": 1, "sec": 1, "secs": 1, "second": 1, "seconds": 1, } tokens = TOKEN_SHIFT_RE.findall(body) if not tokens or len(tokens) % 2 != 0: raise ValueError("time shift units must look like +1h30m or -2 minutes") seconds = 0 for amount, unit in zip(tokens[::2], tokens[1::2]): if not amount.isdigit() or unit.lower() not in units: raise ValueError("invalid time shift unit") seconds += int(amount) * units[unit.lower()] return sign * timedelta(seconds=seconds) def naive_wall_time(dt: datetime) -> datetime: return dt.replace(tzinfo=None) def collect_media_files(args: list[str]) -> list[Path]: files: list[Path] = [] seen: set[Path] = set() for arg in args: path = Path(arg).expanduser() if path.is_file(): candidates = [path] if path.suffix.lower() in SUPPORTED_EXTS else [] elif path.is_dir(): candidates = [ child for child in path.rglob("*") if child.is_file() and child.suffix.lower() in SUPPORTED_EXTS ] else: print(f"Skipping missing path: {arg}") continue for candidate in candidates: resolved = candidate.resolve() if resolved not in seen: seen.add(resolved) files.append(resolved) return sorted(files, key=lambda p: str(p).lower()) def existing_input_paths(args: list[str]) -> list[Path]: paths: list[Path] = [] for arg in args: path = Path(arg).expanduser() if path.exists(): paths.append(path.resolve()) return paths def determine_working_dir(args: list[str]) -> Path: paths = existing_input_paths(args) if not paths: raise RuntimeError("No existing input paths were provided.") if len(paths) == 1: path = paths[0] return path if path.is_dir() else path.parent try: common = Path(os.path.commonpath([str(path) for path in paths])) except ValueError as exc: raise RuntimeError("Input paths do not have a common ancestor.") from exc if str(common) == common.anchor: raise RuntimeError("Input paths only share the filesystem root; choose items closer together.") return common def require_exiftool() -> str: executable = which("exiftool") if executable: return executable raise RuntimeError( "exiftool was not found on PATH. Install ExifTool and make sure the " "'exiftool' command works before running this script." ) def run_exiftool_json(executable: str, paths: list[Path]) -> list[dict]: results: list[dict] = [] for start in range(0, len(paths), 80): chunk = paths[start : start + 80] command = [ executable, "-j", "-G", "-api", "QuickTimeUTC=1", "-charset", "filename=UTF8", *[str(path) for path in chunk], ] completed = subprocess.run(command, check=True, capture_output=True, text=True) results.extend(json.loads(completed.stdout or "[]")) return results def run_exiftool_write(executable: str, path: Path, args: list[str]) -> None: if not args: return command = [ executable, "-overwrite_original", "-charset", "filename=UTF8", *args, str(path), ] subprocess.run(command, check=True) def first_tag(metadata: dict, tags: Iterable[str]) -> object | None: for tag in tags: if tag in metadata: return metadata[tag] return None def find_sidecar(video_path: Path) -> XmlSidecar | None: expected = f"{video_path.stem}M01.XML" for child in video_path.parent.iterdir(): if child.is_file() and child.name.lower() == expected.lower(): return parse_xml_sidecar(child) return None def parse_xml_sidecar(path: Path) -> XmlSidecar: sidecar = XmlSidecar(path=path) try: root = ET.parse(path).getroot() except ET.ParseError as exc: sidecar.device = f"unreadable XML: {exc}" return sidecar duration_value: float | None = None fps_value: float | None = None def strip_ns(name: str) -> str: return name.rsplit("}", 1)[-1] for element in root.iter(): name = strip_ns(element.tag) if name == "CreationDate": value = element.attrib.get("value") if value: sidecar.creation = parse_exif_datetime(value) if sidecar.creation and sidecar.creation.tzinfo: sidecar.timezone_value = sidecar.creation.tzinfo elif name == "Duration": value = element.attrib.get("value") if value: duration_value = parse_duration_seconds(value) elif name == "VideoFrame": fps_text = element.attrib.get("captureFps") or element.attrib.get("formatFps") if fps_text: fps_match = re.search(r"\d+(?:\.\d+)?", fps_text) if fps_match: fps_value = float(fps_match.group(0)) elif name == "Device": manufacturer = element.attrib.get("manufacturer") model = element.attrib.get("modelName") sidecar.device = " ".join(part for part in (manufacturer, model) if part) if duration_value is not None and fps_value: sidecar.duration_seconds = duration_value / fps_value else: sidecar.duration_seconds = duration_value return sidecar def build_records(paths: list[Path], metadata_items: list[dict]) -> list[MediaRecord]: by_source = { Path(item.get("SourceFile", "")).resolve(): item for item in metadata_items if item.get("SourceFile") } records: list[MediaRecord] = [] for path in paths: metadata = by_source.get(path.resolve(), {}) ext = path.suffix.lower() kind = "video" if ext in VIDEO_EXTS else "image" record = MediaRecord(path=path, metadata=metadata, kind=kind) if record.is_image: record.original_time = parse_exif_datetime(first_tag(metadata, IMAGE_TIME_TAGS)) record.adjusted_time = record.original_time offset_value = first_tag(metadata, IMAGE_OFFSET_TAGS) if offset_value: try: record.timezone_value = parse_timezone_offset(str(offset_value)) except ValueError: record.warnings.append(f"Invalid timezone offset metadata: {offset_value}") record.subsec = parse_subsec( first_tag( metadata, ( "EXIF:SubSecTimeOriginal", "EXIF:SubSecTimeDigitized", "EXIF:SubSecTime", ), ) ) else: begin_time_value = first_tag(metadata, VIDEO_BEGIN_TIME_TAGS) if ext == ".mts" and begin_time_value is not None: record.original_time = parse_exif_datetime(begin_time_value, assume_utc=True) record.video_time_is_beginning = True else: record.original_time = parse_exif_datetime( first_tag(metadata, VIDEO_TIME_TAGS), assume_utc=True ) record.adjusted_time = record.original_time record.timezone_value = timezone.utc record.duration_seconds = parse_duration_seconds(first_tag(metadata, VIDEO_DURATION_TAGS)) record.sidecar = find_sidecar(path) if record.sidecar and record.duration_seconds is not None: xml_duration = record.sidecar.duration_seconds if xml_duration is not None and abs(xml_duration - record.duration_seconds) > 1: record.warnings.append( f"XML duration {xml_duration:g}s differs from metadata duration " f"{record.duration_seconds:g}s" ) record.sequence = parse_sequence(first_tag(metadata, SEQUENCE_TAGS)) record.camera_key = camera_key(record) records.append(record) return records def camera_key(record: MediaRecord) -> str: metadata = record.metadata make = first_tag(metadata, ("EXIF:Make", "QuickTime:Make", "MakerNotes:Make")) or "" if not make: make = first_tag(metadata, ("H264:Make",)) or "" model = first_tag(metadata, ("EXIF:Model", "QuickTime:Model", "MakerNotes:Model")) or "" if not model: model = first_tag(metadata, ("H264:Model",)) or "" serial = first_tag( metadata, ( "EXIF:SerialNumber", "MakerNotes:SerialNumber", "MakerNotes:InternalSerialNumber", "Composite:SerialNumber", ), ) if make or model or serial: return "|".join(str(part).strip() for part in (make, model, serial) if part) return f"dir:{record.path.parent.resolve()}" def parse_filename_timestamp(record: MediaRecord) -> datetime | None: try: return parse_datetime(record.path.stem) except ValueError: return None def resolve_reference_time(records: list[MediaRecord], value: str) -> datetime: text = value.strip() lower = text.lower() for record in records: candidates = { str(record.path).lower(), record.path.name.lower(), record.path.stem.lower(), } if lower in candidates: if record.adjusted_time is None: raise ValueError(f"{value} has no readable timestamp") return naive_wall_time(record.adjusted_time) return naive_wall_time(parse_datetime(text)) def infer_reference_target(source_time: datetime, target_text: str) -> datetime: try: return naive_wall_time(parse_datetime(target_text)) except ValueError: pass base = naive_wall_time(source_time) target_same_date = parse_datetime(target_text, default_date=base) candidates = [ target_same_date - timedelta(days=1), target_same_date, target_same_date + timedelta(days=1), ] def score(candidate: datetime) -> tuple[int, float]: delta_hours = abs((candidate - base).total_seconds()) / 3600 same_date_penalty = 0 if candidate.date() == base.date() else 1 return (0 if delta_hours <= 12 else 1, delta_hours + same_date_penalty) return min(candidates, key=score) def prompt_time_offset(records: list[MediaRecord]) -> timedelta | None: while True: mode = prompt_input("Time correction: Enter=none, o=offset, r=reference photo/clock: ").strip().lower() if mode == "": return None if mode in {"o", "offset"}: while True: text = prompt_input("Enter time shift (+1:30, -02:00:00, +1h 2m): ").strip() try: return parse_timeshift(text) except ValueError as exc: print(exc) if mode in {"r", "ref", "reference"}: while True: source_text = prompt_input("Reference source timestamp, filename, or path: ").strip() target_text = prompt_input("Correct reference-clock time: ").strip() try: source_time = resolve_reference_time(records, source_text) target_time = infer_reference_target(source_time, target_text) offset = target_time - source_time print(f"Computed time shift: {format_timedelta(offset)}") return offset except ValueError as exc: print(exc) print("Please choose Enter, o, or r.") def format_timedelta(value: timedelta) -> str: total_seconds = int(value.total_seconds()) sign = "+" if total_seconds >= 0 else "-" total_seconds = abs(total_seconds) hours = total_seconds // 3600 minutes = (total_seconds % 3600) // 60 seconds = total_seconds % 60 return f"{sign}{hours:02d}:{minutes:02d}:{seconds:02d}" def prompt_timezone(records: list[MediaRecord], rename_files: bool) -> tuple[timezone | None, timezone | None, bool, timezone | None]: while True: text = prompt_input("Enter timezone offset (+09:00), or blank to keep/infer: ").strip() if not text: explicit = None break try: explicit = parse_timezone_offset(text) break except ValueError as exc: print(exc) photo_zones = { timezone_to_string(record.timezone_value): record.timezone_value for record in records if record.is_image and record.timezone_value is not None } missing_photo_zones = [ record for record in records if record.is_image and record.timezone_value is None ] authoritative = explicit if authoritative is None and len(photo_zones) == 1: authoritative = next(iter(photo_zones.values())) elif authoritative is None and len(photo_zones) > 1: print("Photo timezone offsets conflict; no authoritative timezone inferred.") write_missing = False if authoritative is not None and explicit is not None: write_missing = True elif authoritative is not None and missing_photo_zones and photo_zones: write_missing = ask_yes_no( f"Write inferred timezone {timezone_to_string(authoritative)} to " f"{len(missing_photo_zones)} photo(s) missing it?", default=True, ) video_tz = None has_videos = any(record.is_video for record in records) if rename_files and has_videos: if authoritative is not None and explicit is None: if ask_yes_no( f"Use photo timezone {timezone_to_string(authoritative)} for video filenames?", default=True, ): video_tz = authoritative elif authoritative is not None: video_tz = authoritative while video_tz is None: text = prompt_input("Video renaming needs local timezone (+09:00): ").strip() try: video_tz = parse_timezone_offset(text) except ValueError as exc: print(exc) return explicit, authoritative, write_missing, video_tz def prompt_artist_action() -> tuple[str, str | None]: while True: text = prompt_input("Artist/author: Enter=leave unchanged, s=set, c=clear: ").strip().lower() if text == "": return "leave", None if text in {"c", "clear"}: return "clear", None if text in {"s", "set"}: value = prompt_input("Artist/author value: ").strip() if value: return "set", value print("Use clear if you want an empty artist/author.") else: print("Please choose Enter, s, or c.") def prompt_choices(records: list[MediaRecord], working_dir: Path) -> UserChoices: missing_time = [record for record in records if record.original_time is None] apply_filename_timestamps = False if missing_time: apply_filename_timestamps = ask_yes_no( f"{len(missing_time)} file(s) have no readable metadata timestamp. " "Try timestamp from filename?", default=True, ) if apply_filename_timestamps: for record in missing_time: parsed = parse_filename_timestamp(record) if parsed is not None: record.original_time = parsed record.adjusted_time = parsed record.parsed_from_filename = True time_offset = prompt_time_offset(records) rename_files = ask_yes_no("Rename files to timestamps?", default=True) explicit_tz, authoritative_tz, write_missing_tz, video_tz = prompt_timezone(records, rename_files) artist_action, artist_value = prompt_artist_action() group_photos = False if sum(1 for record in records if record.is_image and record.adjusted_time) >= 2: group_photos = ask_yes_no("Create automatic burst/HDR groups?", default=True) move_to_working_dir = False if any(record.path.parent.resolve() != working_dir for record in records): move_to_working_dir = ask_yes_no( f"Move ungrouped files from subdirectories into the working directory ({working_dir})?", default=False, ) return UserChoices( working_dir=working_dir, time_offset=time_offset, explicit_timezone=explicit_tz, authoritative_timezone=authoritative_tz, write_missing_photo_timezones=write_missing_tz, video_filename_timezone=video_tz, artist_action=artist_action, artist_value=artist_value, rename_files=rename_files, group_photos=group_photos, move_to_working_dir=move_to_working_dir, apply_filename_timestamps=apply_filename_timestamps, ) def apply_time_offset(records: list[MediaRecord], offset: timedelta | None) -> None: if offset is None: return for record in records: if record.adjusted_time is not None: record.adjusted_time = record.adjusted_time + offset def photo_sort_key(record: MediaRecord) -> tuple: assert record.adjusted_time is not None return ( naive_wall_time(record.adjusted_time), record.subsec if record.subsec is not None else -1, record.sequence if record.sequence is not None else -1, record.path.name.lower(), ) def detect_photo_groups(records: list[MediaRecord]) -> list[list[MediaRecord]]: groups: list[list[MediaRecord]] = [] by_camera: dict[str, list[MediaRecord]] = defaultdict(list) for record in records: if record.is_image and record.adjusted_time is not None: by_camera[record.camera_key].append(record) for camera_records in by_camera.values(): groups.extend(detect_groups_for_camera(sorted(camera_records, key=photo_sort_key))) return groups def detect_groups_for_camera(records: list[MediaRecord]) -> list[list[MediaRecord]]: by_second: dict[datetime, list[MediaRecord]] = defaultdict(list) for record in records: assert record.adjusted_time is not None by_second[naive_wall_time(record.adjusted_time).replace(microsecond=0)].append(record) seconds = sorted(by_second) spans: list[list[datetime]] = [] current: list[datetime] = [] for second in seconds: if not current or second == current[-1] + timedelta(seconds=1): current.append(second) else: spans.append(current) current = [second] if current: spans.append(current) groups: list[list[MediaRecord]] = [] for span in spans: if max(len(by_second[second]) for second in span) < 2: continue span_records = [record for second in span for record in by_second[second]] if all(record.sequence is not None for record in span_records) and len(span_records) >= 2: groups.extend(sequence_groups(span_records)) else: groups.extend(timestamp_groups(span, by_second)) return [group for group in groups if len(group) >= 2] def sequence_groups(records: list[MediaRecord]) -> list[list[MediaRecord]]: ordered = sorted(records, key=photo_sort_key) groups: list[list[MediaRecord]] = [] current: list[MediaRecord] = [ordered[0]] for record in ordered[1:]: previous = current[-1] if record.sequence is not None and previous.sequence is not None and record.sequence > previous.sequence: current.append(record) else: if len(current) >= 2: groups.append(current) current = [record] if len(current) >= 2: groups.append(current) return groups def timestamp_groups( span: list[datetime], by_second: dict[datetime, list[MediaRecord]] ) -> list[list[MediaRecord]]: multi_indices = [index for index, second in enumerate(span) if len(by_second[second]) >= 2] if not multi_indices: return [] clusters: list[list[int]] = [[multi_indices[0]]] for index in multi_indices[1:]: if index - clusters[-1][-1] <= 2: clusters[-1].append(index) else: clusters.append([index]) groups: list[list[MediaRecord]] = [] used_seconds: set[datetime] = set() for cluster in clusters: start = cluster[0] end = cluster[-1] if start > 0 and len(by_second[span[start - 1]]) == 1: start -= 1 if end + 1 < len(span) and len(by_second[span[end + 1]]) == 1: end += 1 seconds = [second for second in span[start : end + 1] if second not in used_seconds] for second in seconds: used_seconds.add(second) records = [record for second in seconds for record in sorted(by_second[second], key=photo_sort_key)] if len(records) >= 2: groups.append(records) return groups def infer_group_subseconds(groups: list[list[MediaRecord]]) -> None: for group in groups: by_second: dict[datetime, list[MediaRecord]] = defaultdict(list) for record in group: assert record.adjusted_time is not None second = naive_wall_time(record.adjusted_time).replace(microsecond=0) by_second[second].append(record) seconds = sorted(by_second) counts = {second: len(records) for second, records in by_second.items()} for index, second in enumerate(seconds): records = sorted(by_second[second], key=photo_sort_key) count = len(records) if count <= 1: continue fps = count if len(seconds) == 2: fps = max(counts[seconds[0]], counts[seconds[1]]) elif index == 0 and len(seconds) > 1: fps = max(count, counts[seconds[index + 1]]) elif index == len(seconds) - 1 and len(seconds) > 1: fps = max(count, counts[seconds[index - 1]]) start_slot = 0 if index == 0 and fps > count: start_slot = fps - count for item_index, record in enumerate(records): record.inferred_subsec = int((start_slot + item_index) * 1000 / fps) def format_filename_stem(dt: datetime) -> str: return naive_wall_time(dt).strftime("%Y%m%d_%H%M%S") def normalized_extension(path: Path) -> str: return path.suffix.lower() def plan_names(records: list[MediaRecord], choices: UserChoices) -> None: if not choices.rename_files: for record in records: record.final_name = record.path.name return for record in records: if record.adjusted_time is None: record.rename_skip_reason = "missing timestamp" continue if record.is_video: if record.duration_seconds is None and not record.parsed_from_filename: record.rename_skip_reason = "missing duration" continue if choices.video_filename_timezone is None: record.rename_skip_reason = "missing video timezone" continue if record.parsed_from_filename and record.original_time is not None: begin = record.adjusted_time elif record.video_time_is_beginning: begin = record.adjusted_time else: duration = round_half_up(record.duration_seconds or 0) begin = record.adjusted_time - timedelta(seconds=duration) if begin.tzinfo is None: begin = begin.replace(tzinfo=timezone.utc) local_begin = begin.astimezone(choices.video_filename_timezone) record.new_stem = format_filename_stem(local_begin) else: record.new_stem = format_filename_stem(record.adjusted_time) assign_unique_names(records) def assign_unique_names(records: list[MediaRecord]) -> None: buckets: dict[tuple[int | str, str], list[MediaRecord]] = defaultdict(list) for record in records: if record.new_stem is None: record.final_name = record.path.name continue container: int | str = record.group_id if record.group_id is not None else "all" buckets[(container, normalized_extension(record.path))].append(record) for (_container, ext), bucket in buckets.items(): by_stem: dict[str, list[MediaRecord]] = defaultdict(list) for record in sorted(bucket, key=lambda item: (item.new_stem or "", item.path.name.lower())): assert record.new_stem is not None by_stem[record.new_stem].append(record) for stem, records_with_stem in by_stem.items(): if len(records_with_stem) == 1: records_with_stem[0].final_name = f"{stem}{ext}" continue for index, record in enumerate(records_with_stem, start=1): record.final_name = f"{stem}-{index}{ext}" def assign_groups(records: list[MediaRecord], groups: list[list[MediaRecord]]) -> None: for index, group in enumerate(groups, start=1): ordered = sorted(group, key=photo_sort_key) for record in ordered: record.group_id = index def assign_group_names(groups: list[list[MediaRecord]]) -> None: for group in groups: ordered = sorted(group, key=photo_sort_key) first = Path(ordered[0].final_name or ordered[0].path.name).stem last = Path(ordered[-1].final_name or ordered[-1].path.name).stem group_name = f"{first}-{last}" for record in ordered: record.group_name = group_name def plan_targets(records: list[MediaRecord], choices: UserChoices) -> None: for record in records: if record.final_name is None: record.final_name = record.path.name out_dir = record.path.parent if record.group_name: out_dir = choices.working_dir / record.group_name elif choices.move_to_working_dir: out_dir = choices.working_dir record.target_path = out_dir / record.final_name if record.is_video and record.sidecar is not None: target_sidecar = out_dir / f"{Path(record.final_name).stem}M01.XML" if target_sidecar.resolve() != record.sidecar.path.resolve(): record.target_sidecar = target_sidecar def build_write_args(record: MediaRecord, choices: UserChoices) -> list[str]: args: list[str] = [] if choices.artist_action == "set": assert choices.artist_value is not None args.extend([f"-Artist={choices.artist_value}", f"-Author={choices.artist_value}"]) elif choices.artist_action == "clear": args.extend(["-Artist=", "-Author="]) if record.is_image: if choices.time_offset is not None or record.parsed_from_filename: if record.adjusted_time is not None: args.append(f"-AllDates={record.adjusted_time.strftime('%Y:%m:%d %H:%M:%S')}") should_write_tz = False tz_to_write = choices.explicit_timezone if tz_to_write is not None: should_write_tz = True elif ( choices.write_missing_photo_timezones and record.timezone_value is None and choices.authoritative_timezone is not None ): tz_to_write = choices.authoritative_timezone should_write_tz = True if should_write_tz and tz_to_write is not None: tz_text = timezone_to_string(tz_to_write) args.extend( [ f"-EXIF:OffsetTimeOriginal={tz_text}", f"-EXIF:OffsetTimeDigitized={tz_text}", f"-EXIF:OffsetTime={tz_text}", ] ) if record.inferred_subsec is not None: args.extend( [ f"-SubSecTimeOriginal={record.inferred_subsec:03d}", f"-SubSecTimeDigitized={record.inferred_subsec:03d}", f"-SubSecTime={record.inferred_subsec:03d}", ] ) elif record.is_video and choices.time_offset is not None and record.adjusted_time is not None: utc_time = record.adjusted_time if utc_time.tzinfo is None: utc_time = utc_time.replace(tzinfo=timezone.utc) utc_text = utc_time.astimezone(timezone.utc).strftime("%Y:%m:%d %H:%M:%S") args.extend( [ f"-QuickTime:CreateDate={utc_text}", f"-QuickTime:ModifyDate={utc_text}", f"-QuickTime:TrackCreateDate={utc_text}", f"-QuickTime:TrackModifyDate={utc_text}", f"-QuickTime:MediaCreateDate={utc_text}", f"-QuickTime:MediaModifyDate={utc_text}", ] ) return args def print_preview(records: list[MediaRecord], choices: UserChoices) -> None: print("\nPreview") print(f"Working directory: {choices.working_dir}") print(f"Move ungrouped files into working directory: {'yes' if choices.move_to_working_dir else 'no'}") print(f"Time shift: {format_timedelta(choices.time_offset) if choices.time_offset else 'none'}") print( "Photo timezone write: " + ( timezone_to_string(choices.explicit_timezone) if choices.explicit_timezone else ( f"missing only -> {timezone_to_string(choices.authoritative_timezone)}" if choices.write_missing_photo_timezones and choices.authoritative_timezone else "unchanged" ) ) ) print( "Video filename timezone: " + (timezone_to_string(choices.video_filename_timezone) if choices.video_filename_timezone else "n/a") ) if choices.artist_action == "set": print(f"Artist/author: set to {choices.artist_value}") elif choices.artist_action == "clear": print("Artist/author: clear") else: print("Artist/author: unchanged") rows = [] for record in sorted(records, key=lambda item: str(item.path).lower()): target = record.target_path or record.path operation = [] if target != record.path: operation.append(f"rename/move -> {target}") if record.inferred_subsec is not None: operation.append(f"subsec {record.inferred_subsec:03d}") if record.rename_skip_reason: operation.append(f"rename skipped: {record.rename_skip_reason}") for warning in record.warnings: operation.append(f"warning: {warning}") if record.target_sidecar and record.sidecar: operation.append(f"sidecar -> {record.target_sidecar}") rows.append((str(record.path), "; ".join(operation) or "metadata only/no change")) path_width = min(max((len(row[0]) for row in rows), default=10), 70) for source, operation in rows: print(f"{source:<{path_width}} {operation}") def unique_existing_target(path: Path, planned_sources: set[Path]) -> Path: if not path.exists() or path.resolve() in planned_sources: return path for index in range(1, 10000): candidate = path.with_name(f"{path.stem}-{index}{path.suffix}") if not candidate.exists() or candidate.resolve() in planned_sources: return candidate raise RuntimeError(f"Could not find available target name for {path}") def apply_changes(executable: str, records: list[MediaRecord], choices: UserChoices) -> list[str]: failures: list[str] = [] planned_sources = {record.path.resolve() for record in records} for index, record in enumerate(records, start=1): try: write_args = build_write_args(record, choices) run_exiftool_write(executable, record.path, write_args) if record.target_path and record.target_path != record.path: target = unique_existing_target(record.target_path, planned_sources) target.parent.mkdir(parents=True, exist_ok=True) shutil.move(str(record.path), str(target)) record.target_path = target if record.sidecar and record.target_sidecar: target_sidecar = unique_existing_target(record.target_sidecar, planned_sources) target_sidecar.parent.mkdir(parents=True, exist_ok=True) if record.sidecar.path.exists(): shutil.move(str(record.sidecar.path), str(target_sidecar)) except (OSError, subprocess.CalledProcessError, RuntimeError) as exc: failures.append(f"{record.path}: {exc}") print(f"Processed {index} of {len(records)}") return failures def prepare_plan(records: list[MediaRecord], choices: UserChoices) -> list[list[MediaRecord]]: apply_time_offset(records, choices.time_offset) groups: list[list[MediaRecord]] = [] if choices.group_photos: groups = detect_photo_groups(records) assign_groups(records, groups) infer_group_subseconds(groups) plan_names(records, choices) if groups: assign_group_names(groups) plan_targets(records, choices) return groups def main(argv: list[str]) -> int: if len(argv) < 2: print("Usage: python3 photo_metadata.py path/to/file/or/directory [...]") return 2 try: working_dir = determine_working_dir(argv[1:]) except RuntimeError as exc: print(exc) return 1 files = collect_media_files(argv[1:]) if not files: print("No supported media files found.") return 1 try: exiftool = require_exiftool() except RuntimeError as exc: print(exc) return 1 print(f"Found {len(files)} supported media file(s). Reading metadata...") try: metadata = run_exiftool_json(exiftool, files) except (subprocess.CalledProcessError, json.JSONDecodeError) as exc: print(f"Failed to read metadata with exiftool: {exc}") return 1 records = build_records(files, metadata) try: choices = prompt_choices(records, working_dir) except RuntimeError as exc: print(exc) return 1 prepare_plan(records, choices) print_preview(records, choices) try: proceed = ask_yes_no("Proceed with these changes?", default=True) except RuntimeError as exc: print(exc) return 1 if not proceed: print("Nothing changed.") return 0 failures = apply_changes(exiftool, records, choices) if failures: print("\nSome files failed:") for failure in failures: print(f" - {failure}") return 1 print("\nDone.") return 0 if __name__ == "__main__": raise SystemExit(main(sys.argv))