from __future__ import annotations import json import re import statistics import subprocess from collections import Counter from dataclasses import dataclass from datetime import datetime, timedelta, timezone from pathlib import Path from tools.executables import require_executable VIDEO_TIME_TAGS = ( "QuickTime:MediaCreateDate", "QuickTime:TrackCreateDate", "QuickTime:CreateDate", "QuickTime:MediaModifyDate", "QuickTime:TrackModifyDate", "QuickTime:ModifyDate", "XMP:DateTimeOriginal", "XMP:CreateDate", ) VIDEO_BEGIN_TIME_TAGS = ( "H264:DateTimeOriginal", ) VIDEO_TIMEZONE_TAGS = ( "QuickTime:AndroidTimeZone", "QuickTime:TimeZone", "EXIF:OffsetTimeOriginal", "XMP:OffsetTimeOriginal", ) GPS_LATITUDE_TAGS = ( "Composite:GPSLatitude", "EXIF:GPSLatitude", "GPS:GPSLatitude", "QuickTime:GPSLatitude", ) GPS_LONGITUDE_TAGS = ( "Composite:GPSLongitude", "EXIF:GPSLongitude", "GPS:GPSLongitude", "QuickTime:GPSLongitude", ) GPS_POSITION_TAGS = ( "Composite:GPSPosition", "QuickTime:GPSCoordinates", "Composite:GPSCoordinates", ) DATETIME_RE = re.compile( r"(?P\d{4})[:\-]?(?P\d{2})[:\-]?(?P\d{2})" r"(?:[ T_])?" r"(?P\d{2}):?(?P\d{2}):?(?P\d{2})" r"(?:[.,](?P\d+))?" r"(?:\s*(?PZ|[+-]\d{2}:?\d{2}))?" ) DMS_RE = re.compile( r"(?P-?\d+(?:\.\d+)?)\D+" r"(?P\d+(?:\.\d+)?)?\D*" r"(?P\d+(?:\.\d+)?)?\D*" r"(?P[NSEW])?", re.IGNORECASE, ) @dataclass(frozen=True) class FrameTimingSummary: frame_count: int first_pts: float | None last_pts: float | None min_delta_ms: float | None median_delta_ms: float | None max_delta_ms: float | None unique_delta_count: int common_deltas_ms: list[tuple[float, int]] timing_kind: str @dataclass(frozen=True) class AudioStreamProbe: codec: str | None channels: int | None sample_rate: int | None bit_rate: int | None @dataclass(frozen=True) class VideoProbe: path: Path codec: str | None width: int | None height: int | None pixel_format: str | None bit_depth: int | None video_bit_rate: int | None color_space: str | None color_matrix: str | None expected_color_matrix: str | None duration_seconds: float | None frame_timestamps: list[float] stream_frame_count: int | None audio_streams: list[AudioStreamProbe] audio_stream_count: int audio_sample_rate: int | None avg_frame_rate: str | None r_frame_rate: str | None time_base: str | None timing: FrameTimingSummary metadata_end: datetime | None metadata_begin_tag: datetime | None source_timezone: timezone | None computed_begin: datetime | None filename_begin: datetime | None gps_latitude: float | None gps_longitude: float | None warnings: list[str] def require_tool(name: str) -> str: return require_executable(name) def probe_video(ffprobe: str, path: Path, metadata: dict) -> VideoProbe: stream, format_info = _probe_stream(ffprobe, path) stream_duration = _to_float(stream.get("duration")) format_duration = _to_float(format_info.get("duration")) duration = stream_duration if stream_duration is not None else format_duration audio_streams = _probe_audio_streams(ffprobe, path, duration) audio_sample_rate = audio_streams[0].sample_rate if audio_streams else None timestamps, video_packet_bytes = _probe_video_packets(ffprobe, path) timing = summarize_frame_timing(timestamps) video_bit_rate = _to_int(stream.get("bit_rate")) or _average_bit_rate( video_packet_bytes, duration, ) end_value = first_tag(metadata, VIDEO_TIME_TAGS) metadata_end = parse_exif_datetime(end_value, assume_utc=True) metadata_begin_tag = parse_exif_datetime( first_tag(metadata, VIDEO_BEGIN_TIME_TAGS), assume_utc=True ) timezone_value = first_tag(metadata, VIDEO_TIMEZONE_TAGS) metadata_timezone = _video_timezone(metadata) source_timezone = metadata_timezone explicit_end = parse_exif_datetime(end_value) if source_timezone is None and explicit_end is not None: source_timezone = explicit_end.tzinfo computed_begin = None filename_begin = None if metadata_end is not None and duration is not None: computed_begin = metadata_end - timedelta(seconds=duration) filename_begin = metadata_end - timedelta(seconds=round_half_up(duration)) gps_latitude, gps_longitude = _gps_coordinates(metadata) warnings: list[str] = [] if metadata_end is None: warnings.append("No metadata end timestamp found.") if timezone_value is not None and metadata_timezone is None: warnings.append(f"Invalid source timezone metadata: {timezone_value}") if duration is None: warnings.append("No ffprobe duration found.") if metadata_begin_tag is not None: warnings.append( "Found a possible beginning timestamp tag; v1 still treats the " "normal video metadata timestamp as the end timestamp." ) if timing.timing_kind == "unknown": warnings.append("Not enough frame timestamps to classify CFR/VFR timing.") return VideoProbe( path=path, codec=_as_str(stream.get("codec_name")), width=_to_int(stream.get("width")), height=_to_int(stream.get("height")), pixel_format=_as_str(stream.get("pix_fmt")), bit_depth=_video_bit_depth(stream), video_bit_rate=video_bit_rate, color_space=_as_str(stream.get("color_space")), color_matrix=_avisynth_matrix_from_color_space(_as_str(stream.get("color_space"))), expected_color_matrix=_expected_color_matrix( _to_int(stream.get("width")), _to_int(stream.get("height")), ), duration_seconds=duration, frame_timestamps=timestamps, stream_frame_count=_to_int(stream.get("nb_frames")), audio_streams=audio_streams, audio_stream_count=len(audio_streams), audio_sample_rate=audio_sample_rate, avg_frame_rate=_as_str(stream.get("avg_frame_rate")), r_frame_rate=_as_str(stream.get("r_frame_rate")), time_base=_as_str(stream.get("time_base")), timing=timing, metadata_end=metadata_end, metadata_begin_tag=metadata_begin_tag, source_timezone=source_timezone, computed_begin=computed_begin, filename_begin=filename_begin, gps_latitude=gps_latitude, gps_longitude=gps_longitude, warnings=warnings, ) def _probe_stream(ffprobe: str, path: Path) -> tuple[dict, dict]: command = [ ffprobe, "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=codec_name,width,height,pix_fmt,bits_per_raw_sample,bits_per_sample,bit_rate,color_space,color_range,color_transfer,color_primaries,avg_frame_rate,r_frame_rate,time_base,duration,nb_frames", "-show_entries", "format=duration", "-of", "json", str(path), ] completed = subprocess.run(command, check=True, capture_output=True, text=True) data = json.loads(completed.stdout or "{}") streams = data.get("streams") or [] if not streams: raise RuntimeError(f"No video stream found in {path}") return streams[0], data.get("format") or {} def _probe_audio_streams( ffprobe: str, path: Path, fallback_duration: float | None, ) -> list[AudioStreamProbe]: command = [ ffprobe, "-v", "error", "-select_streams", "a", "-show_entries", "stream=index,codec_name,channels,sample_rate,bit_rate,duration", "-show_entries", "packet=stream_index,size", "-of", "json", str(path), ] completed = subprocess.run(command, check=True, capture_output=True, text=True) data = json.loads(completed.stdout or "{}") streams = data.get("streams") or [] packet_bytes: Counter[int] = Counter() for packet in data.get("packets") or []: stream_index = _to_int(packet.get("stream_index")) size = _to_int(packet.get("size")) if stream_index is not None and size is not None: packet_bytes[stream_index] += size return [ AudioStreamProbe( codec=_as_str(stream.get("codec_name")), channels=_to_int(stream.get("channels")), sample_rate=_to_int(stream.get("sample_rate")), bit_rate=( _to_int(stream.get("bit_rate")) or _average_bit_rate( packet_bytes[_to_int(stream.get("index")) or 0], _to_float(stream.get("duration")) or fallback_duration, ) ), ) for stream in streams ] def _probe_video_packets(ffprobe: str, path: Path) -> tuple[list[float], int]: command = [ ffprobe, "-v", "error", "-select_streams", "v:0", "-show_entries", "packet=pts_time,size", "-of", "json", str(path), ] completed = subprocess.run(command, check=True, capture_output=True, text=True) data = json.loads(completed.stdout or "{}") timestamps: list[float] = [] total_bytes = 0 for packet in data.get("packets") or []: value = _to_float(packet.get("pts_time")) if value is not None: timestamps.append(value) total_bytes += _to_int(packet.get("size")) or 0 return sorted(timestamps), total_bytes def _average_bit_rate(total_bytes: int, duration: float | None) -> int | None: if total_bytes <= 0 or duration is None or duration <= 0: return None return round(total_bytes * 8 / duration) def summarize_frame_timing(timestamps: list[float]) -> FrameTimingSummary: if len(timestamps) < 2: return FrameTimingSummary( frame_count=len(timestamps), first_pts=timestamps[0] if timestamps else None, last_pts=timestamps[-1] if timestamps else None, min_delta_ms=None, median_delta_ms=None, max_delta_ms=None, unique_delta_count=0, common_deltas_ms=[], timing_kind="unknown", ) deltas = [ (current - previous) * 1000 for previous, current in zip(timestamps, timestamps[1:]) if current >= previous ] if not deltas: timing_kind = "unknown" median_delta = None max_deviation = None else: median_delta = statistics.median(deltas) max_deviation = max(abs(delta - median_delta) for delta in deltas) timing_kind = "vfr" if max_deviation > 2.0 else "cfr" rounded = [round(delta, 3) for delta in deltas] counter = Counter(rounded) return FrameTimingSummary( frame_count=len(timestamps), first_pts=timestamps[0], last_pts=timestamps[-1], min_delta_ms=round(min(deltas), 3) if deltas else None, median_delta_ms=round(median_delta, 3) if median_delta is not None else None, max_delta_ms=round(max(deltas), 3) if deltas else None, unique_delta_count=len(counter), common_deltas_ms=counter.most_common(5), timing_kind=timing_kind, ) def first_tag(metadata: dict, tags: tuple[str, ...]) -> object | None: for tag in tags: if tag in metadata: return metadata[tag] return None def _gps_coordinates(metadata: dict) -> tuple[float | None, float | None]: latitude = _parse_gps_coordinate(first_tag(metadata, GPS_LATITUDE_TAGS)) longitude = _parse_gps_coordinate(first_tag(metadata, GPS_LONGITUDE_TAGS)) if latitude is not None and longitude is not None: return latitude, longitude position = first_tag(metadata, GPS_POSITION_TAGS) if position is None: return latitude, longitude pair = _parse_gps_pair(position) if pair == (None, None): return latitude, longitude return pair def _parse_gps_pair(value: object) -> tuple[float | None, float | None]: if isinstance(value, (list, tuple)) and len(value) >= 2: return _parse_gps_coordinate(value[0]), _parse_gps_coordinate(value[1]) text = str(value).strip() if not text: return None, None if "," in text: left, right = text.split(",", 1) return _parse_gps_coordinate(left), _parse_gps_coordinate(right) parts = text.split() if len(parts) >= 2: return _parse_gps_coordinate(parts[0]), _parse_gps_coordinate(parts[1]) return None, None def _parse_gps_coordinate(value: object) -> float | None: if value is None: return None if isinstance(value, (int, float)): return float(value) text = str(value).strip() if not text: return None try: return float(text) except ValueError: pass hemisphere = text[-1:].upper() sign = -1.0 if hemisphere in {"S", "W"} else 1.0 match = DMS_RE.search(text) if not match: return None degrees = float(match.group("deg")) minutes = float(match.group("min") or 0) seconds = float(match.group("sec") or 0) if degrees < 0: sign = -1.0 degrees = abs(degrees) return sign * (degrees + minutes / 60.0 + seconds / 3600.0) def parse_exif_datetime(value: object, assume_utc: bool = False) -> datetime | None: if value is None: return None match = DATETIME_RE.search(str(value).strip()) if not match: return None tzinfo = None tz_text = match.group("tz") if tz_text == "Z": tzinfo = timezone.utc elif tz_text: tzinfo = _parse_timezone_offset(tz_text) parsed = datetime( int(match.group("Y")), int(match.group("M")), int(match.group("D")), int(match.group("h")), int(match.group("m")), int(match.group("s")), tzinfo=tzinfo, ) if assume_utc and parsed.tzinfo is None: parsed = parsed.replace(tzinfo=timezone.utc) return parsed def _parse_timezone_offset(value: str) -> timezone: text = value.replace(":", "") sign = 1 if text[0] == "+" else -1 hours = int(text[1:3]) minutes = int(text[3:5]) return timezone(sign * timedelta(hours=hours, minutes=minutes)) def _video_timezone(metadata: dict) -> timezone | None: value = first_tag(metadata, VIDEO_TIMEZONE_TAGS) if value is None: return None text = str(value).strip() if not re.fullmatch(r"[+-]\d{2}:?\d{2}", text): return None return _parse_timezone_offset(text) def round_half_up(value: float) -> int: return int(value + 0.5) def _to_float(value: object) -> float | None: if value is None or value == "N/A": return None try: return float(value) except (TypeError, ValueError): return None def _to_int(value: object) -> int | None: if value is None or value == "N/A": return None try: return int(value) except (TypeError, ValueError): return None def _video_bit_depth(stream: dict) -> int | None: for key in ("bits_per_raw_sample", "bits_per_sample"): value = _to_int(stream.get(key)) if value: return value pix_fmt = _as_str(stream.get("pix_fmt")) or "" match = re.search(r"p(?P10|12|14|16)(?:le|be)?$", pix_fmt) if match: return int(match.group("bits")) if pix_fmt: return 8 return None def _avisynth_matrix_from_color_space(color_space: str | None) -> str | None: if not color_space: return None normalized = color_space.lower() if normalized in {"bt709", "bt709nc"}: return "Rec709" if normalized in {"smpte170m", "bt470bg", "fcc"}: return "Rec601" if normalized in {"bt2020nc", "bt2020ncl", "bt2020c", "bt2020cl"}: return "Rec2020" return None def _expected_color_matrix(width: int | None, height: int | None) -> str | None: if width is None or height is None: return None if width >= 3840 or height >= 2160: return "Rec2020" if width >= 1280 or height >= 720: return "Rec709" return "Rec601" def _as_str(value: object) -> str | None: if value is None or value == "N/A": return None return str(value)