480 lines
14 KiB
Python
Executable File
480 lines
14 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Normalize photo and image filenames from phones and other sources."""
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import sys
|
|
from datetime import datetime
|
|
from functools import lru_cache
|
|
from pathlib import Path
|
|
|
|
PHONE_PATTERN = re.compile(
|
|
r'^[A-Za-z]+[_-]'
|
|
r'(\d{4})-?(\d{2})-?(\d{2})'
|
|
r'[_-]'
|
|
r'(.+)$'
|
|
)
|
|
|
|
# Telegram photo/video export, e.g. "photo_94@16-07-2026_22-09-43.jpg"
|
|
# (photo|video)_<n>@<DD>-<MM>-<YYYY>_<HH>-<MM>-<SS>
|
|
TELEGRAM_PATTERN = re.compile(
|
|
r'^(?:photo|video)_(\d+)@(\d{2})-(\d{2})-(\d{4})_(\d{2})-(\d{2})-(\d{2})$'
|
|
)
|
|
|
|
NORMALIZED = re.compile(r'^\d{4}-\d{2}-\d{2} .+')
|
|
|
|
# Filename prefixes whose embedded timestamp is UTC rather than local time.
|
|
# The Pixel camera names captures in UTC but records the local wall clock time
|
|
# in EXIF, so for these the metadata is authoritative and the filename is not.
|
|
# Other sources (Samsung's IMG_/VID_, Telegram exports) already name in local
|
|
# time and are left alone.
|
|
UTC_FILENAME_PREFIXES = ('PXL',)
|
|
|
|
# e.g. "PXL_20260917_163001280.MP" -> prefix, timestamp, subseconds, suffix
|
|
UTC_STEM_PATTERN = re.compile(
|
|
r'^(' + '|'.join(UTC_FILENAME_PREFIXES) + r')[_-]'
|
|
r'(\d{8}[_-]\d{6})'
|
|
r'(\d*)'
|
|
r'(.*)$',
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
# Windows hidden-file attribute, mirrored from the stat module.
|
|
FILE_ATTRIBUTE_HIDDEN = 0x2
|
|
|
|
STATUS_RENAME = 'rename'
|
|
STATUS_ALREADY = 'skip (already normalized)'
|
|
STATUS_NO_MATCH = 'skip (no rule matched)'
|
|
STATUS_CONFLICT = 'skip (conflict)'
|
|
|
|
|
|
@lru_cache(maxsize=1)
|
|
def has_exiftool():
|
|
"""Check once whether exiftool is available on PATH."""
|
|
return shutil.which('exiftool') is not None
|
|
|
|
|
|
# Capture timestamps keyed by exif_key(path), filled by prefetch_exif().
|
|
_exif_cache = {}
|
|
|
|
|
|
def exif_key(filepath):
|
|
"""Normalize a path so exiftool's SourceFile and our Path agree."""
|
|
return os.path.normcase(os.path.abspath(filepath))
|
|
|
|
|
|
def prefetch_exif(files):
|
|
"""Read capture timestamps for all files with a single exiftool call.
|
|
|
|
Starting exiftool dominates the cost of reading a few tags, so reading
|
|
every file in one process is far faster than one process per file.
|
|
Paths are passed on stdin to stay clear of command line length limits.
|
|
"""
|
|
files = [f for f in files if exif_key(f) not in _exif_cache]
|
|
if not files or not has_exiftool():
|
|
return
|
|
|
|
# Files without a usable timestamp stay None so they aren't re-queried.
|
|
_exif_cache.update((exif_key(f), None) for f in files)
|
|
|
|
args = [
|
|
'-json', '-fast', '-charset', 'filename=utf8',
|
|
'-d', '%Y%m%d_%H%M%S',
|
|
'-DateTimeOriginal', '-SubSecTimeOriginal',
|
|
*(str(f) for f in files),
|
|
]
|
|
try:
|
|
result = subprocess.run(
|
|
['exiftool', '-@', '-'],
|
|
input='\n'.join(args),
|
|
capture_output=True,
|
|
encoding='utf-8',
|
|
timeout=60 + len(files),
|
|
)
|
|
records = json.loads(result.stdout or '[]')
|
|
except (FileNotFoundError, subprocess.TimeoutExpired, json.JSONDecodeError):
|
|
return
|
|
|
|
for record in records:
|
|
stamp = str(record.get('DateTimeOriginal', ''))
|
|
if not re.match(r'^\d{8}_\d{6}$', stamp):
|
|
continue
|
|
subsec = str(record.get('SubSecTimeOriginal', ''))
|
|
_exif_cache[exif_key(record['SourceFile'])] = (
|
|
stamp, subsec if subsec.isdigit() else '',
|
|
)
|
|
|
|
|
|
def get_exif_date(filepath):
|
|
"""Get the image creation date (YYYY-MM-DD) from EXIF, or None."""
|
|
stamp = get_exif_timestamp(filepath)
|
|
if stamp is None:
|
|
return None
|
|
day = stamp[0]
|
|
return f'{day[:4]}-{day[4:6]}-{day[6:8]}'
|
|
|
|
|
|
def get_exif_timestamp(filepath):
|
|
"""Get the local capture time from EXIF as (YYYYMMDD_HHMMSS, subsec).
|
|
|
|
DateTimeOriginal is the wall clock time at the moment of capture, so it
|
|
needs no timezone correction. Returns None when unavailable.
|
|
"""
|
|
prefetch_exif([filepath])
|
|
return _exif_cache.get(exif_key(filepath))
|
|
|
|
|
|
def resolve_stem(filepath):
|
|
"""Return the file's stem, with UTC-named timestamps corrected from EXIF.
|
|
|
|
Pixel filenames encode UTC, so the name alone is off by the UTC offset in
|
|
effect at capture. Falls back to the original stem when the file is not a
|
|
UTC-named capture, or when its metadata has no usable timestamp.
|
|
"""
|
|
stem = filepath.stem
|
|
|
|
match = UTC_STEM_PATTERN.match(stem)
|
|
if not match or not has_exiftool():
|
|
return stem
|
|
|
|
prefix, _, subsec, rest = match.groups()
|
|
stamp = get_exif_timestamp(filepath)
|
|
if stamp is None:
|
|
return stem
|
|
|
|
exif_stamp, exif_subsec = stamp
|
|
return f'{prefix}_{exif_stamp}{exif_subsec or subsec}{rest}'
|
|
|
|
|
|
def get_file_date(filepath):
|
|
"""Get file creation date (or modification time as fallback)."""
|
|
stat = filepath.stat()
|
|
ts = getattr(stat, 'st_birthtime', None) or stat.st_mtime
|
|
return datetime.fromtimestamp(ts).strftime('%Y-%m-%d')
|
|
|
|
|
|
def plan_default_rename(filepath):
|
|
"""Plan a rename using the default pattern-matching mode."""
|
|
if NORMALIZED.match(filepath.name):
|
|
return None, STATUS_ALREADY
|
|
|
|
stem = resolve_stem(filepath)
|
|
ext = filepath.suffix
|
|
|
|
tg_match = TELEGRAM_PATTERN.match(stem)
|
|
if tg_match:
|
|
number, day, month, year, hour, minute, second = tg_match.groups()
|
|
return f'{year}-{month}-{day} {hour}{minute}{second} {number}{ext}', STATUS_RENAME
|
|
|
|
match = PHONE_PATTERN.match(stem)
|
|
if not match:
|
|
match = PHONE_PATTERN.match(f'{stem}{ext}')
|
|
if match:
|
|
year, month, day, rest = match.groups()
|
|
new_name = f'{year}-{month}-{day} {rest}'
|
|
else:
|
|
return None, STATUS_NO_MATCH
|
|
else:
|
|
year, month, day, rest = match.groups()
|
|
new_name = f'{year}-{month}-{day} {rest}{ext}'
|
|
|
|
return new_name, STATUS_RENAME
|
|
|
|
|
|
def plan_parse_rename(filepath):
|
|
"""Plan a rename using EXIF or file date."""
|
|
if NORMALIZED.match(filepath.name):
|
|
return None, STATUS_ALREADY
|
|
|
|
date_str = None
|
|
if has_exiftool():
|
|
date_str = get_exif_date(filepath)
|
|
|
|
if not date_str:
|
|
date_str = get_file_date(filepath)
|
|
|
|
stem = filepath.stem
|
|
ext = filepath.suffix
|
|
|
|
if stem.startswith(date_str):
|
|
new_name = f'{date_str} {stem[len(date_str):].lstrip(" -_")}{ext}'
|
|
else:
|
|
new_name = f'{date_str} {stem}{ext}'
|
|
|
|
if new_name == f'{date_str} {ext}':
|
|
new_name = f'{date_str}{ext}'
|
|
|
|
if new_name == filepath.name:
|
|
return None, STATUS_ALREADY
|
|
|
|
return new_name, STATUS_RENAME
|
|
|
|
|
|
def get_date_for_file(filepath, parse_mode):
|
|
"""Resolve a date string for a file, or None if unavailable."""
|
|
if parse_mode:
|
|
date_str = None
|
|
if has_exiftool():
|
|
date_str = get_exif_date(filepath)
|
|
if not date_str:
|
|
date_str = get_file_date(filepath)
|
|
return date_str
|
|
|
|
# Default pattern mode: extract date from filename
|
|
stem = resolve_stem(filepath)
|
|
|
|
# Check if already normalized (e.g. "2024-06-01 breakfast.jpg")
|
|
norm_match = re.match(r'^(\d{4})-(\d{2})-(\d{2}) ', filepath.name)
|
|
if norm_match:
|
|
return f'{norm_match.group(1)}-{norm_match.group(2)}-{norm_match.group(3)}'
|
|
|
|
# Telegram export pattern (date is DD-MM-YYYY)
|
|
tg_match = TELEGRAM_PATTERN.match(stem)
|
|
if tg_match:
|
|
_, day, month, year = tg_match.groups()[:4]
|
|
return f'{year}-{month}-{day}'
|
|
|
|
# Try the standard pattern
|
|
match = PHONE_PATTERN.match(stem)
|
|
if not match:
|
|
match = PHONE_PATTERN.match(f'{stem}{filepath.suffix}')
|
|
if match:
|
|
year, month, day, _ = match.groups()
|
|
return f'{year}-{month}-{day}'
|
|
|
|
return None
|
|
|
|
|
|
def is_hidden(filepath):
|
|
"""Check whether a file is hidden (dotfile, or hidden attribute on Windows)."""
|
|
if filepath.name.startswith('.'):
|
|
return True
|
|
|
|
try:
|
|
attrs = filepath.stat().st_file_attributes
|
|
except (AttributeError, OSError):
|
|
return False
|
|
return bool(attrs & FILE_ATTRIBUTE_HIDDEN)
|
|
|
|
|
|
def collect_files(paths, include_hidden=False):
|
|
"""Resolve the list of files to process."""
|
|
if not paths:
|
|
paths = ['.']
|
|
|
|
files = []
|
|
for path in paths:
|
|
candidate = Path(path)
|
|
if candidate.is_dir():
|
|
files.extend(sorted(
|
|
file for file in candidate.iterdir()
|
|
if file.is_file() and (include_hidden or not is_hidden(file))
|
|
))
|
|
elif candidate.is_file():
|
|
# Explicitly named files are always processed.
|
|
files.append(candidate)
|
|
return files
|
|
|
|
|
|
def build_plan(files, parse_mode):
|
|
"""Build a list of (original_path, new_name, status) tuples."""
|
|
plan = []
|
|
seen_targets = {}
|
|
|
|
for file_path in files:
|
|
if parse_mode:
|
|
new_name, status = plan_parse_rename(file_path)
|
|
else:
|
|
new_name, status = plan_default_rename(file_path)
|
|
|
|
if status == STATUS_RENAME and new_name:
|
|
target = file_path.parent / new_name
|
|
target_key = str(target).casefold()
|
|
|
|
if target.exists() and target.resolve() != file_path.resolve():
|
|
status = STATUS_CONFLICT
|
|
elif target_key in seen_targets:
|
|
prev_idx = seen_targets[target_key]
|
|
plan[prev_idx] = (plan[prev_idx][0], plan[prev_idx][1], STATUS_CONFLICT)
|
|
status = STATUS_CONFLICT
|
|
else:
|
|
seen_targets[target_key] = len(plan)
|
|
|
|
plan.append((file_path, new_name, status))
|
|
|
|
return plan
|
|
|
|
|
|
def build_number_plan(files, parse_mode, force=False):
|
|
"""Build a rename plan using sequential numbering per day."""
|
|
entries = []
|
|
for file_path in files:
|
|
if not force and NORMALIZED.match(file_path.name):
|
|
entries.append((file_path, None, STATUS_ALREADY))
|
|
continue
|
|
date_str = get_date_for_file(file_path, parse_mode)
|
|
entries.append((file_path, date_str))
|
|
|
|
day_counter = {}
|
|
plan = []
|
|
seen_targets = {}
|
|
|
|
for entry in entries:
|
|
if len(entry) == 3:
|
|
plan.append((entry[0], None, entry[2]))
|
|
continue
|
|
file_path, date_str = entry
|
|
if date_str is None:
|
|
plan.append((file_path, None, STATUS_NO_MATCH))
|
|
continue
|
|
|
|
day_counter[date_str] = day_counter.get(date_str, 0) + 1
|
|
seq = day_counter[date_str]
|
|
ext = file_path.suffix
|
|
new_name = f'{date_str} {seq:02d}{ext}'
|
|
|
|
if new_name == file_path.name:
|
|
plan.append((file_path, None, STATUS_ALREADY))
|
|
continue
|
|
|
|
status = STATUS_RENAME
|
|
target = file_path.parent / new_name
|
|
target_key = str(target).casefold()
|
|
|
|
if target.exists() and target.resolve() != file_path.resolve():
|
|
status = STATUS_CONFLICT
|
|
elif target_key in seen_targets:
|
|
prev_idx = seen_targets[target_key]
|
|
plan[prev_idx] = (plan[prev_idx][0], plan[prev_idx][1], STATUS_CONFLICT)
|
|
status = STATUS_CONFLICT
|
|
else:
|
|
seen_targets[target_key] = len(plan)
|
|
|
|
plan.append((file_path, new_name, status))
|
|
|
|
return plan
|
|
|
|
|
|
def print_preview(plan):
|
|
"""Print a formatted preview table."""
|
|
if not plan:
|
|
print('No files found.')
|
|
return
|
|
|
|
col1 = max(len(file_path.name) for file_path, _, _ in plan)
|
|
col2 = max(len(new_name or '') for _, new_name, _ in plan)
|
|
col1 = max(col1, len('Original'))
|
|
col2 = max(col2, len('New Name'))
|
|
|
|
header = f'{"Original":<{col1}} {"New Name":<{col2}} Status'
|
|
print(header)
|
|
print('-' * len(header))
|
|
|
|
for file_path, new_name, status in plan:
|
|
print(f'{file_path.name:<{col1}} {(new_name or ""):<{col2}} {status}')
|
|
|
|
print()
|
|
|
|
|
|
def apply_renames(plan):
|
|
"""Execute the planned renames."""
|
|
count = 0
|
|
for file_path, new_name, status in plan:
|
|
if status != STATUS_RENAME:
|
|
continue
|
|
file_path.rename(file_path.parent / new_name)
|
|
count += 1
|
|
return count
|
|
|
|
|
|
def confirm(prompt='Apply renames? [Y/n] '):
|
|
"""Ask the user for confirmation. An empty answer accepts the default (yes)."""
|
|
while True:
|
|
try:
|
|
answer = input(prompt).strip().lower()
|
|
except EOFError:
|
|
print()
|
|
return True
|
|
if answer in ('', 'y', 'yes'):
|
|
return True
|
|
if answer in ('n', 'no'):
|
|
return False
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(description='Normalize photo and image filenames.')
|
|
parser.add_argument(
|
|
'-p',
|
|
'--parse',
|
|
action='store_true',
|
|
help='Use EXIF data or file dates instead of pattern matching.',
|
|
)
|
|
parser.add_argument(
|
|
'-n',
|
|
'--number',
|
|
action='store_true',
|
|
help=(
|
|
'Discard original filename stems and number files sequentially '
|
|
'per day. Combines with --parse to select the date source. '
|
|
'Skips already-normalized files unless --force is given.'
|
|
),
|
|
)
|
|
parser.add_argument(
|
|
'-f',
|
|
'--force',
|
|
action='store_true',
|
|
help='Force numbering even for already-normalized files (used with -n).',
|
|
)
|
|
parser.add_argument(
|
|
'-a',
|
|
'--all',
|
|
action='store_true',
|
|
help='Include hidden files when scanning directories.',
|
|
)
|
|
parser.add_argument(
|
|
'files',
|
|
nargs='*',
|
|
help='Files or directories to process. Defaults to current directory.',
|
|
)
|
|
|
|
args = parser.parse_args()
|
|
|
|
files = collect_files(args.files, args.all)
|
|
if not files:
|
|
print('No files found.')
|
|
sys.exit(0)
|
|
|
|
prefetch_exif([
|
|
f for f in files
|
|
if (args.force or not NORMALIZED.match(f.name))
|
|
and (args.parse or UTC_STEM_PATTERN.match(f.stem))
|
|
])
|
|
|
|
if args.number:
|
|
plan = build_number_plan(files, args.parse, args.force)
|
|
else:
|
|
plan = build_plan(files, args.parse)
|
|
print_preview(plan)
|
|
|
|
renames = sum(1 for _, _, status in plan if status == STATUS_RENAME)
|
|
if renames == 0:
|
|
print('Nothing to rename.')
|
|
sys.exit(0)
|
|
|
|
if confirm():
|
|
print(f'Renamed {apply_renames(plan)} file(s).')
|
|
else:
|
|
print('Aborted.')
|
|
|
|
|
|
if __name__ == '__main__':
|
|
try:
|
|
main()
|
|
except KeyboardInterrupt:
|
|
print('\nAborted.')
|
|
sys.exit(1)
|