Files
scripts/bin/nn
T
2026-09-25 08:44:15 +02:00

446 lines
13 KiB
Python
Executable File

#!/usr/bin/env python3
"""Normalize photo and image filenames from phones and other sources."""
import argparse
import re
import shutil
import subprocess
import sys
from datetime import datetime
from functools import lru_cache
from pathlib import Path
PHONE_PATTERN = re.compile(
r'^[A-Za-z]+[_-]'
r'(\d{4})-?(\d{2})-?(\d{2})'
r'[_-]'
r'(.+)$'
)
# Telegram photo/video export, e.g. "photo_94@16-07-2026_22-09-43.jpg"
# (photo|video)_<n>@<DD>-<MM>-<YYYY>_<HH>-<MM>-<SS>
TELEGRAM_PATTERN = re.compile(
r'^(?:photo|video)_(\d+)@(\d{2})-(\d{2})-(\d{4})_(\d{2})-(\d{2})-(\d{2})$'
)
NORMALIZED = re.compile(r'^\d{4}-\d{2}-\d{2} .+')
# Filename prefixes whose embedded timestamp is UTC rather than local time.
# The Pixel camera names captures in UTC but records the local wall clock time
# in EXIF, so for these the metadata is authoritative and the filename is not.
# Other sources (Samsung's IMG_/VID_, Telegram exports) already name in local
# time and are left alone.
UTC_FILENAME_PREFIXES = ('PXL',)
# e.g. "PXL_20260917_163001280.MP" -> prefix, timestamp, subseconds, suffix
UTC_STEM_PATTERN = re.compile(
r'^(' + '|'.join(UTC_FILENAME_PREFIXES) + r')[_-]'
r'(\d{8}[_-]\d{6})'
r'(\d*)'
r'(.*)$',
re.IGNORECASE,
)
# Windows hidden-file attribute, mirrored from the stat module.
FILE_ATTRIBUTE_HIDDEN = 0x2
STATUS_RENAME = 'rename'
STATUS_ALREADY = 'skip (already normalized)'
STATUS_NO_MATCH = 'skip (no rule matched)'
STATUS_CONFLICT = 'skip (conflict)'
@lru_cache(maxsize=1)
def has_exiftool():
"""Check once whether exiftool is available on PATH."""
return shutil.which('exiftool') is not None
def get_exif_date(filepath):
"""Try to get the image creation date via exiftool."""
try:
result = subprocess.run(
['exiftool', '-s3', '-d', '%Y-%m-%d', '-DateTimeOriginal', str(filepath)],
capture_output=True,
text=True,
timeout=10,
)
date_str = result.stdout.strip()
if date_str and re.match(r'^\d{4}-\d{2}-\d{2}$', date_str):
return date_str
except (FileNotFoundError, subprocess.TimeoutExpired):
pass
return None
def get_exif_timestamp(filepath):
"""Get the local capture time via exiftool as (YYYYMMDD_HHMMSS, subsec).
DateTimeOriginal is the wall clock time at the moment of capture, so it
needs no timezone correction. Returns None when unavailable.
"""
try:
result = subprocess.run(
[
'exiftool', '-s3', '-d', '%Y%m%d_%H%M%S',
'-DateTimeOriginal', '-SubSecTimeOriginal', str(filepath),
],
capture_output=True,
text=True,
timeout=10,
)
except (FileNotFoundError, subprocess.TimeoutExpired):
return None
lines = result.stdout.strip().splitlines()
if not lines or not re.match(r'^\d{8}_\d{6}$', lines[0].strip()):
return None
subsec = lines[1].strip() if len(lines) > 1 else ''
return lines[0].strip(), subsec if subsec.isdigit() else ''
def resolve_stem(filepath):
"""Return the file's stem, with UTC-named timestamps corrected from EXIF.
Pixel filenames encode UTC, so the name alone is off by the UTC offset in
effect at capture. Falls back to the original stem when the file is not a
UTC-named capture, or when its metadata has no usable timestamp.
"""
stem = filepath.stem
match = UTC_STEM_PATTERN.match(stem)
if not match or not has_exiftool():
return stem
prefix, _, subsec, rest = match.groups()
stamp = get_exif_timestamp(filepath)
if stamp is None:
return stem
exif_stamp, exif_subsec = stamp
return f'{prefix}_{exif_stamp}{exif_subsec or subsec}{rest}'
def get_file_date(filepath):
"""Get file creation date (or modification time as fallback)."""
stat = filepath.stat()
ts = getattr(stat, 'st_birthtime', None) or stat.st_mtime
return datetime.fromtimestamp(ts).strftime('%Y-%m-%d')
def plan_default_rename(filepath):
"""Plan a rename using the default pattern-matching mode."""
if NORMALIZED.match(filepath.name):
return None, STATUS_ALREADY
stem = resolve_stem(filepath)
ext = filepath.suffix
tg_match = TELEGRAM_PATTERN.match(stem)
if tg_match:
number, day, month, year, hour, minute, second = tg_match.groups()
return f'{year}-{month}-{day} {hour}{minute}{second} {number}{ext}', STATUS_RENAME
match = PHONE_PATTERN.match(stem)
if not match:
match = PHONE_PATTERN.match(f'{stem}{ext}')
if match:
year, month, day, rest = match.groups()
new_name = f'{year}-{month}-{day} {rest}'
else:
return None, STATUS_NO_MATCH
else:
year, month, day, rest = match.groups()
new_name = f'{year}-{month}-{day} {rest}{ext}'
return new_name, STATUS_RENAME
def plan_parse_rename(filepath):
"""Plan a rename using EXIF or file date."""
if NORMALIZED.match(filepath.name):
return None, STATUS_ALREADY
date_str = None
if has_exiftool():
date_str = get_exif_date(filepath)
if not date_str:
date_str = get_file_date(filepath)
stem = filepath.stem
ext = filepath.suffix
if stem.startswith(date_str):
new_name = f'{date_str} {stem[len(date_str):].lstrip(" -_")}{ext}'
else:
new_name = f'{date_str} {stem}{ext}'
if new_name == f'{date_str} {ext}':
new_name = f'{date_str}{ext}'
if new_name == filepath.name:
return None, STATUS_ALREADY
return new_name, STATUS_RENAME
def get_date_for_file(filepath, parse_mode):
"""Resolve a date string for a file, or None if unavailable."""
if parse_mode:
date_str = None
if has_exiftool():
date_str = get_exif_date(filepath)
if not date_str:
date_str = get_file_date(filepath)
return date_str
# Default pattern mode: extract date from filename
stem = resolve_stem(filepath)
# Check if already normalized (e.g. "2024-06-01 breakfast.jpg")
norm_match = re.match(r'^(\d{4})-(\d{2})-(\d{2}) ', filepath.name)
if norm_match:
return f'{norm_match.group(1)}-{norm_match.group(2)}-{norm_match.group(3)}'
# Telegram export pattern (date is DD-MM-YYYY)
tg_match = TELEGRAM_PATTERN.match(stem)
if tg_match:
_, day, month, year = tg_match.groups()[:4]
return f'{year}-{month}-{day}'
# Try the standard pattern
match = PHONE_PATTERN.match(stem)
if not match:
match = PHONE_PATTERN.match(f'{stem}{filepath.suffix}')
if match:
year, month, day, _ = match.groups()
return f'{year}-{month}-{day}'
return None
def is_hidden(filepath):
"""Check whether a file is hidden (dotfile, or hidden attribute on Windows)."""
if filepath.name.startswith('.'):
return True
try:
attrs = filepath.stat().st_file_attributes
except (AttributeError, OSError):
return False
return bool(attrs & FILE_ATTRIBUTE_HIDDEN)
def collect_files(paths, include_hidden=False):
"""Resolve the list of files to process."""
if not paths:
paths = ['.']
files = []
for path in paths:
candidate = Path(path)
if candidate.is_dir():
files.extend(sorted(
file for file in candidate.iterdir()
if file.is_file() and (include_hidden or not is_hidden(file))
))
elif candidate.is_file():
# Explicitly named files are always processed.
files.append(candidate)
return files
def build_plan(files, parse_mode):
"""Build a list of (original_path, new_name, status) tuples."""
plan = []
seen_targets = {}
for file_path in files:
if parse_mode:
new_name, status = plan_parse_rename(file_path)
else:
new_name, status = plan_default_rename(file_path)
if status == STATUS_RENAME and new_name:
target = file_path.parent / new_name
target_key = str(target).casefold()
if target.exists() and target.resolve() != file_path.resolve():
status = STATUS_CONFLICT
elif target_key in seen_targets:
prev_idx = seen_targets[target_key]
plan[prev_idx] = (plan[prev_idx][0], plan[prev_idx][1], STATUS_CONFLICT)
status = STATUS_CONFLICT
else:
seen_targets[target_key] = len(plan)
plan.append((file_path, new_name, status))
return plan
def build_number_plan(files, parse_mode, force=False):
"""Build a rename plan using sequential numbering per day."""
entries = []
for file_path in files:
if not force and NORMALIZED.match(file_path.name):
entries.append((file_path, None, STATUS_ALREADY))
continue
date_str = get_date_for_file(file_path, parse_mode)
entries.append((file_path, date_str))
day_counter = {}
plan = []
seen_targets = {}
for entry in entries:
if len(entry) == 3:
plan.append((entry[0], None, entry[2]))
continue
file_path, date_str = entry
if date_str is None:
plan.append((file_path, None, STATUS_NO_MATCH))
continue
day_counter[date_str] = day_counter.get(date_str, 0) + 1
seq = day_counter[date_str]
ext = file_path.suffix
new_name = f'{date_str} {seq:02d}{ext}'
if new_name == file_path.name:
plan.append((file_path, None, STATUS_ALREADY))
continue
status = STATUS_RENAME
target = file_path.parent / new_name
target_key = str(target).casefold()
if target.exists() and target.resolve() != file_path.resolve():
status = STATUS_CONFLICT
elif target_key in seen_targets:
prev_idx = seen_targets[target_key]
plan[prev_idx] = (plan[prev_idx][0], plan[prev_idx][1], STATUS_CONFLICT)
status = STATUS_CONFLICT
else:
seen_targets[target_key] = len(plan)
plan.append((file_path, new_name, status))
return plan
def print_preview(plan):
"""Print a formatted preview table."""
if not plan:
print('No files found.')
return
col1 = max(len(file_path.name) for file_path, _, _ in plan)
col2 = max(len(new_name or '') for _, new_name, _ in plan)
col1 = max(col1, len('Original'))
col2 = max(col2, len('New Name'))
header = f'{"Original":<{col1}} {"New Name":<{col2}} Status'
print(header)
print('-' * len(header))
for file_path, new_name, status in plan:
print(f'{file_path.name:<{col1}} {(new_name or ""):<{col2}} {status}')
print()
def apply_renames(plan):
"""Execute the planned renames."""
count = 0
for file_path, new_name, status in plan:
if status != STATUS_RENAME:
continue
file_path.rename(file_path.parent / new_name)
count += 1
return count
def confirm(prompt='Apply renames? [Y/n] '):
"""Ask the user for confirmation. An empty answer accepts the default (yes)."""
while True:
try:
answer = input(prompt).strip().lower()
except EOFError:
print()
return True
if answer in ('', 'y', 'yes'):
return True
if answer in ('n', 'no'):
return False
def main():
parser = argparse.ArgumentParser(description='Normalize photo and image filenames.')
parser.add_argument(
'-p',
'--parse',
action='store_true',
help='Use EXIF data or file dates instead of pattern matching.',
)
parser.add_argument(
'-n',
'--number',
action='store_true',
help=(
'Discard original filename stems and number files sequentially '
'per day. Combines with --parse to select the date source. '
'Skips already-normalized files unless --force is given.'
),
)
parser.add_argument(
'-f',
'--force',
action='store_true',
help='Force numbering even for already-normalized files (used with -n).',
)
parser.add_argument(
'-a',
'--all',
action='store_true',
help='Include hidden files when scanning directories.',
)
parser.add_argument(
'files',
nargs='*',
help='Files or directories to process. Defaults to current directory.',
)
args = parser.parse_args()
files = collect_files(args.files, args.all)
if not files:
print('No files found.')
sys.exit(0)
if args.number:
plan = build_number_plan(files, args.parse, args.force)
else:
plan = build_plan(files, args.parse)
print_preview(plan)
renames = sum(1 for _, _, status in plan if status == STATUS_RENAME)
if renames == 0:
print('Nothing to rename.')
sys.exit(0)
if confirm():
print(f'Renamed {apply_renames(plan)} file(s).')
else:
print('Aborted.')
if __name__ == '__main__':
try:
main()
except KeyboardInterrupt:
print('\nAborted.')
sys.exit(1)