Files
scripts/bin/nn
T
luxick a8b1d41398 Update nn
Exclude hidden files per default. Use `-a` to include them
2026-08-24 10:54:58 +02:00

382 lines
11 KiB
Python
Executable File

#!/usr/bin/env python3
"""Normalize photo and image filenames from phones and other sources."""
import argparse
import re
import shutil
import subprocess
import sys
import termios
import tty
from datetime import datetime
from pathlib import Path
PHONE_PATTERN = re.compile(
r'^[A-Za-z]+[_-]'
r'(\d{4})-?(\d{2})-?(\d{2})'
r'[_-]'
r'(.+)$'
)
# Telegram photo/video export, e.g. "photo_94@16-07-2026_22-09-43.jpg"
# (photo|video)_<n>@<DD>-<MM>-<YYYY>_<HH>-<MM>-<SS>
TELEGRAM_PATTERN = re.compile(
r'^(?:photo|video)_(\d+)@(\d{2})-(\d{2})-(\d{4})_(\d{2})-(\d{2})-(\d{2})$'
)
NORMALIZED = re.compile(r'^\d{4}-\d{2}-\d{2} .+')
# Windows hidden-file attribute, mirrored from the stat module.
FILE_ATTRIBUTE_HIDDEN = 0x2
STATUS_RENAME = 'rename'
STATUS_ALREADY = 'skip (already normalized)'
STATUS_NO_MATCH = 'skip (no rule matched)'
STATUS_CONFLICT = 'skip (conflict)'
def get_exif_date(filepath):
"""Try to get the image creation date via exiftool."""
try:
result = subprocess.run(
['exiftool', '-s3', '-d', '%Y-%m-%d', '-DateTimeOriginal', str(filepath)],
capture_output=True,
text=True,
timeout=10,
)
date_str = result.stdout.strip()
if date_str and re.match(r'^\d{4}-\d{2}-\d{2}$', date_str):
return date_str
except (FileNotFoundError, subprocess.TimeoutExpired):
pass
return None
def get_file_date(filepath):
"""Get file creation date (or modification time as fallback)."""
stat = filepath.stat()
ts = getattr(stat, 'st_birthtime', None) or stat.st_mtime
return datetime.fromtimestamp(ts).strftime('%Y-%m-%d')
def plan_default_rename(filepath):
"""Plan a rename using the default pattern-matching mode."""
stem = filepath.stem
ext = filepath.suffix
if NORMALIZED.match(filepath.name):
return None, STATUS_ALREADY
tg_match = TELEGRAM_PATTERN.match(stem)
if tg_match:
number, day, month, year, hour, minute, second = tg_match.groups()
return f'{year}-{month}-{day} {hour}{minute}{second} {number}{ext}', STATUS_RENAME
match = PHONE_PATTERN.match(stem)
if not match:
match = PHONE_PATTERN.match(filepath.name)
if match:
year, month, day, rest = match.groups()
new_name = f'{year}-{month}-{day} {rest}'
else:
return None, STATUS_NO_MATCH
else:
year, month, day, rest = match.groups()
new_name = f'{year}-{month}-{day} {rest}{ext}'
return new_name, STATUS_RENAME
def plan_parse_rename(filepath):
"""Plan a rename using EXIF or file date."""
if NORMALIZED.match(filepath.name):
return None, STATUS_ALREADY
date_str = None
if shutil.which('exiftool') is not None:
date_str = get_exif_date(filepath)
if not date_str:
date_str = get_file_date(filepath)
stem = filepath.stem
ext = filepath.suffix
if stem.startswith(date_str):
new_name = f'{date_str} {stem[len(date_str):].lstrip(" -_")}{ext}'
else:
new_name = f'{date_str} {stem}{ext}'
if new_name == f'{date_str} {ext}':
new_name = f'{date_str}{ext}'
if new_name == filepath.name:
return None, STATUS_ALREADY
return new_name, STATUS_RENAME
def get_date_for_file(filepath, parse_mode):
"""Resolve a date string for a file, or None if unavailable."""
if parse_mode:
date_str = None
if shutil.which('exiftool') is not None:
date_str = get_exif_date(filepath)
if not date_str:
date_str = get_file_date(filepath)
return date_str
# Default pattern mode: extract date from filename
stem = filepath.stem
# Check if already normalized (e.g. "2024-06-01 breakfast.jpg")
norm_match = re.match(r'^(\d{4})-(\d{2})-(\d{2}) ', filepath.name)
if norm_match:
return f'{norm_match.group(1)}-{norm_match.group(2)}-{norm_match.group(3)}'
# Telegram export pattern (date is DD-MM-YYYY)
tg_match = TELEGRAM_PATTERN.match(stem)
if tg_match:
_, day, month, year = tg_match.groups()[:4]
return f'{year}-{month}-{day}'
# Try the standard pattern
match = PHONE_PATTERN.match(stem)
if not match:
match = PHONE_PATTERN.match(filepath.name)
if match:
year, month, day, _ = match.groups()
return f'{year}-{month}-{day}'
return None
def is_hidden(filepath):
"""Check whether a file is hidden (dotfile, or hidden attribute on Windows)."""
if filepath.name.startswith('.'):
return True
try:
attrs = filepath.stat().st_file_attributes
except (AttributeError, OSError):
return False
return bool(attrs & FILE_ATTRIBUTE_HIDDEN)
def collect_files(paths, include_hidden=False):
"""Resolve the list of files to process."""
if not paths:
paths = ['.']
files = []
for path in paths:
candidate = Path(path)
if candidate.is_dir():
files.extend(sorted(
file for file in candidate.iterdir()
if file.is_file() and (include_hidden or not is_hidden(file))
))
elif candidate.is_file():
# Explicitly named files are always processed.
files.append(candidate)
return files
def build_plan(files, parse_mode):
"""Build a list of (original_path, new_name, status) tuples."""
plan = []
seen_targets = {}
for file_path in files:
if parse_mode:
new_name, status = plan_parse_rename(file_path)
else:
new_name, status = plan_default_rename(file_path)
if status == STATUS_RENAME and new_name:
target = file_path.parent / new_name
target_key = str(target).casefold()
if target.exists() and target.resolve() != file_path.resolve():
status = STATUS_CONFLICT
elif target_key in seen_targets:
prev_idx = seen_targets[target_key]
plan[prev_idx] = (plan[prev_idx][0], plan[prev_idx][1], STATUS_CONFLICT)
status = STATUS_CONFLICT
else:
seen_targets[target_key] = len(plan)
plan.append((file_path, new_name, status))
return plan
def build_number_plan(files, parse_mode, force=False):
"""Build a rename plan using sequential numbering per day."""
entries = []
for file_path in files:
if not force and NORMALIZED.match(file_path.name):
entries.append((file_path, None, STATUS_ALREADY))
continue
date_str = get_date_for_file(file_path, parse_mode)
entries.append((file_path, date_str))
day_counter = {}
plan = []
seen_targets = {}
for entry in entries:
if len(entry) == 3:
plan.append((entry[0], None, entry[2]))
continue
file_path, date_str = entry
if date_str is None:
plan.append((file_path, None, STATUS_NO_MATCH))
continue
day_counter[date_str] = day_counter.get(date_str, 0) + 1
seq = day_counter[date_str]
ext = file_path.suffix
new_name = f'{date_str} {seq:02d}{ext}'
if new_name == file_path.name:
plan.append((file_path, None, STATUS_ALREADY))
continue
status = STATUS_RENAME
target = file_path.parent / new_name
target_key = str(target).casefold()
if target.exists() and target.resolve() != file_path.resolve():
status = STATUS_CONFLICT
elif target_key in seen_targets:
prev_idx = seen_targets[target_key]
plan[prev_idx] = (plan[prev_idx][0], plan[prev_idx][1], STATUS_CONFLICT)
status = STATUS_CONFLICT
else:
seen_targets[target_key] = len(plan)
plan.append((file_path, new_name, status))
return plan
def print_preview(plan):
"""Print a formatted preview table."""
if not plan:
print('No files found.')
return
col1 = max(len(file_path.name) for file_path, _, _ in plan)
col2 = max(len(new_name or '') for _, new_name, _ in plan)
col1 = max(col1, len('Original'))
col2 = max(col2, len('New Name'))
header = f'{"Original":<{col1}} {"New Name":<{col2}} Status'
print(header)
print('-' * len(header))
for file_path, new_name, status in plan:
print(f'{file_path.name:<{col1}} {(new_name or ""):<{col2}} {status}')
print()
def apply_renames(plan):
"""Execute the planned renames."""
count = 0
for file_path, new_name, status in plan:
if status != STATUS_RENAME:
continue
file_path.rename(file_path.parent / new_name)
count += 1
return count
def confirm(prompt='Apply renames? [y/n] '):
"""Ask the user for confirmation, reacting to a single keypress."""
print(prompt, end='', flush=True)
if not sys.stdin.isatty():
return sys.stdin.read(1).strip().lower() == 'y'
fd = sys.stdin.fileno()
old_settings = termios.tcgetattr(fd)
try:
# cbreak (not raw) keeps Ctrl-C as SIGINT and newlines well behaved.
tty.setcbreak(fd)
while (key := sys.stdin.read(1).lower()) not in ('y', 'n', ''):
pass
finally:
termios.tcsetattr(fd, termios.TCSADRAIN, old_settings)
print(key)
return key == 'y'
def main():
parser = argparse.ArgumentParser(description='Normalize photo and image filenames.')
parser.add_argument(
'-p',
'--parse',
action='store_true',
help='Use EXIF data or file dates instead of pattern matching.',
)
parser.add_argument(
'-n',
'--number',
action='store_true',
help=(
'Discard original filename stems and number files sequentially '
'per day. Combines with --parse to select the date source. '
'Skips already-normalized files unless --force is given.'
),
)
parser.add_argument(
'-f',
'--force',
action='store_true',
help='Force numbering even for already-normalized files (used with -n).',
)
parser.add_argument(
'-a',
'--all',
action='store_true',
help='Include hidden files when scanning directories.',
)
parser.add_argument(
'files',
nargs='*',
help='Files or directories to process. Defaults to current directory.',
)
args = parser.parse_args()
files = collect_files(args.files, args.all)
if not files:
print('No files found.')
sys.exit(0)
if args.number:
plan = build_number_plan(files, args.parse, args.force)
else:
plan = build_plan(files, args.parse)
print_preview(plan)
renames = sum(1 for _, _, status in plan if status == STATUS_RENAME)
if renames == 0:
print('Nothing to rename.')
sys.exit(0)
if confirm():
print(f'Renamed {apply_renames(plan)} file(s).')
else:
print('Aborted.')
if __name__ == '__main__':
try:
main()
except KeyboardInterrupt:
print('\nAborted.')
sys.exit(1)