From c092140d2f2c8897fe1b99692061f9528e9c3eef Mon Sep 17 00:00:00 2001 From: luxick Date: Sat, 26 Sep 2026 17:38:49 +0200 Subject: [PATCH] Update nn Improve exif parsing performance --- bin/nn | 96 +++++++++++++++++++++++++++++++++++++++------------------- 1 file changed, 65 insertions(+), 31 deletions(-) diff --git a/bin/nn b/bin/nn index 0fdb7f1..86a0adf 100755 --- a/bin/nn +++ b/bin/nn @@ -2,6 +2,8 @@ """Normalize photo and image filenames from phones and other sources.""" import argparse +import json +import os import re import shutil import subprocess @@ -56,48 +58,74 @@ def has_exiftool(): return shutil.which('exiftool') is not None -def get_exif_date(filepath): - """Try to get the image creation date via exiftool.""" +# Capture timestamps keyed by exif_key(path), filled by prefetch_exif(). +_exif_cache = {} + + +def exif_key(filepath): + """Normalize a path so exiftool's SourceFile and our Path agree.""" + return os.path.normcase(os.path.abspath(filepath)) + + +def prefetch_exif(files): + """Read capture timestamps for all files with a single exiftool call. + + Starting exiftool dominates the cost of reading a few tags, so reading + every file in one process is far faster than one process per file. + Paths are passed on stdin to stay clear of command line length limits. + """ + files = [f for f in files if exif_key(f) not in _exif_cache] + if not files or not has_exiftool(): + return + + # Files without a usable timestamp stay None so they aren't re-queried. + _exif_cache.update((exif_key(f), None) for f in files) + + args = [ + '-json', '-fast', '-charset', 'filename=utf8', + '-d', '%Y%m%d_%H%M%S', + '-DateTimeOriginal', '-SubSecTimeOriginal', + *(str(f) for f in files), + ] try: result = subprocess.run( - ['exiftool', '-s3', '-d', '%Y-%m-%d', '-DateTimeOriginal', str(filepath)], + ['exiftool', '-@', '-'], + input='\n'.join(args), capture_output=True, - text=True, - timeout=10, + encoding='utf-8', + timeout=60 + len(files), ) - date_str = result.stdout.strip() - if date_str and re.match(r'^\d{4}-\d{2}-\d{2}$', date_str): - return date_str - except (FileNotFoundError, subprocess.TimeoutExpired): - pass - return None + records = json.loads(result.stdout or '[]') + except (FileNotFoundError, subprocess.TimeoutExpired, json.JSONDecodeError): + return + + for record in records: + stamp = str(record.get('DateTimeOriginal', '')) + if not re.match(r'^\d{8}_\d{6}$', stamp): + continue + subsec = str(record.get('SubSecTimeOriginal', '')) + _exif_cache[exif_key(record['SourceFile'])] = ( + stamp, subsec if subsec.isdigit() else '', + ) + + +def get_exif_date(filepath): + """Get the image creation date (YYYY-MM-DD) from EXIF, or None.""" + stamp = get_exif_timestamp(filepath) + if stamp is None: + return None + day = stamp[0] + return f'{day[:4]}-{day[4:6]}-{day[6:8]}' def get_exif_timestamp(filepath): - """Get the local capture time via exiftool as (YYYYMMDD_HHMMSS, subsec). + """Get the local capture time from EXIF as (YYYYMMDD_HHMMSS, subsec). DateTimeOriginal is the wall clock time at the moment of capture, so it needs no timezone correction. Returns None when unavailable. """ - try: - result = subprocess.run( - [ - 'exiftool', '-s3', '-d', '%Y%m%d_%H%M%S', - '-DateTimeOriginal', '-SubSecTimeOriginal', str(filepath), - ], - capture_output=True, - text=True, - timeout=10, - ) - except (FileNotFoundError, subprocess.TimeoutExpired): - return None - - lines = result.stdout.strip().splitlines() - if not lines or not re.match(r'^\d{8}_\d{6}$', lines[0].strip()): - return None - - subsec = lines[1].strip() if len(lines) > 1 else '' - return lines[0].strip(), subsec if subsec.isdigit() else '' + prefetch_exif([filepath]) + return _exif_cache.get(exif_key(filepath)) def resolve_stem(filepath): @@ -420,6 +448,12 @@ def main(): print('No files found.') sys.exit(0) + prefetch_exif([ + f for f in files + if (args.force or not NORMALIZED.match(f.name)) + and (args.parse or UTC_STEM_PATTERN.match(f.stem)) + ]) + if args.number: plan = build_number_plan(files, args.parse, args.force) else: