Update nn

Improve exif parsing performance
This commit is contained in:
2026-09-26 17:38:49 +02:00
parent 51b63eb6b6
commit c092140d2f
+64 -30
View File
@@ -2,6 +2,8 @@
"""Normalize photo and image filenames from phones and other sources.""" """Normalize photo and image filenames from phones and other sources."""
import argparse import argparse
import json
import os
import re import re
import shutil import shutil
import subprocess import subprocess
@@ -56,48 +58,74 @@ def has_exiftool():
return shutil.which('exiftool') is not None return shutil.which('exiftool') is not None
def get_exif_date(filepath): # Capture timestamps keyed by exif_key(path), filled by prefetch_exif().
"""Try to get the image creation date via exiftool.""" _exif_cache = {}
def exif_key(filepath):
"""Normalize a path so exiftool's SourceFile and our Path agree."""
return os.path.normcase(os.path.abspath(filepath))
def prefetch_exif(files):
"""Read capture timestamps for all files with a single exiftool call.
Starting exiftool dominates the cost of reading a few tags, so reading
every file in one process is far faster than one process per file.
Paths are passed on stdin to stay clear of command line length limits.
"""
files = [f for f in files if exif_key(f) not in _exif_cache]
if not files or not has_exiftool():
return
# Files without a usable timestamp stay None so they aren't re-queried.
_exif_cache.update((exif_key(f), None) for f in files)
args = [
'-json', '-fast', '-charset', 'filename=utf8',
'-d', '%Y%m%d_%H%M%S',
'-DateTimeOriginal', '-SubSecTimeOriginal',
*(str(f) for f in files),
]
try: try:
result = subprocess.run( result = subprocess.run(
['exiftool', '-s3', '-d', '%Y-%m-%d', '-DateTimeOriginal', str(filepath)], ['exiftool', '-@', '-'],
input='\n'.join(args),
capture_output=True, capture_output=True,
text=True, encoding='utf-8',
timeout=10, timeout=60 + len(files),
) )
date_str = result.stdout.strip() records = json.loads(result.stdout or '[]')
if date_str and re.match(r'^\d{4}-\d{2}-\d{2}$', date_str): except (FileNotFoundError, subprocess.TimeoutExpired, json.JSONDecodeError):
return date_str return
except (FileNotFoundError, subprocess.TimeoutExpired):
pass for record in records:
stamp = str(record.get('DateTimeOriginal', ''))
if not re.match(r'^\d{8}_\d{6}$', stamp):
continue
subsec = str(record.get('SubSecTimeOriginal', ''))
_exif_cache[exif_key(record['SourceFile'])] = (
stamp, subsec if subsec.isdigit() else '',
)
def get_exif_date(filepath):
"""Get the image creation date (YYYY-MM-DD) from EXIF, or None."""
stamp = get_exif_timestamp(filepath)
if stamp is None:
return None return None
day = stamp[0]
return f'{day[:4]}-{day[4:6]}-{day[6:8]}'
def get_exif_timestamp(filepath): def get_exif_timestamp(filepath):
"""Get the local capture time via exiftool as (YYYYMMDD_HHMMSS, subsec). """Get the local capture time from EXIF as (YYYYMMDD_HHMMSS, subsec).
DateTimeOriginal is the wall clock time at the moment of capture, so it DateTimeOriginal is the wall clock time at the moment of capture, so it
needs no timezone correction. Returns None when unavailable. needs no timezone correction. Returns None when unavailable.
""" """
try: prefetch_exif([filepath])
result = subprocess.run( return _exif_cache.get(exif_key(filepath))
[
'exiftool', '-s3', '-d', '%Y%m%d_%H%M%S',
'-DateTimeOriginal', '-SubSecTimeOriginal', str(filepath),
],
capture_output=True,
text=True,
timeout=10,
)
except (FileNotFoundError, subprocess.TimeoutExpired):
return None
lines = result.stdout.strip().splitlines()
if not lines or not re.match(r'^\d{8}_\d{6}$', lines[0].strip()):
return None
subsec = lines[1].strip() if len(lines) > 1 else ''
return lines[0].strip(), subsec if subsec.isdigit() else ''
def resolve_stem(filepath): def resolve_stem(filepath):
@@ -420,6 +448,12 @@ def main():
print('No files found.') print('No files found.')
sys.exit(0) sys.exit(0)
prefetch_exif([
f for f in files
if (args.force or not NORMALIZED.match(f.name))
and (args.parse or UTC_STEM_PATTERN.match(f.stem))
])
if args.number: if args.number:
plan = build_number_plan(files, args.parse, args.force) plan = build_number_plan(files, args.parse, args.force)
else: else: