Update nn
Improve exif parsing performance
This commit is contained in:
@@ -2,6 +2,8 @@
|
|||||||
"""Normalize photo and image filenames from phones and other sources."""
|
"""Normalize photo and image filenames from phones and other sources."""
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
|
import json
|
||||||
|
import os
|
||||||
import re
|
import re
|
||||||
import shutil
|
import shutil
|
||||||
import subprocess
|
import subprocess
|
||||||
@@ -56,48 +58,74 @@ def has_exiftool():
|
|||||||
return shutil.which('exiftool') is not None
|
return shutil.which('exiftool') is not None
|
||||||
|
|
||||||
|
|
||||||
def get_exif_date(filepath):
|
# Capture timestamps keyed by exif_key(path), filled by prefetch_exif().
|
||||||
"""Try to get the image creation date via exiftool."""
|
_exif_cache = {}
|
||||||
|
|
||||||
|
|
||||||
|
def exif_key(filepath):
|
||||||
|
"""Normalize a path so exiftool's SourceFile and our Path agree."""
|
||||||
|
return os.path.normcase(os.path.abspath(filepath))
|
||||||
|
|
||||||
|
|
||||||
|
def prefetch_exif(files):
|
||||||
|
"""Read capture timestamps for all files with a single exiftool call.
|
||||||
|
|
||||||
|
Starting exiftool dominates the cost of reading a few tags, so reading
|
||||||
|
every file in one process is far faster than one process per file.
|
||||||
|
Paths are passed on stdin to stay clear of command line length limits.
|
||||||
|
"""
|
||||||
|
files = [f for f in files if exif_key(f) not in _exif_cache]
|
||||||
|
if not files or not has_exiftool():
|
||||||
|
return
|
||||||
|
|
||||||
|
# Files without a usable timestamp stay None so they aren't re-queried.
|
||||||
|
_exif_cache.update((exif_key(f), None) for f in files)
|
||||||
|
|
||||||
|
args = [
|
||||||
|
'-json', '-fast', '-charset', 'filename=utf8',
|
||||||
|
'-d', '%Y%m%d_%H%M%S',
|
||||||
|
'-DateTimeOriginal', '-SubSecTimeOriginal',
|
||||||
|
*(str(f) for f in files),
|
||||||
|
]
|
||||||
try:
|
try:
|
||||||
result = subprocess.run(
|
result = subprocess.run(
|
||||||
['exiftool', '-s3', '-d', '%Y-%m-%d', '-DateTimeOriginal', str(filepath)],
|
['exiftool', '-@', '-'],
|
||||||
|
input='\n'.join(args),
|
||||||
capture_output=True,
|
capture_output=True,
|
||||||
text=True,
|
encoding='utf-8',
|
||||||
timeout=10,
|
timeout=60 + len(files),
|
||||||
)
|
)
|
||||||
date_str = result.stdout.strip()
|
records = json.loads(result.stdout or '[]')
|
||||||
if date_str and re.match(r'^\d{4}-\d{2}-\d{2}$', date_str):
|
except (FileNotFoundError, subprocess.TimeoutExpired, json.JSONDecodeError):
|
||||||
return date_str
|
return
|
||||||
except (FileNotFoundError, subprocess.TimeoutExpired):
|
|
||||||
pass
|
for record in records:
|
||||||
return None
|
stamp = str(record.get('DateTimeOriginal', ''))
|
||||||
|
if not re.match(r'^\d{8}_\d{6}$', stamp):
|
||||||
|
continue
|
||||||
|
subsec = str(record.get('SubSecTimeOriginal', ''))
|
||||||
|
_exif_cache[exif_key(record['SourceFile'])] = (
|
||||||
|
stamp, subsec if subsec.isdigit() else '',
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def get_exif_date(filepath):
|
||||||
|
"""Get the image creation date (YYYY-MM-DD) from EXIF, or None."""
|
||||||
|
stamp = get_exif_timestamp(filepath)
|
||||||
|
if stamp is None:
|
||||||
|
return None
|
||||||
|
day = stamp[0]
|
||||||
|
return f'{day[:4]}-{day[4:6]}-{day[6:8]}'
|
||||||
|
|
||||||
|
|
||||||
def get_exif_timestamp(filepath):
|
def get_exif_timestamp(filepath):
|
||||||
"""Get the local capture time via exiftool as (YYYYMMDD_HHMMSS, subsec).
|
"""Get the local capture time from EXIF as (YYYYMMDD_HHMMSS, subsec).
|
||||||
|
|
||||||
DateTimeOriginal is the wall clock time at the moment of capture, so it
|
DateTimeOriginal is the wall clock time at the moment of capture, so it
|
||||||
needs no timezone correction. Returns None when unavailable.
|
needs no timezone correction. Returns None when unavailable.
|
||||||
"""
|
"""
|
||||||
try:
|
prefetch_exif([filepath])
|
||||||
result = subprocess.run(
|
return _exif_cache.get(exif_key(filepath))
|
||||||
[
|
|
||||||
'exiftool', '-s3', '-d', '%Y%m%d_%H%M%S',
|
|
||||||
'-DateTimeOriginal', '-SubSecTimeOriginal', str(filepath),
|
|
||||||
],
|
|
||||||
capture_output=True,
|
|
||||||
text=True,
|
|
||||||
timeout=10,
|
|
||||||
)
|
|
||||||
except (FileNotFoundError, subprocess.TimeoutExpired):
|
|
||||||
return None
|
|
||||||
|
|
||||||
lines = result.stdout.strip().splitlines()
|
|
||||||
if not lines or not re.match(r'^\d{8}_\d{6}$', lines[0].strip()):
|
|
||||||
return None
|
|
||||||
|
|
||||||
subsec = lines[1].strip() if len(lines) > 1 else ''
|
|
||||||
return lines[0].strip(), subsec if subsec.isdigit() else ''
|
|
||||||
|
|
||||||
|
|
||||||
def resolve_stem(filepath):
|
def resolve_stem(filepath):
|
||||||
@@ -420,6 +448,12 @@ def main():
|
|||||||
print('No files found.')
|
print('No files found.')
|
||||||
sys.exit(0)
|
sys.exit(0)
|
||||||
|
|
||||||
|
prefetch_exif([
|
||||||
|
f for f in files
|
||||||
|
if (args.force or not NORMALIZED.match(f.name))
|
||||||
|
and (args.parse or UTC_STEM_PATTERN.match(f.stem))
|
||||||
|
])
|
||||||
|
|
||||||
if args.number:
|
if args.number:
|
||||||
plan = build_number_plan(files, args.parse, args.force)
|
plan = build_number_plan(files, args.parse, args.force)
|
||||||
else:
|
else:
|
||||||
|
|||||||
Reference in New Issue
Block a user