#!/usr/bin/env python3
"""Normalize photo and image filenames from phones and other sources."""

import argparse
import re
import shutil
import subprocess
import sys
from datetime import datetime
from functools import lru_cache
from pathlib import Path

PHONE_PATTERN = re.compile(
    r'^[A-Za-z]+[_-]'
    r'(\d{4})-?(\d{2})-?(\d{2})'
    r'[_-]'
    r'(.+)$'
)

# Telegram photo/video export, e.g. "photo_94@16-07-2026_22-09-43.jpg"
# (photo|video)_<n>@<DD>-<MM>-<YYYY>_<HH>-<MM>-<SS>
TELEGRAM_PATTERN = re.compile(
    r'^(?:photo|video)_(\d+)@(\d{2})-(\d{2})-(\d{4})_(\d{2})-(\d{2})-(\d{2})$'
)

NORMALIZED = re.compile(r'^\d{4}-\d{2}-\d{2} .+')

# Filename prefixes whose embedded timestamp is UTC rather than local time.
# The Pixel camera names captures in UTC but records the local wall clock time
# in EXIF, so for these the metadata is authoritative and the filename is not.
# Other sources (Samsung's IMG_/VID_, Telegram exports) already name in local
# time and are left alone.
UTC_FILENAME_PREFIXES = ('PXL',)

# e.g. "PXL_20260917_163001280.MP" -> prefix, timestamp, subseconds, suffix
UTC_STEM_PATTERN = re.compile(
    r'^(' + '|'.join(UTC_FILENAME_PREFIXES) + r')[_-]'
    r'(\d{8}[_-]\d{6})'
    r'(\d*)'
    r'(.*)$',
    re.IGNORECASE,
)

# Windows hidden-file attribute, mirrored from the stat module.
FILE_ATTRIBUTE_HIDDEN = 0x2

STATUS_RENAME = 'rename'
STATUS_ALREADY = 'skip (already normalized)'
STATUS_NO_MATCH = 'skip (no rule matched)'
STATUS_CONFLICT = 'skip (conflict)'


@lru_cache(maxsize=1)
def has_exiftool():
    """Check once whether exiftool is available on PATH."""
    return shutil.which('exiftool') is not None


def get_exif_date(filepath):
    """Try to get the image creation date via exiftool."""
    try:
        result = subprocess.run(
            ['exiftool', '-s3', '-d', '%Y-%m-%d', '-DateTimeOriginal', str(filepath)],
            capture_output=True,
            text=True,
            timeout=10,
        )
        date_str = result.stdout.strip()
        if date_str and re.match(r'^\d{4}-\d{2}-\d{2}$', date_str):
            return date_str
    except (FileNotFoundError, subprocess.TimeoutExpired):
        pass
    return None


def get_exif_timestamp(filepath):
    """Get the local capture time via exiftool as (YYYYMMDD_HHMMSS, subsec).

    DateTimeOriginal is the wall clock time at the moment of capture, so it
    needs no timezone correction. Returns None when unavailable.
    """
    try:
        result = subprocess.run(
            [
                'exiftool', '-s3', '-d', '%Y%m%d_%H%M%S',
                '-DateTimeOriginal', '-SubSecTimeOriginal', str(filepath),
            ],
            capture_output=True,
            text=True,
            timeout=10,
        )
    except (FileNotFoundError, subprocess.TimeoutExpired):
        return None

    lines = result.stdout.strip().splitlines()
    if not lines or not re.match(r'^\d{8}_\d{6}$', lines[0].strip()):
        return None

    subsec = lines[1].strip() if len(lines) > 1 else ''
    return lines[0].strip(), subsec if subsec.isdigit() else ''


def resolve_stem(filepath):
    """Return the file's stem, with UTC-named timestamps corrected from EXIF.

    Pixel filenames encode UTC, so the name alone is off by the UTC offset in
    effect at capture. Falls back to the original stem when the file is not a
    UTC-named capture, or when its metadata has no usable timestamp.
    """
    stem = filepath.stem

    match = UTC_STEM_PATTERN.match(stem)
    if not match or not has_exiftool():
        return stem

    prefix, _, subsec, rest = match.groups()
    stamp = get_exif_timestamp(filepath)
    if stamp is None:
        return stem

    exif_stamp, exif_subsec = stamp
    return f'{prefix}_{exif_stamp}{exif_subsec or subsec}{rest}'


def get_file_date(filepath):
    """Get file creation date (or modification time as fallback)."""
    stat = filepath.stat()
    ts = getattr(stat, 'st_birthtime', None) or stat.st_mtime
    return datetime.fromtimestamp(ts).strftime('%Y-%m-%d')


def plan_default_rename(filepath):
    """Plan a rename using the default pattern-matching mode."""
    if NORMALIZED.match(filepath.name):
        return None, STATUS_ALREADY

    stem = resolve_stem(filepath)
    ext = filepath.suffix

    tg_match = TELEGRAM_PATTERN.match(stem)
    if tg_match:
        number, day, month, year, hour, minute, second = tg_match.groups()
        return f'{year}-{month}-{day} {hour}{minute}{second} {number}{ext}', STATUS_RENAME

    match = PHONE_PATTERN.match(stem)
    if not match:
        match = PHONE_PATTERN.match(f'{stem}{ext}')
        if match:
            year, month, day, rest = match.groups()
            new_name = f'{year}-{month}-{day} {rest}'
        else:
            return None, STATUS_NO_MATCH
    else:
        year, month, day, rest = match.groups()
        new_name = f'{year}-{month}-{day} {rest}{ext}'

    return new_name, STATUS_RENAME


def plan_parse_rename(filepath):
    """Plan a rename using EXIF or file date."""
    if NORMALIZED.match(filepath.name):
        return None, STATUS_ALREADY

    date_str = None
    if has_exiftool():
        date_str = get_exif_date(filepath)

    if not date_str:
        date_str = get_file_date(filepath)

    stem = filepath.stem
    ext = filepath.suffix

    if stem.startswith(date_str):
        new_name = f'{date_str} {stem[len(date_str):].lstrip(" -_")}{ext}'
    else:
        new_name = f'{date_str} {stem}{ext}'

    if new_name == f'{date_str} {ext}':
        new_name = f'{date_str}{ext}'

    if new_name == filepath.name:
        return None, STATUS_ALREADY

    return new_name, STATUS_RENAME


def get_date_for_file(filepath, parse_mode):
    """Resolve a date string for a file, or None if unavailable."""
    if parse_mode:
        date_str = None
        if has_exiftool():
            date_str = get_exif_date(filepath)
        if not date_str:
            date_str = get_file_date(filepath)
        return date_str

    # Default pattern mode: extract date from filename
    stem = resolve_stem(filepath)

    # Check if already normalized (e.g. "2024-06-01 breakfast.jpg")
    norm_match = re.match(r'^(\d{4})-(\d{2})-(\d{2}) ', filepath.name)
    if norm_match:
        return f'{norm_match.group(1)}-{norm_match.group(2)}-{norm_match.group(3)}'

    # Telegram export pattern (date is DD-MM-YYYY)
    tg_match = TELEGRAM_PATTERN.match(stem)
    if tg_match:
        _, day, month, year = tg_match.groups()[:4]
        return f'{year}-{month}-{day}'

    # Try the standard pattern
    match = PHONE_PATTERN.match(stem)
    if not match:
        match = PHONE_PATTERN.match(f'{stem}{filepath.suffix}')
    if match:
        year, month, day, _ = match.groups()
        return f'{year}-{month}-{day}'

    return None


def is_hidden(filepath):
    """Check whether a file is hidden (dotfile, or hidden attribute on Windows)."""
    if filepath.name.startswith('.'):
        return True

    try:
        attrs = filepath.stat().st_file_attributes
    except (AttributeError, OSError):
        return False
    return bool(attrs & FILE_ATTRIBUTE_HIDDEN)


def collect_files(paths, include_hidden=False):
    """Resolve the list of files to process."""
    if not paths:
        paths = ['.']

    files = []
    for path in paths:
        candidate = Path(path)
        if candidate.is_dir():
            files.extend(sorted(
                file for file in candidate.iterdir()
                if file.is_file() and (include_hidden or not is_hidden(file))
            ))
        elif candidate.is_file():
            # Explicitly named files are always processed.
            files.append(candidate)
    return files


def build_plan(files, parse_mode):
    """Build a list of (original_path, new_name, status) tuples."""
    plan = []
    seen_targets = {}

    for file_path in files:
        if parse_mode:
            new_name, status = plan_parse_rename(file_path)
        else:
            new_name, status = plan_default_rename(file_path)

        if status == STATUS_RENAME and new_name:
            target = file_path.parent / new_name
            target_key = str(target).casefold()

            if target.exists() and target.resolve() != file_path.resolve():
                status = STATUS_CONFLICT
            elif target_key in seen_targets:
                prev_idx = seen_targets[target_key]
                plan[prev_idx] = (plan[prev_idx][0], plan[prev_idx][1], STATUS_CONFLICT)
                status = STATUS_CONFLICT
            else:
                seen_targets[target_key] = len(plan)

        plan.append((file_path, new_name, status))

    return plan


def build_number_plan(files, parse_mode, force=False):
    """Build a rename plan using sequential numbering per day."""
    entries = []
    for file_path in files:
        if not force and NORMALIZED.match(file_path.name):
            entries.append((file_path, None, STATUS_ALREADY))
            continue
        date_str = get_date_for_file(file_path, parse_mode)
        entries.append((file_path, date_str))

    day_counter = {}
    plan = []
    seen_targets = {}

    for entry in entries:
        if len(entry) == 3:
            plan.append((entry[0], None, entry[2]))
            continue
        file_path, date_str = entry
        if date_str is None:
            plan.append((file_path, None, STATUS_NO_MATCH))
            continue

        day_counter[date_str] = day_counter.get(date_str, 0) + 1
        seq = day_counter[date_str]
        ext = file_path.suffix
        new_name = f'{date_str} {seq:02d}{ext}'

        if new_name == file_path.name:
            plan.append((file_path, None, STATUS_ALREADY))
            continue

        status = STATUS_RENAME
        target = file_path.parent / new_name
        target_key = str(target).casefold()

        if target.exists() and target.resolve() != file_path.resolve():
            status = STATUS_CONFLICT
        elif target_key in seen_targets:
            prev_idx = seen_targets[target_key]
            plan[prev_idx] = (plan[prev_idx][0], plan[prev_idx][1], STATUS_CONFLICT)
            status = STATUS_CONFLICT
        else:
            seen_targets[target_key] = len(plan)

        plan.append((file_path, new_name, status))

    return plan


def print_preview(plan):
    """Print a formatted preview table."""
    if not plan:
        print('No files found.')
        return

    col1 = max(len(file_path.name) for file_path, _, _ in plan)
    col2 = max(len(new_name or '') for _, new_name, _ in plan)
    col1 = max(col1, len('Original'))
    col2 = max(col2, len('New Name'))

    header = f'{"Original":<{col1}}  {"New Name":<{col2}}  Status'
    print(header)
    print('-' * len(header))

    for file_path, new_name, status in plan:
        print(f'{file_path.name:<{col1}}  {(new_name or ""):<{col2}}  {status}')

    print()


def apply_renames(plan):
    """Execute the planned renames."""
    count = 0
    for file_path, new_name, status in plan:
        if status != STATUS_RENAME:
            continue
        file_path.rename(file_path.parent / new_name)
        count += 1
    return count


def confirm(prompt='Apply renames? [Y/n] '):
    """Ask the user for confirmation. An empty answer accepts the default (yes)."""
    while True:
        try:
            answer = input(prompt).strip().lower()
        except EOFError:
            print()
            return True
        if answer in ('', 'y', 'yes'):
            return True
        if answer in ('n', 'no'):
            return False


def main():
    parser = argparse.ArgumentParser(description='Normalize photo and image filenames.')
    parser.add_argument(
        '-p',
        '--parse',
        action='store_true',
        help='Use EXIF data or file dates instead of pattern matching.',
    )
    parser.add_argument(
        '-n',
        '--number',
        action='store_true',
        help=(
            'Discard original filename stems and number files sequentially '
            'per day. Combines with --parse to select the date source. '
            'Skips already-normalized files unless --force is given.'
        ),
    )
    parser.add_argument(
        '-f',
        '--force',
        action='store_true',
        help='Force numbering even for already-normalized files (used with -n).',
    )
    parser.add_argument(
        '-a',
        '--all',
        action='store_true',
        help='Include hidden files when scanning directories.',
    )
    parser.add_argument(
        'files',
        nargs='*',
        help='Files or directories to process. Defaults to current directory.',
    )

    args = parser.parse_args()

    files = collect_files(args.files, args.all)
    if not files:
        print('No files found.')
        sys.exit(0)

    if args.number:
        plan = build_number_plan(files, args.parse, args.force)
    else:
        plan = build_plan(files, args.parse)
    print_preview(plan)

    renames = sum(1 for _, _, status in plan if status == STATUS_RENAME)
    if renames == 0:
        print('Nothing to rename.')
        sys.exit(0)

    if confirm():
        print(f'Renamed {apply_renames(plan)} file(s).')
    else:
        print('Aborted.')


if __name__ == '__main__':
    try:
        main()
    except KeyboardInterrupt:
        print('\nAborted.')
        sys.exit(1)
