#!/usr/bin/env python3
"""
Subtitle Pipeline (default: English → Ukrainian)
Priority order per episode:
  1. OpenSubtitles.com — search & download existing target-language subtitles
     └─ timing sync check & correction (linear scale + offset)
  2. Configurable translation chain for entries that still need translation
  3. Claude — optional final fallback when the local usage guard allows it

Usage:
  python3 translate_subtitles.py [max_files] [--min-tokens N] [--debug-tokens]

  python3 translate_subtitles.py 11                  # use the default 30% threshold
  python3 translate_subtitles.py 11 --min-tokens 50  # stop Claude below 50%
  python3 translate_subtitles.py --min-tokens 0      # always allow Claude
  python3 translate_subtitles.py --debug-tokens      # check the explicit usage file

Setup:
  python3 -m venv .venv
  source .venv/bin/activate
  python -m pip install deep-translator argostranslate
  export OPENSUBTITLES_API_KEY="..."
"""

import re, sys, os, json, time, shutil, subprocess, gzip, signal, atexit, math, tempfile
from pathlib import Path
from datetime import datetime
import urllib.request, urllib.parse, urllib.error

# Keep system awake (macOS only) — run caffeinate as background daemon so Ctrl+C still works
if os.uname().sysname == 'Darwin':
    _cafe = subprocess.Popen(
        ['/usr/bin/caffeinate', '-i', '-w', str(os.getpid())],
        stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
    )
    atexit.register(_cafe.terminate)

# ─── Language settings ───────────────────────────────────────────────────────
# Change this block to reuse the pipeline for another language pair.
SOURCE_LANGUAGE_CODE   = "en"
SOURCE_LANGUAGE_NAME   = "English"
SOURCE_LANGUAGE_LOCALE = "en-US"
SOURCE_INPUT_SUFFIX    = "en"

TARGET_LANGUAGE_CODE   = "uk"
TARGET_LANGUAGE_NAME   = "Ukrainian"
TARGET_LANGUAGE_LOCALE = "uk-UA"
TARGET_OUTPUT_SUFFIX   = "uk"

# The script-ratio check catches untranslated text when source and target use
# different writing systems. Set the marker patterns to "" when the target
# language has no useful language-specific characters.
ALL_LETTERS_PATTERN               = r'[a-zA-Zа-яА-ЯіІїЇєЄґҐ]'
TARGET_SCRIPT_PATTERN             = r'[а-яА-ЯіІїЇєЄґҐ]'
TARGET_LANGUAGE_MARKERS_PATTERN   = r'[іІїЇєЄґҐ]'
REJECTED_LANGUAGE_MARKERS_PATTERN = r'[ыЫэЭъЪ]'

# ─── Runtime settings ────────────────────────────────────────────────────────
OPENSUBTITLES_API_KEY      = os.environ.get("OPENSUBTITLES_API_KEY", "")
OPENSUBTITLES_ACCESS_TOKEN = os.environ.get("OPENSUBTITLES_ACCESS_TOKEN", "")
OPENSUBTITLES_USERNAME     = os.environ.get("OPENSUBTITLES_USERNAME", "")
OPENSUBTITLES_PASSWORD     = os.environ.get("OPENSUBTITLES_PASSWORD", "")
OPENSUBTITLES_APP_NAME     = "SubtitleTranslator/1.0"

BATCH_SIZE          = 80
CLAUDE_MODEL        = "claude-haiku-4-5-20251001"
DELAY_BETWEEN_CALLS = 0.4            # seconds between API calls
MAX_FILES_PER_RUN   = 30
MIN_QUALITY         = 0.70           # min target-script ratio accepted from any translator
MAX_SYNC_GAP_MS     = 2000           # max allowed timing error after sync (ms)
MAX_SYNC_DRIFT      = 0.10           # max scale deviation from 1.0 (10% = huge fps diff)
OS_DAILY_LIMIT      = int(os.environ.get("OPENSUBTITLES_DAILY_LIMIT", "20"))
MIN_TOKENS_PCT      = 30             # stop using Claude below this % free tokens (0-100)

# ─── Translation chain (Chain of Responsibility) ─────────────────────────────
# Format: "Service (fallbacks:timeout_inc) -> ..."
#   fallbacks    - retries after the first attempt (0 = one attempt only)
#   timeout_inc  - seconds added to the timeout for each retry
# The first service runs the primary batch. The rest handle unresolved entries.
# Unavailable services are skipped automatically.
TRANSLATOR_CHAIN = "Google (2:30) -> MyMemory (1:10) -> LibreTranslate (1:20) -> Argos -> Claude"

# ─── Local LibreTranslate server ──────────────────────────────────────────────
# POST /translate  {"q": text, "source": SOURCE_LANGUAGE_CODE,
#                    "target": TARGET_LANGUAGE_CODE, "format": "text"}
# Leave this empty to disable LibreTranslate.
LIBRETRANSLATE_URL = os.environ.get("LIBRETRANSLATE_URL", "")
# ─────────────────────────────────────────────────────────────────────────────

_token_pct: float | None = None      # current externally supplied usage estimate (module state)
_argos_translator   = None           # cached argostranslate translation object


def _init_argos():
    """
    Load the configured Argos Translate model.
    If it is missing, install the matching source-to-target package.
    """
    global _argos_translator
    if _argos_translator is not None:
        return _argos_translator
    try:
        from argostranslate import translate

        def _get_translator():
            installed = translate.get_installed_languages()
            source = next(
                (language for language in installed
                 if language.code == SOURCE_LANGUAGE_CODE),
                None,
            )
            target = next(
                (language for language in installed
                 if language.code == TARGET_LANGUAGE_CODE),
                None,
            )
            return source.get_translation(target) if (source and target) else None

        tr = _get_translator()
        if tr is None:
            package = f"translate-{SOURCE_LANGUAGE_CODE}_{TARGET_LANGUAGE_CODE}"
            print(f"  ⬇ Argos: installing {package}...", flush=True)
            subprocess.run(['argospm', 'install', package], check=True, timeout=120)
            tr = _get_translator()

        _argos_translator = tr
    except Exception:
        pass
    return _argos_translator


# ═══════════════════════════════════════════════════════════════════
# CHAIN CONFIG PARSER
# ═══════════════════════════════════════════════════════════════════

_KNOWN_SERVICES = ('Google', 'Argos', 'MyMemory', 'LibreTranslate', 'Claude')


def _parse_chain_config(chain_str: str) -> list:
    """
    Parse TRANSLATOR_CHAIN into list of step dicts.
    Each dict: {'name': str, 'fallbacks': int, 'timeout_inc': int}
      fallbacks=0  → 1 total attempt (no retry)
      fallbacks=N  → N+1 total attempts
    """
    steps = []
    for part in re.split(r'\s*->\s*', chain_str.strip()):
        m = re.fullmatch(
            r'(\w+)\s*(?:\(\s*(\d+)\s*(?::\s*(\d+)\s*)?\))?',
            part.strip(),
        )
        if not m:
            raise ValueError(f"Invalid translator step: {part!r}")
        name        = m.group(1)
        if name not in _KNOWN_SERVICES:
            raise ValueError(f"Unknown translator service: {name!r}")
        fallbacks   = int(m.group(2)) if m.group(2) is not None else 0
        timeout_inc = int(m.group(3)) if m.group(3) is not None else 0
        steps.append({'name': name, 'fallbacks': fallbacks, 'timeout_inc': timeout_inc})
    return steps


def _check_service(name: str, has_argos: bool) -> tuple:
    """Returns (available: bool, reason: str)."""
    if name == 'Argos':
        return (True, 'offline model ready') if has_argos \
               else (
                   False,
                   'not installed '
                   f'(argostranslate + translate-'
                   f'{SOURCE_LANGUAGE_CODE}_{TARGET_LANGUAGE_CODE})',
               )
    if name in ('Google', 'MyMemory'):
        try:
            import deep_translator  # noqa
            return True, 'deep-translator ✓'
        except ImportError:
            return False, 'pip install deep-translator'
    if name == 'LibreTranslate':
        if LIBRETRANSLATE_URL:
            return True, f'POST {LIBRETRANSLATE_URL}/translate'
        return False, 'LIBRETRANSLATE_URL is not set'
    if name == 'Claude':
        executable = shutil.which('claude')
        if executable:
            return True, f'Claude CLI ({executable})'
        return False, 'claude executable not found in PATH'
    return False, 'unknown service'


def _build_chain_description(has_argos: bool) -> tuple:
    """
    Parse TRANSLATOR_CHAIN, check availability of each service.
    Returns (resolved_steps: list, description: str).
    resolved_steps — only available services, in order.
    """
    parsed   = _parse_chain_config(TRANSLATOR_CHAIN)
    resolved = []
    rows     = []

    for i, step in enumerate(parsed):
        name    = step['name']
        fb      = step['fallbacks']
        tinc    = step['timeout_inc']
        avail, reason = _check_service(name, has_argos)

        # Human-readable attempt count
        total_att = fb + 1
        if total_att == 1:
            att_str = '1 attempt'
        elif total_att == 2:
            att_str = '2 attempts'
        else:
            att_str = f'{total_att} attempts'
        if tinc:
            att_str += f', +{tinc}s/retry'

        extra = ''

        is_last   = (i == len(parsed) - 1)
        connector = '└─' if is_last else '├─'

        if avail:
            tag = '✓'
            resolved.append(step)
        else:
            tag = '⚠ skipped'
            reason = f'({reason})'

        rows.append(
            f'    {connector} {name:<10} {att_str:<22}{extra}  {tag}  {reason if not avail else ""}'
        )

    primary = resolved[0]['name'] if resolved else '—'
    sep     = '═' * 64

    lines = [
        f'\n{sep}',
        f' 🔗 TRANSLATOR_CHAIN: {TRANSLATOR_CHAIN}',
        f'',
        f'    Phase 1 (primary):     {primary}',
        f'    Phase 2 (CoR fallback):',
    ] + rows + [sep]

    return resolved, '\n'.join(lines)


# ═══════════════════════════════════════════════════════════════════
# TIMING UTILS
# ═══════════════════════════════════════════════════════════════════

def parse_time(s):
    """HH:MM:SS,mmm → milliseconds"""
    h, m, rest = s.strip().split(':')
    sec, ms = rest.replace('.', ',').split(',')
    return int(h) * 3600000 + int(m) * 60000 + int(sec) * 1000 + int(ms)

def format_time(ms):
    """milliseconds → HH:MM:SS,mmm"""
    ms = max(0, int(round(ms)))
    h, ms = divmod(ms, 3600000)
    m, ms = divmod(ms, 60000)
    s, ms = divmod(ms, 1000)
    return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"

def shift_ts_line(ts_line, offset_ms, scale):
    """Apply (scale, offset) transform to 'HH:MM:SS,mmm --> HH:MM:SS,mmm'."""
    parts = ts_line.split('-->')
    if len(parts) != 2:
        return ts_line
    start = parse_time(parts[0])
    end   = parse_time(parts[1])
    return f"{format_time(scale * start + offset_ms)} --> {format_time(scale * end + offset_ms)}"

def compute_sync(en_blocks, uk_blocks):
    """
    Find linear transform (offset_ms, scale) to align uk_blocks timing to en_blocks.
    Uses first and last subtitle as two calibration points.
    Returns (offset_ms, scale) or None if files are incompatible.
    """
    if len(en_blocks) < 2 or len(uk_blocks) < 2:
        return None

    # Reject if entry count differs too much — probably a different episode version
    count_ratio = len(uk_blocks) / len(en_blocks)
    if not (0.80 <= count_ratio <= 1.20):
        return None

    en_t0 = parse_time(en_blocks[0][1].split('-->')[0])
    en_t1 = parse_time(en_blocks[-1][1].split('-->')[0])
    uk_t0 = parse_time(uk_blocks[0][1].split('-->')[0])
    uk_t1 = parse_time(uk_blocks[-1][1].split('-->')[0])

    if uk_t1 == uk_t0:
        return None

    scale  = (en_t1 - en_t0) / (uk_t1 - uk_t0)
    offset = en_t0 - scale * uk_t0

    if not (1.0 - MAX_SYNC_DRIFT <= scale <= 1.0 + MAX_SYNC_DRIFT):
        return None  # fps difference too large — probably wrong video version

    return (offset, scale)

def apply_sync(uk_blocks, offset_ms, scale):
    return [(num, shift_ts_line(ts, offset_ms, scale), text)
            for num, ts, text in uk_blocks]

def verify_sync(en_blocks, synced_uk_blocks):
    """Check first 5 entries are within MAX_SYNC_GAP_MS after transform."""
    n = min(5, len(en_blocks), len(synced_uk_blocks))
    for i in range(n):
        en_t = parse_time(en_blocks[i][1].split('-->')[0])
        uk_t = parse_time(synced_uk_blocks[i][1].split('-->')[0])
        if abs(en_t - uk_t) > MAX_SYNC_GAP_MS:
            return False
    return True


# ═══════════════════════════════════════════════════════════════════
# SRT UTILS
# ═══════════════════════════════════════════════════════════════════

TAG_RE  = re.compile(r'<[^>]+>')
SKIP_RE = re.compile(r'^[\s♪♫\[\](){}|_\-\.#@!?,*•…♬\d]+$')

def parse_srt(filepath):
    with open(filepath, 'r', encoding='utf-8', errors='replace') as f:
        content = f.read()
    content = content.replace('\r\n', '\n').replace('\r', '\n')
    blocks = []
    for block in re.split(r'\n{2,}', content.strip()):
        lines = block.strip().split('\n')
        if len(lines) >= 2 and '-->' in lines[1]:
            blocks.append((lines[0].strip(), lines[1].strip(), '\n'.join(lines[2:]).strip()))
    return blocks

def _write_srt_blocks_atomically(blocks, output_path):
    output_path = Path(output_path)
    temporary_path = output_path.with_name(f".{output_path.name}.tmp")
    try:
        with open(temporary_path, 'w', encoding='utf-8') as f:
            for num, ts, text in blocks:
                f.write(f"{num}\n{ts}\n{text}\n\n")
            f.flush()
            os.fsync(f.fileno())
        os.replace(temporary_path, output_path)
    finally:
        temporary_path.unlink(missing_ok=True)


def write_srt(blocks, translations, output_path):
    translated_blocks = [
        (num, ts, translations.get(i + 1, orig))
        for i, (num, ts, orig) in enumerate(blocks)
    ]
    _write_srt_blocks_atomically(translated_blocks, output_path)

def strip_tags(text):
    return TAG_RE.sub('', text).strip()

def restore_tags(translated, original):
    m = re.match(r'^(<[a-zA-Z][^>]*>)(.*)(</[a-zA-Z]+>)$', original.strip(), re.DOTALL)
    return m.group(1) + translated + m.group(3) if m else translated

def needs_translation(text):
    clean = strip_tags(text)
    return bool(clean) and not SKIP_RE.match(clean)

def target_script_ratio(text):
    """Return the share of letters that use the configured target script."""
    alpha = re.findall(ALL_LETTERS_PATTERN, text)
    if not alpha:
        return 1.0
    target_letters = re.findall(TARGET_SCRIPT_PATTERN, text)
    return len(target_letters) / len(alpha)


def cyrillic_ratio(text):
    """Backward-compatible name for the original Ukrainian configuration."""
    return target_script_ratio(text)


def is_target_language(text):
    """
    Apply the optional target/rejected marker checks from the language settings.
    This works well for Ukrainian versus Russian. For language pairs that share
    the same alphabet, leave both marker patterns empty and use another quality
    check if detecting untranslated text matters.
    """
    target_letters = len(re.findall(TARGET_SCRIPT_PATTERN, text))
    if target_letters < 20:
        return True  # too little text to judge

    if REJECTED_LANGUAGE_MARKERS_PATTERN:
        rejected = len(re.findall(REJECTED_LANGUAGE_MARKERS_PATTERN, text))
        if rejected > target_letters * 0.02:
            return False

    if not TARGET_LANGUAGE_MARKERS_PATTERN:
        return True

    markers = len(re.findall(TARGET_LANGUAGE_MARKERS_PATTERN, text))
    return markers > target_letters * 0.03


def is_ukrainian(text):
    """Backward-compatible alias for the default target-language check."""
    return is_target_language(text)


# ═══════════════════════════════════════════════════════════════════
# OPENSUBTITLES.COM
# ═══════════════════════════════════════════════════════════════════

OS_BASE = "https://api.opensubtitles.com/api/v1"

def _os_headers():
    headers = {
        "Api-Key": OPENSUBTITLES_API_KEY,
        "Content-Type": "application/json",
        "User-Agent": OPENSUBTITLES_APP_NAME,
        "Accept": "application/json",
    }
    if OPENSUBTITLES_ACCESS_TOKEN:
        headers["Authorization"] = f"Bearer {OPENSUBTITLES_ACCESS_TOKEN}"
    return headers

def _os_get(path, params=None):
    url = f"{OS_BASE}{path}"
    if params:
        url += "?" + urllib.parse.urlencode(params)
    req = urllib.request.Request(url, headers=_os_headers())
    with urllib.request.urlopen(req, timeout=10) as r:
        return json.loads(r.read())

def _os_post(path, body):
    data = json.dumps(body).encode()
    req = urllib.request.Request(f"{OS_BASE}{path}", data=data,
                                  headers=_os_headers(), method="POST")
    with urllib.request.urlopen(req, timeout=10) as r:
        return json.loads(r.read())


def _os_login():
    """Resolve a download token from the environment or account credentials."""
    global OPENSUBTITLES_ACCESS_TOKEN, OS_BASE

    if OPENSUBTITLES_ACCESS_TOKEN:
        return True
    if not (OPENSUBTITLES_USERNAME and OPENSUBTITLES_PASSWORD):
        return False

    response = _os_post(
        "/login",
        {
            "username": OPENSUBTITLES_USERNAME,
            "password": OPENSUBTITLES_PASSWORD,
        },
    )
    token = response.get("token", "")
    if not token:
        return False

    OPENSUBTITLES_ACCESS_TOKEN = token
    base_url = response.get("base_url", "").strip().rstrip("/")
    if base_url:
        if not base_url.startswith(("http://", "https://")):
            base_url = f"https://{base_url}"
        if not base_url.endswith("/api/v1"):
            base_url = f"{base_url}/api/v1"
        OS_BASE = base_url
    return True

def parse_episode_info(stem):
    """'The Fosters S03E15 Minor Offenses.en' → ('The Fosters', 3, 15)"""
    source_suffix = rf'\.{re.escape(SOURCE_INPUT_SUFFIX)}$'
    stem = re.sub(source_suffix, '', stem)
    m = re.search(r'^(.+?)[\s._-]+[Ss](\d+)[Ee](\d+)', stem)
    if m:
        return m.group(1).strip(), int(m.group(2)), int(m.group(3))
    return None, None, None

def try_opensubtitles(en_path, uk_path, max_downloads=5):
    """
    Search OpenSubtitles for target-language subtitles, download, sync timing.
    Returns (success, download_attempts). Download attempts are counted
    conservatively because the API may consume quota before a later step fails.
    """
    if not (OPENSUBTITLES_API_KEY and OPENSUBTITLES_ACCESS_TOKEN):
        return False, 0

    title, season, episode = parse_episode_info(en_path.stem)
    if not title:
        return False, 0

    print(
        f"  🔍 OpenSubtitles: {title} S{season:02d}E{episode:02d} "
        f"[{TARGET_LANGUAGE_CODE}]...",
        end='',
        flush=True,
    )
    try:
        data = _os_get("/subtitles", {
            "query": title,
            "season_number": season,
            "episode_number": episode,
            "languages": TARGET_LANGUAGE_CODE,
        })
    except Exception as e:
        print(f" error ({e})")
        return False, 0

    results = data.get("data", [])
    if not results:
        print(" not found")
        return False, 0

    print(f" {len(results)} result(s)")
    en_blocks = parse_srt(en_path)
    download_attempts = 0

    for item in results[:max(0, min(5, max_downloads))]:
        files = item.get("attributes", {}).get("files", [])
        if not files:
            continue
        file_id = files[0].get("file_id")
        if not file_id:
            continue

        download_attempts += 1
        try:
            info = _os_post("/download", {"file_id": file_id})
            dl_url = info.get("link")
            if not dl_url:
                continue
            with urllib.request.urlopen(dl_url, timeout=15) as r:
                raw = r.read()
            if raw[:2] == b'\x1f\x8b':
                raw = gzip.decompress(raw)
            content = raw.decode('utf-8', errors='replace')
        except Exception as e:
            print(f"     ⚠ Download failed: {e}")
            time.sleep(DELAY_BETWEEN_CALLS)
            continue

        tmp = uk_path.with_suffix('.tmp.srt')
        tmp.write_text(content, encoding='utf-8')
        uk_blocks = parse_srt(str(tmp))
        tmp.unlink(missing_ok=True)

        if not uk_blocks:
            continue

        # Check that the downloaded file matches the configured target language.
        sample_text = ' '.join(t for _, _, t in uk_blocks[:50])
        if target_script_ratio(sample_text) < 0.5:
            print(f"     ⚠ Not enough {TARGET_LANGUAGE_NAME} script, skipping")
            continue
        if not is_target_language(sample_text):
            print(f"     ⚠ Not {TARGET_LANGUAGE_NAME}, skipping")
            continue

        # Compute and apply timing sync
        sync = compute_sync(en_blocks, uk_blocks)
        if sync is None:
            print(f"     ⚠ Timing incompatible ({len(uk_blocks)} vs {len(en_blocks)} entries)")
            time.sleep(DELAY_BETWEEN_CALLS)
            continue

        offset_ms, scale = sync
        synced = apply_sync(uk_blocks, offset_ms, scale)

        if not verify_sync(en_blocks, synced):
            print(f"     ⚠ Sync verification failed (gap > {MAX_SYNC_GAP_MS}ms)")
            time.sleep(DELAY_BETWEEN_CALLS)
            continue

        # Publish the complete file in one filesystem replacement.
        _write_srt_blocks_atomically(synced, uk_path)

        size_kb = uk_path.stat().st_size // 1024
        if abs(offset_ms) < 50 and abs(scale - 1.0) < 0.0005:
            sync_str = "already in sync"
        else:
            sync_str = f"offset={offset_ms/1000:+.2f}s, scale={scale:.4f}"
        print(f"  ✓ Downloaded from OpenSubtitles ({size_kb} KB, {sync_str})")
        return True, download_attempts

    print(f"     No usable subtitles found")
    return False, download_attempts


# ═══════════════════════════════════════════════════════════════════
# ARGOS TRANSLATE  (offline, unlimited)
# ═══════════════════════════════════════════════════════════════════

def argos_translate_batch(entries, timeout=30):
    """Translate offline via argostranslate. Returns {idx: text}. (timeout ignored — offline)"""
    tr = _init_argos()
    if tr is None:
        return {}
    results = {}
    for idx, text in entries:
        try:
            translated = tr.translate(text)
            if translated:
                results[idx] = translated
        except KeyboardInterrupt:
            raise
        except Exception:
            pass
    return results


# ═══════════════════════════════════════════════════════════════════
# GOOGLE / MYMEMORY / LIBRETRANSLATE  (online fallbacks)
# ═══════════════════════════════════════════════════════════════════

_TRANSLATOR_WORKER_CODE = """
import contextlib, json, sys
from deep_translator import GoogleTranslator, MyMemoryTranslator
payload = json.load(sys.stdin)
with contextlib.redirect_stdout(sys.stderr):
    translator = {"Google": GoogleTranslator, "MyMemory": MyMemoryTranslator}[payload["service"]](
        source=payload["source"], target=payload["target"])
    result = translator.translate_batch(payload["texts"])
json.dump(result, sys.stdout)
"""


def _call_translator(service, texts, source, target, timeout=30):
    """Run one chunk in a process that is killed and reaped at its deadline."""
    if service not in ('Google', 'MyMemory'):
        raise ValueError('Unsupported translator worker')
    result = subprocess.run(
        [sys.executable, '-c', _TRANSLATOR_WORKER_CODE],
        input=json.dumps(dict(service=service, source=source, target=target, texts=texts)),
        capture_output=True, text=True, timeout=timeout,
    )
    if result.returncode:
        raise RuntimeError(f'{service} worker failed (exit {result.returncode})')
    translated = json.loads(result.stdout)
    if (not isinstance(translated, list) or len(translated) != len(texts)
            or any(not isinstance(text, str) for text in translated)):
        raise ValueError('Invalid translator response')
    return translated


def _translate_in_chunks(service, idx_list, texts, source, target, chunk=20, timeout=30):
    """Keep completed chunks when a later chunk fails or times out."""
    results = {}
    for start in range(0, len(texts), chunk):
        chunk_texts = texts[start:start + chunk]
        chunk_idxs = idx_list[start:start + chunk]
        try:
            translated = _call_translator(service, chunk_texts, source, target, timeout=timeout)
            results.update((idx, text) for idx, text in zip(chunk_idxs, translated) if text)
        except KeyboardInterrupt:
            raise
        except subprocess.TimeoutExpired:
            print(f" ⚠ {service} chunk timeout ({timeout}s)", end='')
            break
        except Exception:
            break
    return results


def _mymemory_translate(idx_list, texts, timeout=20):
    """MyMemory fallback — free, no API key."""
    try:
        from deep_translator import MyMemoryTranslator
    except ImportError:
        return {}
    # Try the configured short codes, locales, and language names.
    language_pairs = [
        (SOURCE_LANGUAGE_CODE, TARGET_LANGUAGE_CODE),
        (SOURCE_LANGUAGE_LOCALE, TARGET_LANGUAGE_LOCALE),
        (SOURCE_LANGUAGE_NAME.lower(), TARGET_LANGUAGE_NAME.lower()),
    ]
    for src, tgt in language_pairs:
        try:
            # Select a supported language pair before making network requests.
            MyMemoryTranslator(source=src, target=tgt)
        except Exception:
            continue
        return _translate_in_chunks('MyMemory', idx_list, texts, src, tgt, chunk=5, timeout=timeout)
    return {}



def _libretranslate_batch(idx_list, texts, timeout=10):
    """
    POST to local LibreTranslate server (LIBRETRANSLATE_URL).
    Returns {} if URL not configured or server unreachable.
    """
    if not LIBRETRANSLATE_URL:
        return {}
    base = LIBRETRANSLATE_URL.rstrip('/')
    results = {}
    try:
        for i, text in enumerate(texts):
            url  = f"{base}/translate"
            body = json.dumps({
                "q": text,
                "source": SOURCE_LANGUAGE_CODE,
                "target": TARGET_LANGUAGE_CODE,
                "format": "text", "api_key": "",
            }).encode()
            req = urllib.request.Request(
                url, data=body,
                headers={"Content-Type": "application/json", "Accept": "application/json"},
                method="POST",
            )
            try:
                with urllib.request.urlopen(req, timeout=timeout) as r:
                    t = json.loads(r.read()).get("translatedText", "")
                if t:
                    results[idx_list[i]] = t
            except KeyboardInterrupt:
                raise
            except Exception:
                break   # server error — return what we have so far
    except KeyboardInterrupt:
        raise
    except Exception:
        pass
    return results


def google_translate_batch(entries, timeout=30):
    """
    Translate via Google Translate. Returns {idx: text} for successful entries.
    timeout: seconds per chunk, enforced by a disposable worker process.
    """
    try:
        from deep_translator import GoogleTranslator
    except ImportError:
        return {}

    idx_list = [idx for idx, _ in entries]
    texts    = [text for _, text in entries]

    try:
        results = _translate_in_chunks(
            'Google', idx_list, texts, SOURCE_LANGUAGE_CODE, TARGET_LANGUAGE_CODE,
            chunk=20, timeout=timeout,
        )
        if results:
            return results
    except KeyboardInterrupt:
        raise
    except Exception as e:
        print(f" ⚠ Google error: {e}", end='')

    return {}


# ═══════════════════════════════════════════════════════════════════
# CLAUDE FALLBACK
# ═══════════════════════════════════════════════════════════════════

def claude_translate_batch(entries, min_tokens_pct: float = 0, timeout=30):
    """Translate via Claude CLI. Returns {idx: text}."""
    # Check token budget before calling
    pct = get_free_tokens_pct()
    tok_str = token_status_str(pct)
    print(f" [{tok_str}]", end='', flush=True)

    if not claude_allowed(min_tokens_pct):
        print(f" ✗ skipped (< {min_tokens_pct:.0f}% tokens)", end='')
        return {}

    lines  = [f"{idx}:{text.replace(chr(10), '§')}" for idx, text in entries]
    prompt = (
        f"Translate subtitles from {SOURCE_LANGUAGE_NAME} "
        f"to {TARGET_LANGUAGE_NAME}.\n"
        "Rules: reply ONLY with lines in format NUMBER:translated_text — no preamble, no markdown, no extra text.\n"
        "Example input:  1:Hello world\n"
        f"Example output format: 1:<translated {TARGET_LANGUAGE_NAME} text>\n\n"
        + '\n'.join(lines)
    )
    valid_idxs = {idx for idx, _ in entries}

    try:
        with tempfile.TemporaryDirectory(prefix='subtitle_claude_') as workdir:
            result = subprocess.run(
                ['claude', '--print', '--output-format', 'text', '--model', CLAUDE_MODEL,
                 '--tools', '', '--strict-mcp-config', '--mcp-config', '{"mcpServers":{}}',
                 '--disable-slash-commands', '--setting-sources', '', '--no-session-persistence'],
                input=prompt, capture_output=True, text=True, timeout=timeout, cwd=workdir,
            )
        if result.returncode != 0:
            print(f" ✗ Claude exited with code {result.returncode}", end='')
            return {}
        translations = {}
        for line in result.stdout.strip().split('\n'):
            match = re.match(r'^(\d+)[:.]\s*(.*)', line.strip())
            if match:
                idx = int(match.group(1))
                text = match.group(2).strip().replace(' § ', '\n').replace('§', '\n')
                if idx in valid_idxs and text:
                    translations[idx] = text
        return translations
    except subprocess.TimeoutExpired:
        print(f" ⚠ Claude timeout ({timeout}s)", end='')
    except OSError:
        print(" ✗ Claude could not be started", end='')
    return {}


# ═══════════════════════════════════════════════════════════════════
# TRANSLATE FILE  (config-driven CoR pipeline)
# ═══════════════════════════════════════════════════════════════════

def translate_file(en_path, uk_path, min_tokens_pct: float = 0):
    blocks = parse_srt(en_path)
    if not blocks:
        print(f"  ✗ Cannot parse")
        return False, {}

    total = len(blocks)
    entry_clean = [
        (i + 1, strip_tags(orig).replace('\n', ' ').strip(), orig)
        for i, (_, _, orig) in enumerate(blocks)
    ]

    # Deduplicate: only translate unique phrases
    text_cache    = {}   # clean_text → translated
    unique_needed = []
    for idx, clean, orig in entry_clean:
        if needs_translation(orig) and clean not in text_cache:
            text_cache[clean] = None
            unique_needed.append((idx, clean))

    translatable_count = len(unique_needed)
    skipped = total - translatable_count
    print(f"  Phrases: {translatable_count} unique to translate, {skipped} skipped")

    stats: dict[str, int] = {}

    # ── Resolve chain from TRANSLATOR_CHAIN config ─────────────────
    has_argos = _init_argos() is not None
    resolved  = [s for s in _parse_chain_config(TRANSLATOR_CHAIN)
                 if _check_service(s['name'], has_argos)[0]]

    if not resolved:
        print("  ✗ No translation service is available!")
        return False, stats

    # ── fn registry (uniform signature: fn(entries, timeout) → dict) ─
    def _google_fn(b, timeout=30):
        return google_translate_batch(b, timeout=timeout)
    def _argos_fn(b, timeout=30):
        return argos_translate_batch(b)
    def _mymem_fn(b, timeout=30):
        return _mymemory_translate(
            [i for i, _ in b], [t for _, t in b], timeout=timeout
        )
    def _libre_fn(b, timeout=30):
        return _libretranslate_batch(
            [i for i, _ in b], [t for _, t in b], timeout=timeout
        )
    def _claude_fn(b, timeout=30):
        return claude_translate_batch(b, min_tokens_pct, timeout=timeout)

    _FN_MAP = {
        'Google':   _google_fn,
        'Argos':    _argos_fn,
        'MyMemory': _mymem_fn,
        'LibreTranslate':   _libre_fn,
        'Claude':   _claude_fn,
    }

    # ── Phase 1: primary batch pass (first resolved service) ───────
    needs_retry  = []
    primary      = resolved[0]
    primary_name = primary['name']
    primary_fn   = _FN_MAP[primary_name]

    if unique_needed:
        total_batches = (len(unique_needed) + BATCH_SIZE - 1) // BATCH_SIZE
        for start in range(0, len(unique_needed), BATCH_SIZE):
            batch = unique_needed[start:start + BATCH_SIZE]
            bn    = start // BATCH_SIZE + 1
            print(f"  [{primary_name} {bn}/{total_batches}] {len(batch)} entries...",
                  end='', flush=True)

            result     = primary_fn(batch, timeout=30)
            failed_now = []
            for idx, clean in batch:
                tr = result.get(idx)
                if tr and target_script_ratio(tr) >= MIN_QUALITY:
                    text_cache[clean] = tr
                    stats[primary_name] = stats.get(primary_name, 0) + 1
                else:
                    failed_now.append((idx, clean))
                    needs_retry.append((idx, clean))

            ok_count = len(batch) - len(failed_now)
            suffix   = f" ({len(failed_now)} → retry)" if failed_now else ""
            print(f" ✓ {ok_count}/{len(batch)}{suffix}")

            if start + BATCH_SIZE < len(unique_needed):
                time.sleep(DELAY_BETWEEN_CALLS)

    # ── Phase 2: CoR for failed entries ────────────────────────────

    class TranslatorStep:
        def __init__(self, name, fn, fallbacks=0, timeout_inc=0, timeout_base=30):
            self.name         = name
            self.fn           = fn
            self.fallbacks    = fallbacks    # retries after first attempt (0 = no retry)
            self.timeout_inc  = timeout_inc  # added to timeout per retry
            self.timeout_base = timeout_base
            self.next: 'TranslatorStep | None' = None

        def set_next(self, step):
            self.next = step
            return step

        def execute(self, entries, attempts_already_used=0):
            remaining     = entries
            total_att     = self.fallbacks + 1
            first_attempt = min(attempts_already_used, total_att)
            for attempt in range(first_attempt, total_att):
                if not remaining:
                    return
                timeout = self.timeout_base + attempt * self.timeout_inc
                label   = self.name if attempt == 0 else f"{self.name}↺{attempt}"
                print(f"  [{label}] {len(remaining)} entries...", end='', flush=True)
                result    = self.fn(remaining, timeout=timeout)
                still_bad = []
                for idx, clean in remaining:
                    tr = result.get(idx)
                    if tr and target_script_ratio(tr) >= MIN_QUALITY:
                        text_cache[clean] = tr
                        stats[self.name] = stats.get(self.name, 0) + 1
                    else:
                        still_bad.append((idx, clean))
                got      = len(remaining) - len(still_bad)
                has_more = attempt < total_att - 1 or self.next
                suffix   = f" ({len(still_bad)} → {'retry' if attempt < total_att - 1 else 'next'})" \
                           if still_bad and has_more else ""
                print(f" ✓ {got}/{len(remaining)}{suffix}")
                remaining = still_bad

            if remaining and self.next:
                self.next.execute(remaining)

    # Build linked list from resolved chain (same order as config)
    steps = [
        TranslatorStep(
            name         = s['name'],
            fn           = _FN_MAP[s['name']],
            fallbacks    = s['fallbacks'],
            timeout_inc  = s['timeout_inc'],
        )
        for s in resolved
    ]
    for i in range(len(steps) - 1):
        steps[i].set_next(steps[i + 1])

    if needs_retry and steps:
        steps[0].execute(needs_retry, attempts_already_used=1)

    # ── Build final translations ────────────────────────────────────
    translations = {}
    for idx, clean, orig in entry_clean:
        if not needs_translation(orig):
            translations[idx] = orig
        elif text_cache.get(clean) is not None:
            translations[idx] = restore_tags(text_cache[clean], orig)

    translatable = sum(1 for _, c, o in entry_clean if needs_translation(o))
    translated   = sum(1 for _, c, o in entry_clean
                       if needs_translation(o) and text_cache.get(c) is not None)
    coverage     = translated / translatable * 100 if translatable else 100

    write_srt(blocks, translations, uk_path)

    if coverage < 60:
        uk_path.unlink(missing_ok=True)
        print(f"  ✗ FAILED: {coverage:.0f}% - file removed; it will be retried on the next run")
        return False, stats

    size_kb = uk_path.stat().st_size // 1024

    def _fmt_pct(n):
        pct = n / translatable_count * 100 if translatable_count else 0
        return f"{pct:.0f}%" if pct >= 1 else f"{pct:.2f}%"

    order     = ['Google', 'Argos', 'MyMemory', 'LibreTranslate', 'Claude']
    breakdown = ', '.join(
        f"{name}: {_fmt_pct(stats[name])}"
        for name in order if stats.get(name, 0) > 0
    )
    print(f"  ✓ {uk_path.name} ({size_kb} KB, {coverage:.0f}% — {breakdown})")
    return True, stats


# ═══════════════════════════════════════════════════════════════════
# CLAUDE TOKEN TRACKING
# ═══════════════════════════════════════════════════════════════════

def get_free_tokens_pct(debug: bool = False) -> 'float | None':
    """Read only an explicitly supplied, fresh usage estimate; never probe the CLI."""
    global _token_pct
    _token_pct = None
    usage_path = os.environ.get('CLAUDE_USAGE_FILE', '')
    if usage_path:
        try:
            with Path(usage_path).expanduser().open(encoding='utf-8') as source:
                age = time.time() - os.fstat(source.fileno()).st_mtime
                data = json.loads(source.read(4097))
            value = data.get('remaining_percent') if isinstance(data, dict) else None
            if (0 <= age <= 300 and isinstance(value, (int, float))
                    and not isinstance(value, bool) and math.isfinite(value) and 0 <= value <= 100):
                _token_pct = float(value)
        except (OSError, ValueError, OverflowError):
            pass
    if debug:
        print('  Claude usage file: ' + ('fresh estimate' if _token_pct is not None
                                         else 'missing, stale, or invalid; usage unknown'))
    return _token_pct


def token_status_str(pct: 'float | None') -> str:
    if pct is None:
        return "tokens: unknown"
    bar = '█' * int(pct / 10) + '░' * (10 - int(pct / 10))
    return f"tokens: {bar} {pct:.0f}% free"

def claude_allowed(min_pct: float) -> bool:
    """Allow unknown usage only when the caller explicitly disables the guard."""
    if _token_pct is None:
        return min_pct == 0
    return _token_pct >= min_pct


# ═══════════════════════════════════════════════════════════════════
# LOGGING
# ═══════════════════════════════════════════════════════════════════

LOG_HEADER    = "| Date and time | Execution result |"
LOG_SEPARATOR = "|---|---|"
LOG_OS_TAG    = re.compile(r'\[OS:(\d+)\]')   # machine-readable OS download count


class _Tee:
    """Write to both the original stdout and a log file simultaneously."""
    def __init__(self, log_file: Path):
        self._orig  = sys.stdout
        self._fh    = open(log_file, 'a', encoding='utf-8', buffering=1)
        sys.stdout  = self

    def write(self, data):
        self._orig.write(data)
        self._fh.write(data)

    def flush(self):
        self._orig.flush()
        self._fh.flush()

    def close(self):
        sys.stdout = self._orig
        self._fh.close()

    # Needed so print(..., file=sys.stdout) works
    def fileno(self):
        return self._orig.fileno()


_tee: '_Tee | None' = None   # module-level handle so atexit can close it


class OpenSubtitlesDownloadCheckpoint:
    """Track which download attempts have already been persisted to Log.md."""

    def __init__(self):
        self._logged_total = 0

    def take_delta(self, current_total):
        delta = max(0, current_total - self._logged_total)
        self._logged_total = max(self._logged_total, current_total)
        return delta


def count_today_os_downloads(log_path):
    """Read Log.md and sum OS downloads recorded today."""
    if not log_path or not log_path.exists():
        return 0
    today = datetime.now().strftime("%Y-%m-%d")
    total = 0
    try:
        for line in log_path.read_text(encoding='utf-8').splitlines():
            if today in line:
                m = LOG_OS_TAG.search(line)
                if m:
                    total += int(m.group(1))
    except Exception:
        pass
    return total

def write_log(log_path, entry, os_downloaded):
    """Prepend a new row to Log.md (creates file if missing)."""
    if not log_path:
        return
    now     = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
    tag     = f" [OS:{os_downloaded}]" if os_downloaded else ""
    new_row = f"| {now} | {entry}{tag} |"
    try:
        if log_path.exists():
            lines = log_path.read_text(encoding='utf-8').splitlines()
            # Insert after separator line
            for i, line in enumerate(lines):
                if line.strip().startswith('|---'):
                    lines.insert(i + 1, new_row)
                    break
            else:
                lines = [LOG_HEADER, LOG_SEPARATOR, new_row] + lines
            log_path.write_text('\n'.join(lines) + '\n', encoding='utf-8')
        else:
            log_path.write_text(
                f"{LOG_HEADER}\n{LOG_SEPARATOR}\n{new_row}\n",
                encoding='utf-8'
            )
    except Exception as e:
        print(f"⚠ Could not write log: {e}")


# ═══════════════════════════════════════════════════════════════════
# FILE DISCOVERY
# ═══════════════════════════════════════════════════════════════════

def find_srt_files(source_dir, output_dir):
    tasks         = []
    skipped       = 0
    seen_outputs  = set()
    source_suffix = f'.{SOURCE_INPUT_SUFFIX}.srt'
    target_suffix = f'.{TARGET_OUTPUT_SUFFIX}.srt'

    for srt_file in sorted(source_dir.rglob('*.srt')):
        name = srt_file.name
        stem = srt_file.stem
        rel  = srt_file.parent.relative_to(source_dir)
        out  = output_dir / rel

        if name.endswith(target_suffix):
            continue

        if name.endswith(source_suffix):
            base = stem[:-(len(SOURCE_INPUT_SUFFIX) + 1)]
        else:
            base = stem

        target_out = out / f"{base}.{TARGET_OUTPUT_SUFFIX}.srt"
        source_out = out / f"{base}.{SOURCE_INPUT_SUFFIX}.srt"

        if target_out.exists():
            skipped += 1
            continue

        if target_out in seen_outputs:
            continue  # .srt and language-suffixed source point to the same output

        seen_outputs.add(target_out)
        tasks.append((srt_file, target_out, source_out))

    return tasks, skipped, skipped + len(tasks)


# ═══════════════════════════════════════════════════════════════════
# MAIN
# ═══════════════════════════════════════════════════════════════════

def main():
    # ── Parse CLI args ─────────────────────────────────────────────
    # Usage: translate_subtitles.py [max_files] [--min-tokens N]
    max_files      = MAX_FILES_PER_RUN
    min_tokens_pct = MIN_TOKENS_PCT

    args = sys.argv[1:]
    debug_tokens = False
    i = 0
    while i < len(args):
        if args[i] == '--min-tokens' and i + 1 < len(args):
            try:
                min_tokens_pct = max(0.0, min(100.0, float(args[i + 1])))
            except ValueError:
                print(f"⚠ Invalid --min-tokens value, using default {MIN_TOKENS_PCT}%")
            i += 2
        elif args[i] == '--debug-tokens':
            debug_tokens = True
            i += 1
        else:
            try:
                max_files = int(args[i])
            except ValueError:
                print(f"⚠ Unknown argument '{args[i]}', ignored")
            i += 1

    script_dir  = Path(__file__).parent.resolve()
    source_dir  = script_dir / "subtitles_source"
    output_dir  = script_dir / "subtitles_output"
    log_path    = script_dir / "Log.md"
    console_log = script_dir / "Log_console.txt"
    output_dir.mkdir(exist_ok=True)

    # Tee: duplicate all print() output to Log_console.txt
    global _tee
    _tee = _Tee(console_log)
    atexit.register(_tee.close)

    now_str = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
    print(f"\n{'='*60}")
    print(f"=== Run started at {now_str} ===")

    has_argos = _init_argos() is not None

    os_authenticated = False
    os_auth_error = ""
    if OPENSUBTITLES_API_KEY:
        try:
            os_authenticated = _os_login()
        except Exception as error:
            os_auth_error = str(error)

    os_used_today   = count_today_os_downloads(log_path)
    os_remaining    = OS_DAILY_LIMIT - os_used_today
    start_token_pct = get_free_tokens_pct(debug=debug_tokens)
    if debug_tokens:
        print()

    # ── Chain analysis ──────────────────────────────────────────────
    resolved_chain, chain_desc = _build_chain_description(has_argos)

    print(f"📁 Source:  {source_dir}")
    print(f"📁 Output:  {output_dir}")
    print(f"🔧 Model: {CLAUDE_MODEL} | Batch: {BATCH_SIZE} | Max files: {max_files}")
    if os_authenticated:
        print(f"🌐 OpenSubtitles: ✓ | {os_remaining}/{OS_DAILY_LIMIT} downloads left today")
    elif OPENSUBTITLES_API_KEY and os_auth_error:
        print(f"🌐 OpenSubtitles: ✗ authentication failed ({os_auth_error})")
    elif OPENSUBTITLES_API_KEY:
        print("🌐 OpenSubtitles: ✗ access token or account credentials are not set")
    else:
        print("🌐 OpenSubtitles: ✗ API key is not set")
    print(f"🔋 {token_status_str(start_token_pct)} | Claude threshold: {min_tokens_pct:.0f}%")
    print(chain_desc)
    print()

    tasks, skipped, total_count = find_srt_files(source_dir, output_dir)

    if not tasks:
        print("✅ Nothing to do — all files already translated.")
        return

    total_pending = len(tasks)
    if total_pending > max_files:
        print(f"⚠  {total_pending} pending — limiting to {max_files} this run.")
        tasks = tasks[:max_files]

    print(f"Total: {total_count} | Done: {skipped} | Pending: {total_pending} | This run: {len(tasks)}\n")
    for t in tasks:
        print(f"  → {t[1].relative_to(output_dir)}")
    print()

    success     = 0
    os_this_run = 0
    os_log_checkpoint = OpenSubtitlesDownloadCheckpoint()
    total_stats: dict[str, int] = {}   # per-translator phrase counts (this run)
    failed_names = []

    def _tok(pct):
        return f"{pct:.0f}%" if pct is not None else "?"

    def _flush_log(interrupted: bool = False):
        """Write current progress to Log.md (called after each file and on exit)."""
        end_pct = get_free_tokens_pct()
        status  = "✗ interrupted" if interrupted else f"✓ {success}/{len(tasks)}"
        parts   = [status]
        if os_this_run:
            parts.append(f"OS×{os_this_run}")
        for svc in ('Google', 'Argos', 'MyMemory', 'LibreTranslate', 'Claude'):
            n = total_stats.get(svc, 0)
            if n:
                parts.append(f"{svc}×{n}")
        if failed_names:
            parts.append(f"✗ {len(failed_names)} errors")
        os_left = OS_DAILY_LIMIT - os_used_today - os_this_run
        if OPENSUBTITLES_API_KEY:
            parts.append(f"OS:{os_left}/{OS_DAILY_LIMIT}")
        parts.append(f"tokens: {_tok(start_token_pct)}→{_tok(end_pct)}")
        write_log(
            log_path,
            " | ".join(parts),
            os_log_checkpoint.take_delta(os_this_run),
        )

    try:
        for i, (orig_path, target_path, source_path) in enumerate(tasks, 1):
            label = f"[{skipped + i}/{total_count}]"
            target_path.parent.mkdir(parents=True, exist_ok=True)

            # Prepare a source-language-suffixed SRT file.
            source_suffix = f'.{SOURCE_INPUT_SUFFIX}.srt'
            if not orig_path.name.endswith(source_suffix):
                print(f"{label} {orig_path.name}")
                if not source_path.exists():
                    shutil.copy2(orig_path, source_path)
                source_file = source_path
            else:
                source_file = orig_path
                print(f"{label} {source_file.name}")

            # ── Priority 1: OpenSubtitles ──
            os_budget = max(0, OS_DAILY_LIMIT - os_used_today - os_this_run)
            if os_budget and os_authenticated:
                os_success, os_downloads = try_opensubtitles(
                    source_file,
                    target_path,
                    max_downloads=os_budget,
                )
                os_this_run += os_downloads
            elif not os_authenticated:
                os_success = False
            else:
                os_success = False
                print("  OpenSubtitles: local daily download limit reached")

            if os_success:
                success     += 1
                print()
            else:
                # ── Priority 2 & 3: translator chain ──
                result, run_stats = translate_file(
                    source_file,
                    target_path,
                    min_tokens_pct,
                )
                for svc, n in run_stats.items():
                    total_stats[svc] = total_stats.get(svc, 0) + n
                if result:
                    success += 1
                else:
                    failed_names.append(orig_path.stem)
                print()

            _flush_log()   # update Log.md after every file

    except KeyboardInterrupt:
        print("\n⛔ Interrupted by user")
        _flush_log(interrupted=True)
        return

    # ── Final summary ─────────────────────────────────────────────
    end_token_pct = get_free_tokens_pct()
    print("=" * 60)
    print(f"✅ Done: {success}/{len(tasks)} files")
    if failed_names:
        print(f"⚠  Failed ({len(failed_names)}): {', '.join(failed_names)}")
    remaining_files = total_pending - len(tasks)
    if remaining_files > 0:
        print(f"ℹ  {remaining_files} remaining — run again to continue")
    print(f"🔋 {token_status_str(end_token_pct)}")


if __name__ == '__main__':
    main()
