"""
Shared, dependency-free core for the manga downloaders.

Provides:
  - a polite HTTP fetcher (single sequential connection, fixed delays,
    exponential backoff, and a hard stop if the server keeps returning 429),
  - a resumable chapter download engine (page_NNN naming, self-healing
    reconciliation of existing files, download_state.json),
  - a Source adapter interface that site modules implement.

A Source adapter provides:
  name            str   identifier, e.g. "mangaread"
  site            str   origin used for links, e.g. "https://www.mangaread.org"
  output_dir(url) Path  archive directory for a manga URL
  chapter_dirname(ch_id) str  e.g. "chapter_12" or "chapter_71_1"
  get_chapters(manga_url) -> list[(ch_id, ch_url, label)]  in READING order
  get_pages(chapter_url) -> list[str]  page image URLs in reading order
"""

import json
import os
import re
import sys
import time
import urllib.error
import urllib.request
from pathlib import Path

USER_AGENT = (
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
    "(KHTML, like Gecko) Chrome/124.0 Safari/124.0"
)

# Politeness knobs (seconds). Deliberately slower than any human would need.
DELAY_CHAPTER_PAGE = 2.0   # before each chapter page fetch
DELAY_IMAGE = 0.75         # before each image download
PAUSE_EVERY_N_IMAGES = 25  # take a longer breather every N images
PAUSE_LONG = 5.0
MAX_RETRIES = 5
RETRY_BACKOFF_BASE = 8.0
ABORT_ON_429_RETRIES = 3   # consecutive 429s before giving up entirely

IMG_EXTS = {".webp", ".jpg", ".jpeg", ".png", ".gif", ".avif", ".bmp"}


def log(msg: str) -> None:
    print(time.strftime("[%H:%M:%S] ") + msg, flush=True)


def fetch(url: str, referer: str | None = None, timeout: float = 30.0) -> bytes | None:
    """GET a URL politely. Returns bytes, or None after repeated failure.

    Backs off on 5xx/network errors; on 429 it backs off exponentially and
    aborts the whole run if the rate limiting persists.
    """
    headers = {"User-Agent": USER_AGENT, "Accept": "*/*"}
    if referer:
        headers["Referer"] = referer
    for attempt in range(1, MAX_RETRIES + 1):
        try:
            req = urllib.request.Request(url, headers=headers)
            with urllib.request.urlopen(req, timeout=timeout) as resp:
                return resp.read()
        except urllib.error.HTTPError as e:
            if e.code == 429:
                if attempt >= ABORT_ON_429_RETRIES:
                    log("Server keeps returning 429 — stopping out of respect.")
                    sys.exit(2)
                wait = RETRY_BACKOFF_BASE * (2 ** attempt)
                log(f"  429 rate-limited. Waiting {wait:.0f}s (attempt {attempt}/{MAX_RETRIES})...")
                time.sleep(wait)
                continue
            if 500 <= e.code < 600:
                wait = RETRY_BACKOFF_BASE * attempt
                log(f"  HTTP {e.code}. Retrying in {wait:.0f}s (attempt {attempt}/{MAX_RETRIES})...")
                time.sleep(wait)
                continue
            log(f"  HTTP {e.code} for {url} — giving up on this URL.")
            return None
        except (urllib.error.URLError, TimeoutError, OSError) as e:
            wait = RETRY_BACKOFF_BASE * attempt
            log(f"  Network error ({e}). Retrying in {wait:.0f}s (attempt {attempt}/{MAX_RETRIES})...")
            time.sleep(wait)
            continue
    log(f"  Failed after {MAX_RETRIES} attempts: {url}")
    return None


def fetch_html(url: str, referer: str | None = None) -> str | None:
    data = fetch(url, referer=referer)
    return data.decode("utf-8", errors="replace") if data is not None else None


def load_state(path: Path) -> dict:
    if path.exists():
        try:
            return json.loads(path.read_text())
        except Exception:
            return {}
    return {}


def save_state(path: Path, state: dict) -> None:
    tmp = path.with_suffix(".tmp")
    tmp.write_text(json.dumps(state, indent=2))
    tmp.replace(path)


def download_chapter(dest_dir: Path, urls: list[str], referer: str,
                     state: dict, state_file: Path, label: str) -> bool:
    """Download one chapter's pages as page_001.ext, page_002.ext, ...

    Self-healing: existing files are matched by the URL basename (old naming)
    or the final page_NNN name and renamed into place; missing pages are
    downloaded; stray image files that belong to neither are removed.
    Returns True when the chapter is complete.
    """
    dest_dir.mkdir(parents=True, exist_ok=True)

    existing = {p.name: p for p in dest_dir.iterdir()
                if p.is_file() and not p.name.endswith(".part")}
    assigned = set()
    plan = []
    for i, url in enumerate(urls, 1):
        base = os.path.basename(url)
        dest_name = f"page_{i:03d}{os.path.splitext(base)[1]}"
        src = None
        for cand_name in (dest_name, base):
            cand = existing.get(cand_name)
            if cand is not None and id(cand) not in assigned:
                src = cand
                break
        if src is not None:
            assigned.add(id(src))
        plan.append((dest_name, src, url))

    removed = 0
    for name, p in existing.items():
        if id(p) not in assigned and p.suffix.lower() in IMG_EXTS:
            p.unlink()
            removed += 1
    if removed:
        log(f"  removed {removed} stray image files from chapter dir.")

    failed = 0
    for i, (dest_name, src, url) in enumerate(plan, 1):
        dest = dest_dir / dest_name
        if src is not None and src != dest:
            src.rename(dest)
        if dest.exists() and dest.stat().st_size > 0:
            continue
        time.sleep(DELAY_IMAGE)
        data = fetch(url, referer=referer, timeout=60.0)
        if data is None:
            failed += 1
            continue
        tmp = dest.with_suffix(dest.suffix + ".part")
        tmp.write_bytes(data)
        tmp.replace(dest)
        if i % PAUSE_EVERY_N_IMAGES == 0 and i < len(plan):
            time.sleep(PAUSE_LONG)

    if failed:
        log(f"{label}: {failed}/{len(urls)} pages failed — retry next run.")
        return False

    state.setdefault("done", {})[label] = {
        "url": referer,
        "images": len(urls),
        "finished": time.strftime("%Y-%m-%dT%H:%M:%S"),
    }
    save_state(state_file, state)
    log(f"{label}: done ({len(urls)} pages).")
    return True


def run(source, manga_url: str) -> None:
    """Download every chapter of a manga through a Source adapter."""
    out = source.output_dir(manga_url)
    out.mkdir(parents=True, exist_ok=True)
    state_file = out / "download_state.json"
    state = load_state(state_file)

    log(f"Fetching manga page: {manga_url}")
    time.sleep(DELAY_CHAPTER_PAGE)
    chapters = source.get_chapters(manga_url)
    if not chapters:
        sys.exit("No chapter links found — site layout may have changed.")
    log(f"Found {len(chapters)} chapters (in reading order).")
    log(f"Output dir: {out.resolve()}")

    ok = incomplete = 0
    for i, (ch_id, ch_url, label) in enumerate(chapters, 1):
        log(f"\n=== {label} ({i}/{len(chapters)}) ===")
        time.sleep(DELAY_CHAPTER_PAGE)
        pages = source.get_pages(ch_url)
        if not pages:
            log(f"{label}: could not determine page list — retry next run.")
            incomplete += 1
            continue
        dest = out / source.chapter_dirname(ch_id)
        if download_chapter(dest, pages, ch_url, state, state_file, label):
            ok += 1
        else:
            incomplete += 1

    log(f"\nAll chapters processed: {ok} ok, {incomplete} incomplete.")
    if incomplete:
        log("Re-run this script to retry the incomplete chapters.")
        sys.exit(1)
    log("Next: run generate_html.py on this directory to (re)build the reader pages.")


def slug_from_url(url: str) -> str:
    """Trailing non-empty path segment, minus a trailing slash."""
    m = re.search(r"([^/]+?)/?$", url)
    return m.group(1) if m else "manga"
