"""
Source adapter for mangaread.org (WordPress "Madara" theme, wp-manga reader).

Manga pages list their chapters as <li class="wp-manga-chapter"> items,
newest first. Chapter readers embed their pages as
    <img ... class="wp-manga-chapter-img" src="...">  (or data-src)
inside <div class="reading-content"> — in display order, which is the
canonical page set and reading order for the chapter.

Chapter slugs can carry fix suffixes (e.g. chapter-71-1), so chapter ids are
the raw slugs; directory names turn '-' into '_'.
"""

import re
import sys

from . import core


class MangaReadSource:
    name = "mangaread"
    site = "https://www.mangaread.org"

    # --- adapter interface -------------------------------------------------

    def output_dir(self, manga_url: str) -> "core.Path":
        return core.Path(core.slug_from_url(manga_url))

    def chapter_dirname(self, ch_id: str) -> str:
        # ch_id is the URL slug, e.g. "chapter-71-1" -> "chapter_71_1"
        return ch_id.replace("chapter-", "chapter_").replace("-", "_")

    def get_chapters(self, manga_url: str) -> list[tuple[str, str, str]]:
        """-> [(ch_id, ch_url, label)] in reading order (oldest first)."""
        html = core.fetch_html(manga_url)
        if html is None:
            sys.exit("Could not fetch manga page.")
        items = re.findall(
            r'<li class="wp-manga-chapter[^"]*">\s*<a href="([^"]+)">(.*?)</a>',
            html, re.S)
        if not items:
            sys.exit("No chapter list found — site layout may have changed.")
        # The list is newest-first; reverse to reading order.
        chapters = []
        seen = set()
        for url, raw_label in reversed(items):
            label = " ".join(raw_label.split())
            ch_id = core.slug_from_url(url)
            if ch_id in seen:
                continue
            seen.add(ch_id)
            chapters.append((ch_id, url, label or f"chapter {ch_id}"))
        return chapters

    def get_pages(self, chapter_url: str) -> list[str]:
        html = core.fetch_html(chapter_url)
        if html is None:
            return []
        # Scope to the reader container so ads/previews elsewhere on the
        # page can never leak in.
        start = html.find('<div class="reading-content"')
        if start == -1:
            start = html.find('class="read-container"')
        if start == -1:
            core.log("  reader container not found — layout may have changed.")
            return []
        end = html.find('<select class="select-page', start)
        if end == -1:
            end = len(html)
        scope = html[start:end]
        urls = []
        for m in re.finditer(r'<img[^>]*wp-manga-chapter-img[^>]*>', scope):
            tag = m.group(0)
            src = re.search(r'\bdata-src\s*=\s*"([^"]+)"', tag) or \
                  re.search(r'\bsrc\s*=\s*"([^"]+)"', tag)
            if not src:
                continue
            url = src.group(1).strip()
            if url and not url.startswith("data:"):
                urls.append(url)
        return urls
