"""
Scribd document scraper service. Runs multiple strategies to extract a document
from a Scribd URL — returns a file path or raises an error.
"""

import json
import re
import tempfile
from pathlib import Path
from urllib.parse import urlparse

import requests
from bs4 import BeautifulSoup
from django.conf import settings

SESSION = requests.Session()
SESSION.headers.update({
    "User-Agent": (
        "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
        "AppleWebKit/537.36 (KHTML, like Gecko) "
        "Chrome/131.0.0.0 Safari/537.36"
    ),
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "en-US,en;q=0.9,id;q=0.8",
})


def extract_doc_id(url: str) -> str | None:
    m = re.search(r"/document/(\d+)", url)
    return m.group(1) if m else None


def sanitize_filename(name: str) -> str:
    return re.sub(r"[^\w\-_\. ]", "_", name).strip()


def scrape(url: str) -> dict:
    """Run all strategies and return result dict with 'path' or 'error'."""
    doc_id = extract_doc_id(url)
    if not doc_id:
        return {"error": "Could not extract document ID from URL"}

    strategies = [
        ("embedded-json",       lambda: _strategy_embedded_json(doc_id)),
        ("download-api",        lambda: _strategy_download_api(doc_id)),
        ("text-api",            lambda: _strategy_text_api(doc_id)),
        ("alternative-search",  lambda: _strategy_alternative_search(doc_id)),
        ("playwright",          lambda: _strategy_playwright(doc_id)),
    ]

    for name, fn in strategies:
        try:
            result = fn()
        except Exception as exc:
            continue

        if result is None:
            continue

        if Path(result).exists() and result.endswith((".pdf", ".txt")):
            title = Path(result).stem
            return {"path": result, "strategy": name, "title": title}

        if result.startswith("http"):
            try:
                output = Path(tempfile.gettempdir()) / f"scribd_{doc_id}.pdf"
                _download_file(result, output)
                return {"path": str(output), "strategy": name, "title": doc_id}
            except Exception:
                continue

    return {"error": "All strategies exhausted"}


# ---------------------------------------------------------------------------
# Strategy 1 — embedded JSON
# ---------------------------------------------------------------------------

def _strategy_embedded_json(doc_id: str) -> str | None:
    url = f"https://www.scribd.com/document/{doc_id}"
    resp = SESSION.get(url, timeout=30)
    resp.raise_for_status()
    soup = BeautifulSoup(resp.text, "html.parser")

    for script in soup.find_all("script"):
        if not script.string:
            continue
        for key in ("window.__INITIAL_STATE__", "window.__PRELOADED_STATE__"):
            idx = script.string.find(key)
            if idx == -1:
                continue
            start = script.string.find("{", idx)
            end = script.string.rfind("}")
            if start == -1 or end == -1:
                continue
            try:
                data = json.loads(script.string[start:end + 1])
            except json.JSONDecodeError:
                continue

            doc_info = _dig(data, "document", "documentData", "document")
            if not isinstance(doc_info, dict):
                continue

            dl = doc_info.get("download_url") or doc_info.get("pdf_url") or doc_info.get("original_document_url")
            if dl:
                return dl

            pages = _dig(doc_info, "text_pages") or _dig(doc_info, "textPages") or []
            if isinstance(pages, list) and pages:
                title = doc_info.get("title", doc_id)
                return _pages_to_pdf(pages, title)

    return None


# ---------------------------------------------------------------------------
# Strategy 2 — download API
# ---------------------------------------------------------------------------

def _strategy_download_api(doc_id: str) -> str | None:
    endpoints = [
        f"https://www.scribd.com/document_downloads/{doc_id}",
        f"https://www.scribd.com/document_downloads/direct/{doc_id}",
        f"https://www.scribd.com/doc/{doc_id}?download=1",
        f"https://www.scribd.com/document/{doc_id}/download",
    ]
    for url in endpoints:
        resp = SESSION.head(url, allow_redirects=True, timeout=15)
        if resp.status_code == 200:
            ct = resp.headers.get("Content-Type", "")
            cd = resp.headers.get("Content-Disposition", "")
            if "pdf" in ct.lower() or "attachment" in cd.lower():
                return url
    return None


# ---------------------------------------------------------------------------
# Strategy 3 — text API
# ---------------------------------------------------------------------------

def _strategy_text_api(doc_id: str) -> str | None:
    session = requests.Session()
    session.headers.update({
        "User-Agent": SESSION.headers["User-Agent"],
        "Accept": "application/json, text/plain, */*",
        "Origin": "https://www.scribd.com",
        "Referer": f"https://www.scribd.com/document/{doc_id}",
    })

    for endpoint in [
        f"https://www.scribd.com/document/{doc_id}/text",
        f"https://www.scribd.com/doc/{doc_id}/text",
        f"https://www.scribd.com/doc/{doc_id}/text?format=txt",
        f"https://www.scribd.com/fullscreen/{doc_id}?content_mode=text",
    ]:
        try:
            resp = session.get(endpoint, timeout=30)
        except Exception:
            continue
        if resp.status_code != 200:
            continue

        ct = resp.headers.get("Content-Type", "").lower()

        if "application/pdf" in ct:
            output = Path(tempfile.gettempdir()) / f"scribd_{doc_id}.pdf"
            output.write_bytes(resp.content)
            return str(output)

        if "text/html" in ct:
            continue

        if "json" in ct:
            try:
                data = resp.json()
            except Exception:
                continue
            text = _extract_text_from_json(data)
            if text:
                return _pages_to_pdf(text, doc_id)
        elif "text/plain" in ct:
            return _pages_to_pdf(resp.text.split("\n\n"), doc_id)

    return None


# ---------------------------------------------------------------------------
# Strategy 4 — alternative search
# ---------------------------------------------------------------------------

def _strategy_alternative_search(doc_id: str) -> str | None:
    try:
        from playwright.sync_api import sync_playwright
    except ImportError:
        return None

    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True)
        context = browser.new_context(
            user_agent=SESSION.headers["User-Agent"],
        )
        page = context.new_page()
        title = ""

        try:
            page.goto(f"https://www.scribd.com/document/{doc_id}", wait_until="domcontentloaded", timeout=20000)
            title = page.title().replace(" | PDF", "").replace(" | Scribd", "").strip()
        except Exception:
            pass

        if not title:
            browser.close()
            return None

        search_url = f"https://www.google.com/search?q={requests.utils.quote(title + ' filetype:pdf')}"
        try:
            page.goto(search_url, wait_until="domcontentloaded", timeout=20000)
        except Exception:
            browser.close()
            return None

        page.wait_for_timeout(3000)
        links = page.evaluate("""() => {
            const results = document.querySelectorAll('a[href*=".pdf"]');
            return Array.from(results).map(a => a.href).slice(0, 5);
        }""")

        browser.close()

        for link in links:
            try:
                resp = SESSION.get(link, stream=True, timeout=30)
                if resp.status_code == 200 and "application/pdf" in resp.headers.get("Content-Type", "").lower():
                    output = Path(tempfile.gettempdir()) / f"scribd_{doc_id}_found.pdf"
                    output.write_bytes(resp.content)
                    return str(output)
            except Exception:
                continue

    return None


# ---------------------------------------------------------------------------
# Strategy 5 — Playwright headless browser
# ---------------------------------------------------------------------------

def _strategy_playwright(doc_id: str) -> str | None:
    try:
        from playwright.sync_api import sync_playwright
    except ImportError:
        return None

    doc_url = f"https://www.scribd.com/document/{doc_id}"
    fullscreen_url = f"https://www.scribd.com/fullscreen/{doc_id}"

    with sync_playwright() as p:
        browser = p.chromium.launch(
            headless=True,
            args=["--no-sandbox", "--disable-blink-features=AutomationControlled"],
        )
        context = browser.new_context(
            user_agent=SESSION.headers["User-Agent"],
            locale="id-ID",
            viewport={"width": 1440, "height": 900},
        )
        page = context.new_page()

        found_pdf: list[str] = []

        def on_response(response):
            if response.status != 200:
                return
            host = urlparse(response.url).hostname or ""
            ct = response.headers.get("content-type", "").lower()
            if host.endswith("scribd.com") and "application/pdf" in ct:
                found_pdf.append(response.url)

        page.on("response", on_response)

        try:
            page.goto(fullscreen_url, wait_until="domcontentloaded", timeout=30000)
        except Exception:
            try:
                page.goto(doc_url, wait_until="domcontentloaded", timeout=30000)
            except Exception:
                browser.close()
                return None

        page.wait_for_timeout(8000)

        try:
            page.click("button:has-text('Terima'), button:has-text('Accept'), button:has-text('Setuju')", timeout=5000)
            page.wait_for_timeout(2000)
        except Exception:
            pass

        try:
            page.click("span:has-text('Lewati ke konten utama')", timeout=3000)
            page.wait_for_timeout(2000)
        except Exception:
            pass

        for _ in range(5):
            page.keyboard.press("PageDown")
            page.wait_for_timeout(1500)

        title = page.title().replace(" | PDF", "").replace(" | Scribd", "").strip() or doc_id

        if found_pdf:
            pdf_url = found_pdf[0]
            browser.close()
            return pdf_url

        viewer_selectors = [
            ".text_editor_root", ".text_editor",
            '[class*="text_layer"]', '[class*="page_content"]',
            ".outer_page", ".inner_page",
            '[data-page-number]', '.scribd_viewer',
            "#doc_outer_container",
        ]
        viewer_text = ""
        for sel in viewer_selectors:
            try:
                els = page.query_selector_all(sel)
                if els:
                    viewer_text = "\n\n".join(el.inner_text() for el in els if el.inner_text().strip())
                    if len(viewer_text) > 300:
                        break
            except Exception:
                continue

        if viewer_text and len(viewer_text) > 300:
            result = _pages_to_pdf([viewer_text], title)
            browser.close()
            if result:
                return result

        body_text = page.inner_text("body") or ""
        if body_text and len(body_text) > 500:
            result = _pages_to_pdf(body_text.split("\n\n"), title)
            browser.close()
            if result:
                return result

        output = Path(tempfile.gettempdir()) / f"scribd_{doc_id}.pdf"
        page.pdf(path=str(output), print_background=True)
        browser.close()
        return str(output)


# ---------------------------------------------------------------------------
# Helpers
# ---------------------------------------------------------------------------

def _dig(data: dict, *keys: str):
    for key in keys:
        if isinstance(data, dict):
            data = data.get(key)
        else:
            return None
    return data


def _extract_text_from_json(data) -> list[str] | None:
    pages = _dig(data, "pages") or _dig(data, "text") or data
    if isinstance(pages, list):
        return [str(p) if isinstance(p, str) else p.get("text", "") for p in pages if isinstance(p, (str, dict))]
    if isinstance(pages, dict):
        return list(pages.values()) if all(isinstance(v, str) for v in pages.values()) else None
    return None


def _pages_to_pdf(pages: list, title: str) -> str | None:
    cleaned_title = sanitize_filename(title)
    pages = [p for p in pages if p.strip()]

    try:
        from reportlab.lib.pagesizes import A4
        from reportlab.platypus import Paragraph, SimpleDocTemplate
        from reportlab.lib.styles import getSampleStyleSheet
    except ImportError:
        output = Path(tempfile.gettempdir()) / f"scribd_{cleaned_title}.txt"
        output.write_text("\n\n".join(pages), encoding="utf-8")
        return str(output)

    output = Path(tempfile.gettempdir()) / f"scribd_{cleaned_title}.pdf"
    doc = SimpleDocTemplate(str(output), pagesize=A4)
    styles = getSampleStyleSheet()
    story = []
    for p_text in pages:
        safe = p_text.replace("&", "&amp;").replace("<", "&lt;").replace(">", "&gt;")
        safe = safe.replace("\n", "<br/>")
        try:
            story.append(Paragraph(safe, styles["Normal"]))
        except Exception:
            story.append(Paragraph(safe[:1000], styles["Normal"]))
    try:
        doc.build(story)
    except Exception:
        output_txt = Path(tempfile.gettempdir()) / f"scribd_{cleaned_title}.txt"
        output_txt.write_text("\n\n".join(pages), encoding="utf-8")
        return str(output_txt)
    return str(output)


def _download_file(url: str, output_path: Path) -> Path:
    resp = SESSION.get(url, stream=True, timeout=120)
    resp.raise_for_status()
    output_path.write_bytes(resp.content)
    return output_path
