"""Fetch favicon + title + description from a registered site's live domain.

Synchronous, stdlib-only (WSGI/cPanel friendly). Called at site create and via
the refresh_metadata action. Network failures are swallowed: metadata is
best-effort, never a reason to block site creation.
"""
import ipaddress
import re
import socket
from urllib.parse import urljoin
from urllib.request import Request, urlopen

def _user_agent():
    """Identify the calling deployment rather than a hardcoded host.

    This codebase runs as both api.caridu.id and api.nocthink.com; a fixed URL
    would misattribute outbound fetches to the wrong deployment in the logs of
    whatever site is being scraped. Read lazily — this module is otherwise
    stdlib-only and imported at module scope.
    """
    from django.conf import settings

    base = getattr(settings, "ANALYTICS_PUBLIC_BASE_URL", "") or "https://api.caridu.id"
    return f"Mozilla/5.0 (compatible; KHubAnalytics/1.0; +{base})"
_MAX_BYTES = 200_000
_TIMEOUT = 5


def _hostname(domain: str) -> str:
    h = domain.strip().lower()
    h = h.replace("https://", "").replace("http://", "")
    return h.strip("/").split("/")[0].split(":")[0]


def _is_public(host: str) -> bool:
    """Block SSRF to private / loopback / link-local addresses."""
    try:
        for res in socket.getaddrinfo(host, None):
            ip = ipaddress.ip_address(res[4][0])
            if ip.is_private or ip.is_loopback or ip.is_link_local or ip.is_reserved or ip.is_multicast:
                return False
        return True
    except Exception:
        return False


def _meta_content(html: str, attr: str, value: str) -> str:
    m = re.search(
        rf'<meta[^>]+{attr}=["\']{re.escape(value)}["\'][^>]+content=["\'](.*?)["\']',
        html, re.I | re.S,
    )
    if not m:
        m = re.search(
            rf'<meta[^>]+content=["\'](.*?)["\'][^>]+{attr}=["\']{re.escape(value)}["\']',
            html, re.I | re.S,
        )
    return m.group(1).strip() if m else ""


def fetch_site_metadata(domain: str) -> dict:
    host = _hostname(domain)
    if not host or not _is_public(host):
        return {}

    url = f"https://{host}/"
    try:
        with urlopen(Request(url, headers={"User-Agent": _user_agent()}), timeout=_TIMEOUT) as resp:
            html = resp.read(_MAX_BYTES).decode("utf-8", "ignore")
            final = resp.geturl() or url
    except Exception:
        # Site unreachable — still record the conventional favicon path.
        return {"favicon_url": f"https://{host}/favicon.ico"}

    title = ""
    m = re.search(r"<title[^>]*>(.*?)</title>", html, re.I | re.S)
    if m:
        title = re.sub(r"\s+", " ", m.group(1)).strip()[:255]

    desc = _meta_content(html, "name", "description") or _meta_content(html, "property", "og:description")
    desc = desc[:500]

    icon = ""
    for tag in re.findall(r"<link[^>]+>", html, re.I):
        if re.search(r'rel=["\'][^"\']*icon[^"\']*["\']', tag, re.I):
            href = re.search(r'href=["\'](.*?)["\']', tag, re.I)
            if href:
                icon = urljoin(final, href.group(1))
                break
    if not icon:
        icon = urljoin(final, "/favicon.ico")

    return {"favicon_url": icon[:512], "title": title, "description": desc}
