"""Bounded local OCR for bank-statement review. It never posts transactions."""
import re
import shutil
import subprocess
import tempfile
from pathlib import Path


DATE_RE = re.compile(r"\b(\d{1,2}[/-]\d{1,2}[/-]\d{2,4}|\d{4}[/-]\d{1,2}[/-]\d{1,2})\b")
AMOUNT_RE = re.compile(r"(?<!\w)(?:Rp\s*)?[-+]?\d[\d.,]*\d(?:[.,]\d{2})?(?!\w)", re.I)


def extract_bank_statement_pdf(file_path):
    if not shutil.which("pdftoppm") or not shutil.which("tesseract"):
        return {"lines": [], "warnings": ["OCR service is unavailable on this server."], "provider_status": "unavailable"}, None
    with tempfile.TemporaryDirectory(prefix="khub-bank-ocr-") as temp_dir:
        prefix = str(Path(temp_dir) / "page")
        try:
            subprocess.run(["pdftoppm", "-f", "1", "-l", "5", "-r", "180", "-png", str(file_path), prefix], check=True, timeout=45, capture_output=True)
            texts = []
            for image in sorted(Path(temp_dir).glob("page-*.png")):
                result = subprocess.run(["tesseract", str(image), "stdout", "--psm", "6"], check=True, timeout=30, capture_output=True, text=True)
                texts.append(result.stdout)
        except (subprocess.SubprocessError, OSError) as exc:
            return {"lines": [], "warnings": [f"OCR could not read this PDF: {type(exc).__name__}."], "provider_status": "failed"}, None
    raw_text = "\n".join(texts)
    candidates = []
    for raw in raw_text.splitlines():
        date = DATE_RE.search(raw)
        amounts = AMOUNT_RE.findall(raw)
        if date and amounts:
            candidates.append({"date": date.group(1), "raw": raw.strip(), "amount_candidates": amounts, "description": raw[: date.start()].strip(), "confidence": 0.6})
    warnings = [] if candidates else ["No transaction rows were confidently detected. Review the extracted text manually."]
    return {"lines": candidates, "raw_text": raw_text, "warnings": warnings, "provider_status": "complete", "page_limit": 5}, 60 if candidates else 30
