"""Scraper Krajowego Rejestru Adwokatów (rejestradwokatow.pl).

Strona oficjalnego rejestru adwokatury (KRAIA) — udostępnia formularz POST
z filtrami po nazwisku/mieście/izbie. Nie ma filtra "specjalizacja",
więc filtrowanie po prawie rodzinnym musi nastąpić dalej w pipeline.
"""

from __future__ import annotations

import logging
import re
from typing import Iterator
from urllib.parse import urljoin

from bs4 import BeautifulSoup

from .. import config
from ..notifier import is_billing_error, notify_billing_error
from .base import Lead, ScraperSource

log = logging.getLogger(__name__)

BASE = "https://rejestradwokatow.pl"
SEARCH_URL = f"{BASE}/adwokat/wyszukaj"

# Statusy: 2 = Wykonujący zawód
SEARCH_STATUSES = ["2"]


class OraScraper(ScraperSource):
    name = "ora"

    def __init__(self, limit: int | None = None, cities: list[str] | None = None):
        super().__init__(limit=limit)
        self.cities = cities or config.CITIES

    # ---------- search ----------
    def _search_city(self, city: str) -> list[str]:
        """Zwraca listę URL-i profili adwokatów z danego miasta."""
        log.info("[ora] searching city: %s", city)
        try:
            r = self.session.post(
                SEARCH_URL,
                data={
                    "nazwisko": "",
                    "imie": "",
                    "imie2": "",
                    "ulica": "",
                    "miasto": city,
                    "id_j": "0",
                    "status[]": SEARCH_STATUSES,
                    "btn_wyszukaj": "Wyszukaj",
                },
                timeout=config.REQUEST_TIMEOUT,
            )
        except Exception as e:
            log.warning("[ora] search failed for %s: %s", city, e)
            return []
        if r.status_code != 200:
            log.warning("[ora] search HTTP %d for %s", r.status_code, city)
            if is_billing_error(r.text, r.status_code):
                notify_billing_error(
                    "ora",
                    f"HTTP {r.status_code} dla city={city!r}: {r.text[:300]}",
                )
            return []

        soup = BeautifulSoup(r.text, "lxml")
        urls: list[str] = []
        table = soup.find("table")
        if not table:
            return urls
        for row in table.find_all("tr")[1:]:
            a = row.find("a", href=True)
            if a:
                urls.append(urljoin(BASE, a["href"]))
        log.info("[ora] %s → %d profiles", city, len(urls))
        return urls

    # ---------- detail ----------
    def _parse_detail(self, url: str) -> Lead | None:
        r = self.get(url)
        if r is None:
            return None
        soup = BeautifulSoup(r.text, "lxml")

        # Imię i nazwisko — pierwsze h2 zawierające słowa-nazwiska (drugie h2 na stronie).
        person_name: str | None = None
        for h2 in soup.find_all("h2"):
            txt = h2.get_text(" ", strip=True)
            if txt and "Szczegółow" not in txt and len(txt) < 100:
                person_name = txt
                break

        # Email jest obfuscowany w <div class="address_e" data-ea="..." data-eb="...">
        # Składamy: data-ea + "@" + data-eb.  Pomijamy adresy izb (ora.*@adwokatura.pl).
        email: str | None = None
        for addr_div in soup.find_all("div", class_="address_e"):
            ea = (addr_div.get("data-ea") or "").strip()
            eb = (addr_div.get("data-eb") or "").strip()
            if ea and eb:
                candidate = f"{ea}@{eb}"
                if "adwokatura.pl" not in eb.lower():
                    email = candidate
                    break

        # Sekcja "MIEJSCE WYKONYWANIA ZAWODU"
        kancelaria_name: str | None = None
        street: str | None = None
        postal: str | None = None
        city: str | None = None
        phone: str | None = None

        section = None
        for h3 in soup.find_all("h3"):
            if "MIEJSCE WYKONYWANIA" in h3.get_text(" ", strip=True).upper():
                section = h3.find_next(class_="line_list_K") or h3.find_next("div")
                break

        if section:
            block = section.find("div")
            if block:
                # Rozbij na linie po <br>
                lines = _split_on_br(block)
                raw = "\n".join(lines)

                if not email:
                    email = self.extract_email(raw)
                phone = self.extract_phone(raw)

                # Sekwencja typowa:
                # [0]= "Kancelaria"  (label ze <strong>)
                # [1]= "Kancelaria Adwokacka XYZ"   ← nazwa firmy
                # [2]= "ul. Jana Pawła II 27"      ← ulica
                # [3]= "00-867 Warszawa"           ← kod + miasto
                # [4..]= "Komórkowy: 600...", "Email: ..."
                content_lines = [l for l in lines if l and l != "Kancelaria"]

                # kod pocztowy + miasto
                for i, line in enumerate(content_lines):
                    m = re.match(r"(\d{2}-\d{3})\s+(.+)", line)
                    if m:
                        postal = m.group(1)
                        city = m.group(2).strip()
                        # Linia powyżej to zwykle ulica
                        if i > 0:
                            candidate = content_lines[i - 1]
                            if not candidate.lower().startswith(("email", "komórkowy", "stacjonarny", "brak")):
                                street = candidate
                        # Linia powyżej ulicy to zwykle nazwa kancelarii
                        if i > 1 and not kancelaria_name:
                            candidate = content_lines[i - 2]
                            if not re.search(r"\d{2}-\d{3}", candidate):
                                kancelaria_name = candidate
                        break

        # Złożenie nazwy — zawsze dołącz imię i nazwisko adwokata, by uniknąć
        # generycznych "Kancelaria Adwokacka" dla wielu różnych osób.
        if kancelaria_name and person_name and person_name not in kancelaria_name:
            display_name = f"{kancelaria_name} — {person_name}"
        elif kancelaria_name:
            display_name = kancelaria_name
        elif person_name:
            display_name = f"Adwokat {person_name}"
        else:
            return None

        lead = Lead(
            name=display_name,
            source=self.name,
            nameMore=f"Adwokat / {person_name}" if person_name and kancelaria_name else "Adwokat",
            city=city,
            street=street,
            postalCode=postal,
            email=email,
            phone=phone,
            extra={"profile_url": url},
        )
        return lead

    # ---------- main loop ----------
    def fetch(self) -> Iterator[Lead]:
        yielded = 0
        for city in self.cities:
            urls = self._search_city(city)
            for url in urls:
                if self.limit and yielded >= self.limit:
                    return
                lead = self._parse_detail(url)
                if not lead:
                    continue
                if not lead.is_valid():
                    log.debug("[ora] skipped invalid lead: %s", lead.name)
                    continue
                yielded += 1
                yield lead
            if self.limit and yielded >= self.limit:
                return


def _split_on_br(tag) -> list[str]:
    """Rozbija zawartość tagu na linie według znaczników <br>."""
    lines: list[str] = []
    buf: list[str] = []
    for child in tag.descendants:
        if getattr(child, "name", None) == "br":
            text = " ".join(buf).strip()
            if text:
                lines.append(re.sub(r"\s+", " ", text))
            buf = []
        elif isinstance(child, str):
            buf.append(child.strip())
    text = " ".join(buf).strip()
    if text:
        lines.append(re.sub(r"\s+", " ", text))
    return lines
