"""Scraper LinkedIn przez Apify.

Apify uruchamia aktora typu "linkedin search/company scraper" — zwraca listę
kancelarii prawnych z LinkedIn. Domyślnie używamy aktora
`curious_coder/linkedin-search-scraper`, ale można podmienić go zmienną
środowiskową `APIFY_LINKEDIN_ACTOR` lub konstruktorem.

Każdy aktor LinkedIn ma własny schemat inputu — dostosuj `_build_input()`
jeśli zmienisz aktora.
"""

from __future__ import annotations

import logging
import re
from typing import Iterator
from urllib.parse import urlparse

import requests

from .. import config
from ..notifier import is_billing_error, notify_billing_error
from .base import Lead, ScraperSource

log = logging.getLogger(__name__)

APIFY_BASE = "https://api.apify.com/v2"


class ApifyError(Exception):
    pass


class LinkedinScraper(ScraperSource):
    name = "linkedin"

    def __init__(
        self,
        limit: int | None = None,
        actor: str | None = None,
        api_key: str | None = None,
        max_items: int | None = None,
        search_query: str | None = None,
        locations: list[str] | None = None,
    ):
        super().__init__(limit=limit)
        self.actor = actor or config.APIFY_LINKEDIN_ACTOR
        self.api_key = api_key or config.APIFY_API_KEY
        self.max_items = max_items or config.APIFY_LINKEDIN_MAX_ITEMS
        # LinkedIn często zwraca 0 wyników dla polskich fraz — iterujemy
        # przez listę zapytań aż któreś zwróci rekordy z Polski.
        if search_query:
            self.search_queries = [search_query]
        else:
            self.search_queries = [
                "law firm Poland",
                "family law Poland",
                "kancelaria prawna",
                "divorce lawyer Poland",
            ]
        self.search_query = self.search_queries[0]
        self.locations = locations or ["Poland"]

    # ---------- Apify API ----------
    def _actor_path(self) -> str:
        # Apify używa "~" zamiast "/" w ID aktora w URL-u.
        return self.actor.replace("/", "~")

    def _build_input(self, query: str | None = None) -> dict:
        """Input dla aktora harvestapi/linkedin-company-search.

        Twardo wymuszamy locations=["Poland"] — Apify i tak czasem zwraca firmy
        spoza filtra, więc dodatkowo post-filtrujemy w `fetch()`.
        """
        return {
            "searchQuery": query or self.search_query,
            "locations": ["Poland"],
            "maxItems": self.max_items,
            "scraperMode": "full",
        }

    # ---------- filtr geograficzny ----------
    @staticmethod
    def _country_strings(item: dict) -> list[str]:
        """Zbiera wszystkie pola country/location/headquarters z rekordu Apify."""
        out: list[str] = []

        def push(v):
            if isinstance(v, str) and v.strip():
                out.append(v.strip())

        for key in ("country", "countryCode", "headquarters", "location", "addressCountry"):
            v = item.get(key)
            if isinstance(v, str):
                push(v)
            elif isinstance(v, dict):
                for k in ("country", "countryCode", "name", "city"):
                    push(v.get(k))

        for loc in item.get("locations") or []:
            if not isinstance(loc, dict):
                continue
            for k in ("country", "countryCode", "city", "line1", "line2"):
                push(loc.get(k))
            parsed = loc.get("parsed")
            if isinstance(parsed, dict):
                for k in ("country", "countryCode", "city"):
                    push(parsed.get(k))

        return out

    @staticmethod
    def _has_cyrillic(text: str) -> bool:
        return bool(re.search(r"[\u0400-\u04FF]", text or ""))

    @classmethod
    def _is_in_poland(cls, item: dict) -> tuple[bool, str]:
        """Zwraca (czy_w_PL, opis_lokalizacji_do_logu).

        Twarde odrzucenie jeśli którekolwiek pole lokalizacji zawiera cyrylicę
        — to jednoznaczny sygnał firmy spoza Polski (RU/UA/BG/itd.).
        """
        strings = cls._country_strings(item)
        joined = " | ".join(strings) if strings else "(brak danych lokalizacji)"
        if cls._has_cyrillic(joined):
            return False, joined + " [cyrylica]"
        lower = joined.lower()
        if "poland" in lower or "polska" in lower:
            return True, joined
        for s in strings:
            if s.strip().upper() == "PL":
                return True, joined
        return False, joined

    def _run_actor(self, query: str | None = None) -> list[dict]:
        """Synchronicznie uruchamia aktora i zwraca elementy datasetu."""
        url = f"{APIFY_BASE}/acts/{self._actor_path()}/run-sync-get-dataset-items"
        params = {"token": self.api_key, "timeout": config.APIFY_RUN_TIMEOUT}
        payload = self._build_input(query)
        log.info(
            "[linkedin] Uruchamiam aktora %s (query=%r, max_items=%d)",
            self.actor,
            payload["searchQuery"],
            self.max_items,
        )
        try:
            r = requests.post(url, params=params, json=payload, timeout=config.APIFY_RUN_TIMEOUT + 30)
        except requests.RequestException as e:
            raise ApifyError(f"Network error: {e}") from e
        if r.status_code >= 400:
            err = f"HTTP {r.status_code}: {r.text[:500]}"
            if is_billing_error(r.text, r.status_code):
                notify_billing_error("linkedin", f"Apify {err}")
            raise ApifyError(err)
        try:
            return r.json()
        except ValueError as e:
            raise ApifyError(f"Invalid JSON from Apify: {e}") from e

    # ---------- mapowanie ----------
    @staticmethod
    def _norm_phone(value) -> str | None:
        if not value:
            return None
        # Aktor zwraca telefon czasem jako dict {"number":..., "extension":...}
        if isinstance(value, dict):
            value = value.get("number") or value.get("phone") or ""
        if not isinstance(value, str):
            return None
        cleaned = re.sub(r"[\s\-()]", "", value)
        return cleaned or None

    def _to_lead(self, item: dict) -> Lead | None:
        name = item.get("name") or item.get("companyName") or item.get("title")
        if not name:
            return None

        website = item.get("website") or item.get("websiteUrl")
        description = item.get("description") or item.get("tagline") or ""

        # Telefon
        phone = self._norm_phone(item.get("phone"))
        if not phone:
            phone = self.extract_phone(description)

        # Email — aktor harvestapi nie zwraca go bezpośrednio, więc szukamy w opisie
        email = item.get("email") or item.get("companyEmail")
        if not email:
            email = self.extract_email(description)

        # Adres — locations[0] z headquarter=True albo pierwsza
        city = street = postal = None
        locations = item.get("locations") or []
        hq = next((l for l in locations if isinstance(l, dict) and l.get("headquarter")), None)
        if not hq and locations and isinstance(locations[0], dict):
            hq = locations[0]
        if hq:
            city = hq.get("city") or (hq.get("parsed") or {}).get("city")
            street = hq.get("line1")
            if hq.get("line2"):
                street = f"{street} {hq['line2']}".strip()
            postal = hq.get("postalCode")

        # Branża
        industries = item.get("industries") or []
        industry = None
        if industries and isinstance(industries[0], dict):
            industry = industries[0].get("name")
        more = f"LinkedIn / {industry}" if industry else "LinkedIn"

        return Lead(
            name=name,
            source=self.name,
            nameMore=more,
            city=city,
            street=street,
            postalCode=postal,
            email=email,
            phone=phone,
            website=website,
            extra={"linkedin_url": item.get("linkedinUrl") or item.get("url")},
        )

    # ---------- email enrichment ----------
    def _extract_email_from_url(self, url: str) -> str | None:
        r = self.get(url)
        if not r:
            return None
        return self.extract_email(r.text)

    def _enrich_email_from_website(self, website: str) -> str | None:
        """Pobiera email ze strony www firmy. Sprawdza też /kontakt i /contact."""
        if not website:
            return None
        # Normalizacja URL
        if not website.startswith(("http://", "https://")):
            website = "https://" + website
        try:
            parsed = urlparse(website)
        except Exception:
            return None
        if not parsed.netloc:
            return None
        base = f"{parsed.scheme}://{parsed.netloc}"

        for url in (base, f"{base}/kontakt", f"{base}/contact"):
            email = self._extract_email_from_url(url)
            if email:
                return email
        return None

    # ---------- main loop ----------
    def fetch(self) -> Iterator[Lead]:
        if not self.api_key:
            log.warning("[linkedin] Brak APIFY_API_KEY — pomijam źródło.")
            return

        # Iteruj przez listę zapytań — zatrzymaj się przy pierwszym, które
        # zwróci jakiekolwiek rekordy z Polski.
        items: list[dict] = []
        used_query: str | None = None
        for query in self.search_queries:
            try:
                batch = self._run_actor(query)
            except ApifyError as e:
                log.error("[linkedin] Apify run failed (query=%r): %s", query, e)
                continue
            log.info("[linkedin] query=%r → %d rekordów", query, len(batch))
            pl_count = sum(1 for it in batch if self._is_in_poland(it)[0])
            if pl_count:
                log.info(
                    "[linkedin] query=%r zwrócił %d firm z Polski — używam tego zapytania",
                    query,
                    pl_count,
                )
                items = batch
                used_query = query
                break
            log.info("[linkedin] query=%r — 0 firm z Polski, próbuję następne", query)

        if not items:
            log.info(
                "[linkedin] SKIP — żadne z zapytań %s nie zwróciło firm z Polski.",
                self.search_queries,
            )
            return

        self.search_query = used_query or self.search_query
        log.info("[linkedin] Apify zwrócił %d rekordów (query=%r)", len(items), used_query)

        yielded = 0
        seen: set[tuple[str, str]] = set()
        for item in items:
            if self.limit and yielded >= self.limit:
                return
            in_pl, loc_str = self._is_in_poland(item)
            if not in_pl:
                nm = (
                    item.get("name")
                    or item.get("companyName")
                    or item.get("title")
                    or "(bez nazwy)"
                )
                log.info("[linkedin] SKIP — firma poza Polską: %s | %s", nm, loc_str)
                continue
            lead = self._to_lead(item)
            if not lead:
                continue
            if not lead.email and lead.website:
                log.info("[linkedin] Brak emaila dla '%s' — szukam na %s", lead.name, lead.website)
                lead.email = self._enrich_email_from_website(lead.website)
            if not lead.email:
                log.info("[linkedin] SKIPPED (brak emaila): %s", lead.name)
                continue
            if not lead.is_valid():
                continue
            key = (lead.email or "", lead.phone or "")
            if key in seen:
                continue
            seen.add(key)
            yielded += 1
            yield lead
