#!/usr/bin/env python3
"""Classify all FB groups as public/private/unavailable using the old actor (no login).

Runs in batches of 10 with pauses between batches.
Saves access_status + access_checked_at to fb_groups table.
"""

from __future__ import annotations

import json
import sqlite3
import sys
import time
from datetime import datetime
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "src"))

import httpx
from dotenv import load_dotenv

load_dotenv()

from agent_samochodowy.config import Settings

APIFY_BASE = "https://api.apify.com/v2"
RESULTS_LIMIT = 3
BATCH_SIZE = 10
BATCH_PAUSE = 5  # seconds between batches


def classify_group(
    client: httpx.Client, token: str, actor_id: str, group_url: str,
) -> tuple[str, int, str]:
    """Returns (status, post_count, sample_text)."""
    try:
        resp = client.post(
            f"{APIFY_BASE}/acts/{actor_id}/runs",
            params={"token": token},
            json={
                "startUrls": [{"url": group_url}],
                "resultsLimit": RESULTS_LIMIT,
                "sortBy": "New posts",
            },
        )
        if resp.status_code != 201:
            return "error", 0, f"start_http_{resp.status_code}"
    except Exception as e:
        return "error", 0, str(e)[:100]

    run_data = resp.json().get("data", {})
    run_id = run_data.get("id")
    dataset_id = run_data.get("defaultDatasetId")
    if not run_id:
        return "error", 0, "no_run_id"

    # Poll for completion
    status = "UNKNOWN"
    for _ in range(45):
        time.sleep(2)
        try:
            sr = client.get(
                f"{APIFY_BASE}/actor-runs/{run_id}", params={"token": token}
            )
            status = sr.json().get("data", {}).get("status", "UNKNOWN")
            if status in ("SUCCEEDED", "FAILED", "ABORTED", "TIMED-OUT"):
                break
        except Exception:
            continue

    if status != "SUCCEEDED":
        return "error", 0, f"run_{status}"

    # Fetch items
    try:
        items_resp = client.get(
            f"{APIFY_BASE}/datasets/{dataset_id}/items",
            params={"token": token, "format": "json"},
        )
        items = items_resp.json()
    except Exception as e:
        return "error", 0, str(e)[:100]

    if not items:
        return "unavailable", 0, "empty_dataset"

    # Check for error items
    errors = [i for i in items if i.get("error")]
    if errors:
        err = errors[0].get("error", "")
        desc = errors[0].get("errorDescription", "")
        combined = f"{err} {desc}".upper()
        if "PRIVATE" in combined:
            return "private", 0, ""
        if "NOT AVAILABLE" in combined or "CONTENT ISN'T" in combined:
            return "unavailable", 0, ""
        return "unavailable", 0, f"{err}: {desc}"[:100]

    # Clean items with text
    clean = [i for i in items if (i.get("text") or i.get("message") or "").strip()]
    if not clean:
        # Items exist but no text — could be private with skeleton data
        return "unavailable", 0, "items_no_text"

    sample = (clean[0].get("text") or clean[0].get("message") or "")[:200]
    return "public", len(clean), sample


def main() -> None:
    settings = Settings()
    conn = sqlite3.connect(settings.db_path)
    conn.row_factory = sqlite3.Row

    groups = conn.execute(
        "SELECT group_id, nazwa, url, kategoria FROM fb_groups ORDER BY kategoria, nazwa"
    ).fetchall()

    total = len(groups)
    print(f"Classifying {total} groups in batches of {BATCH_SIZE}...")
    print()

    results: list[dict] = []
    now_iso = datetime.now().isoformat()

    with httpx.Client(timeout=120) as client:
        for batch_start in range(0, total, BATCH_SIZE):
            batch = groups[batch_start : batch_start + BATCH_SIZE]
            batch_num = batch_start // BATCH_SIZE + 1
            total_batches = (total + BATCH_SIZE - 1) // BATCH_SIZE
            print(f"--- Batch {batch_num}/{total_batches} ---")

            for g in batch:
                gid = g["group_id"]
                url = g["url"]
                nazwa = g["nazwa"][:50]

                access, posts, sample = classify_group(
                    client, settings.apify_token, settings.apify_actor_id, url,
                )

                # Save to DB
                conn.execute(
                    "UPDATE fb_groups SET access_status=?, access_checked_at=? WHERE group_id=?",
                    (access, now_iso, gid),
                )
                conn.commit()

                results.append({
                    "group_id": gid,
                    "nazwa": nazwa,
                    "kategoria": g["kategoria"],
                    "access": access,
                    "posts": posts,
                    "sample": sample,
                })

                tag = {"public": "PUB", "private": "PRV", "unavailable": "N/A", "error": "ERR"}
                done = len(results)
                print(
                    f"  [{done:3d}/{total}] {tag.get(access, '???')} "
                    f"({posts}p) {nazwa}"
                )

            # Pause between batches
            if batch_start + BATCH_SIZE < total:
                print(f"  (pause {BATCH_PAUSE}s)")
                time.sleep(BATCH_PAUSE)

    # --- Summary ---
    print()
    print("=" * 80)

    from collections import Counter

    by_access = Counter(r["access"] for r in results)
    print(f"TOTAL: {total}")
    for status in ["public", "private", "unavailable", "error"]:
        n = by_access.get(status, 0)
        pct = n / total * 100
        print(f"  {status:12s}: {n:3d} ({pct:.0f}%)")

    # Public groups by category
    public = [r for r in results if r["access"] == "public"]
    print(f"\nPUBLIC GROUPS ({len(public)}) by category:")
    by_cat = Counter(r["kategoria"] for r in public)
    for cat, n in sorted(by_cat.items()):
        print(f"  {cat}: {n}")
        for r in sorted(public, key=lambda x: x["kategoria"]):
            if r["kategoria"] == cat:
                sample_short = r["sample"][:70].replace("\n", " ")
                print(f"    {r['nazwa'][:55]:55s} | {r['posts']}p | {sample_short}")

    # Demand vs supply classification
    demand_keywords = [
        "kupię", "kupie", "szukam", "potrzebuję", "potrzebuje",
        "poszukuję", "poszukuje", "kto ma", "ktoś ma", "ktos ma",
        "pilnie", "pomocy", "usterka", "awaria", "problem",
        "nie działa", "nie dziala", "zepsuł", "zepsul",
    ]
    supply_keywords = [
        "sprzedam", "sprzedaż", "oferuję", "oferuje", "zapraszam",
        "wyróżnić", "wyroznic", "promocja", "reklama", "firma",
        "usługi", "uslugi", "serwis", "oferta",
    ]

    print(f"\nDEMAND vs SUPPLY (from sample texts):")
    demand_count = 0
    supply_count = 0
    unclear_count = 0
    for r in public:
        sample_lower = r["sample"].lower()
        is_demand = any(kw in sample_lower for kw in demand_keywords)
        is_supply = any(kw in sample_lower for kw in supply_keywords)
        if is_demand and not is_supply:
            tag = "DEMAND"
            demand_count += 1
        elif is_supply and not is_demand:
            tag = "SUPPLY"
            supply_count += 1
        elif is_demand and is_supply:
            tag = "MIXED"
            demand_count += 1  # count mixed as demand too
        else:
            tag = "UNCLEAR"
            unclear_count += 1
        print(f"  {tag:7s} | {r['nazwa'][:50]:50s} | {r['sample'][:60].replace(chr(10),' ')}")
    print(f"\n  Demand/mixed: {demand_count} | Supply: {supply_count} | Unclear: {unclear_count}")

    # Activate public, deactivate rest
    conn.execute("UPDATE fb_groups SET aktywna = 0")
    conn.execute("UPDATE fb_groups SET aktywna = 1 WHERE access_status = 'public'")
    activated = conn.execute("SELECT count(*) FROM fb_groups WHERE aktywna = 1").fetchone()[0]
    conn.commit()
    conn.close()

    print(f"\nDB updated: {activated} groups set aktywna=1 (public), rest deactivated.")
    print("Done.")


if __name__ == "__main__":
    main()
