"""Extraction pipeline: prefilter → LLM (with regex fallback) → PartQuery."""

from __future__ import annotations

import logging

from agent_samochodowy.config import Settings
from agent_samochodowy.extraction.prefilter import is_part_request
from agent_samochodowy.extraction.regex_extractor import extract_with_regex
from agent_samochodowy.models import PartQuery, Post

logger = logging.getLogger(__name__)


def _extract_with_llm(post: Post, settings: Settings) -> PartQuery:
    """Dispatch to the configured LLM provider."""
    if settings.llm_provider == "anthropic":
        from agent_samochodowy.extraction.llm_extractor import extract_with_llm

        return extract_with_llm(post, settings)

    from agent_samochodowy.extraction.openai_extractor import extract_with_openai

    return extract_with_openai(post, settings)


def extract(post: Post, settings: Settings) -> PartQuery | None:
    """Run the full extraction pipeline on a post.

    Returns PartQuery if the post is a part request, None if it should be skipped.
    """
    # Step 1: cheap pre-filter
    if not is_part_request(post.text):
        logger.debug("Post %s skipped by pre-filter", post.post_id)
        return None

    # Step 2: LLM extraction (with regex fallback)
    try:
        if settings.llm_api_key:
            query = _extract_with_llm(post, settings)
        else:
            logger.info("No LLM API key, using regex fallback for post %s", post.post_id)
            query = extract_with_regex(post)
    except Exception:
        logger.exception("LLM extraction failed for post %s, falling back to regex", post.post_id)
        query = extract_with_regex(post)

    # Respect LLM's classification
    if not query.is_part_request:
        logger.debug("Post %s classified as non-request by extractor", post.post_id)
        return None

    return query
