"""Integration test: full pipeline with CSV source, in-memory index, dry-run dgred."""

import logging

from agent_samochodowy.config import Settings
from agent_samochodowy.dgred.client import DgredClient
from agent_samochodowy.ingestor.csv_source import CsvSource
from agent_samochodowy.matcher.index import CatalogIndex
from agent_samochodowy.models import Offer
from agent_samochodowy.orchestrator import Pipeline


def _build_test_index() -> CatalogIndex:
    idx = CatalogIndex(":memory:")
    offers = [
        Offer(
            offer_id="1", platform="allegro", url="https://a.pl/1",
            title="Pompa wody VOLVO FH13 D13A 20744939",
            brand="VOLVO", model="FH13", category="pompa",
            engine_code="D13A",
            oe_numbers=["20744939", "20734268"],
            price=450.0,
        ),
        Offer(
            offer_id="2", platform="ebay", url="https://e.com/2",
            title="Alternator OPEL Astra F C18NZ 0120488234",
            brand="OPEL", model="Astra F", category="alternator",
            engine_code="C18NZ",
            oe_numbers=["0120488234"],
            price=280.0,
        ),
        Offer(
            offer_id="3", platform="allegro", url="https://a.pl/3",
            title="Filtr oleju SCANIA R500",
            brand="SCANIA", model="R500", category="filtr",
            oe_numbers=[],
            price=120.0,
        ),
    ]
    idx.build_from_offers(offers)
    return idx


def _settings() -> Settings:
    return Settings(dry_run=True, db_path=":memory:")


class TestPipelineIntegration:
    def test_full_pipeline_with_sample_posts(self, caplog, tmp_path):
        """Run full pipeline on sample posts with in-memory index and dry-run CRM."""
        settings = _settings()
        settings.db_path = str(tmp_path / "catalog.db")
        source = CsvSource("data/samples/posts.json")
        index = _build_test_index()
        dgred = DgredClient(settings)

        pipeline = Pipeline(settings, source, index, dgred)

        with caplog.at_level(logging.DEBUG):
            stats = pipeline.run()

        # 30 posts total
        assert stats.total_posts == 30

        # Several selling and off-topic posts should be skipped by pre-filter
        assert stats.skipped_prefilter >= 5

        # Many posts should create leads (OE matches, attribute matches, etc.)
        leads_created = stats.matched + stats.review + stats.no_match
        assert leads_created >= 1  # At least some matches should work

        # No errors
        assert stats.errors == 0

        dgred.close()
        index.close()

    def test_dedup_on_second_run(self, caplog, tmp_path):
        """Running the same posts twice should mark them as duplicates."""
        settings = _settings()
        settings.db_path = str(tmp_path / "catalog.db")
        source = CsvSource("data/samples/posts.json")
        index = _build_test_index()
        dgred = DgredClient(settings)

        pipeline = Pipeline(settings, source, index, dgred)

        # First run
        stats1 = pipeline.run()
        assert stats1.duplicates == 0

        # Second run with fresh source (same posts)
        source2 = CsvSource("data/samples/posts.json")
        pipeline.source = source2
        stats2 = pipeline.run()
        assert stats2.duplicates == stats1.total_posts

        dgred.close()
        index.close()
