import csv
import sys
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parent.parent))

from google.genai import types
from pydantic import BaseModel

from app.config import CATEGORIES, GEMINI_API_KEY, GEMINI_MODEL
from app.content.gemini_client import _client
from app.dedup.normalize import normalize, text_hash
from app.dedup.store import connect

CSV_PATH = Path(__file__).resolve().parent.parent / "gecmis" / "shorts.csv"


class FactCategoryPair(BaseModel):
    fact: str
    category: str


class ClassificationResponse(BaseModel):
    items: list[FactCategoryPair]


def load_facts() -> list[str]:
    with open(CSV_PATH, encoding="utf-8") as f:
        reader = csv.reader(f)
        return [row[0].strip() for row in reader if row and row[0].strip()]


def classify(facts: list[str]) -> list[FactCategoryPair]:
    category_list = "\n".join(f"- {slug}: {cat['display_name']}" for slug, cat in CATEGORIES.items())
    numbered = "\n".join(f"{i+1}. {f}" for i, f in enumerate(facts))

    prompt = f"""Aşağıdaki Türkçe "ilginç bilgi" cümlelerinin her birini, verilen 8 kategoriden
TAM OLARAK birine ata. Her bilgi için, bilginin metnini AYNEN (değiştirmeden) ve en uygun
kategori slug'ını döndür. Kategori slug'ı MUTLAKA aşağıdaki listeden birebir olmalı.

Kategoriler:
{category_list}

Bilgiler:
{numbered}
"""

    response = _client.models.generate_content(
        model=GEMINI_MODEL,
        contents=prompt,
        config=types.GenerateContentConfig(
            response_mime_type="application/json",
            response_schema=ClassificationResponse,
            temperature=0.0,
        ),
    )
    result = ClassificationResponse.model_validate_json(response.text)
    return result.items


def main():
    facts = load_facts()
    print(f"{len(facts)} geçmiş bilgi yüklendi, sınıflandırılıyor...")

    valid_slugs = set(CATEGORIES.keys())
    original_by_normalized = {normalize(f): f for f in facts}

    classified = classify(facts)

    inserted, skipped_invalid, skipped_duplicate, unmatched = 0, 0, 0, 0
    conn = connect()
    try:
        for item in classified:
            category = item.category.strip()
            if category not in valid_slugs:
                skipped_invalid += 1
                print(f"  [GEÇERSİZ KATEGORİ: {category!r}] {item.fact[:60]}...")
                continue

            normalized = normalize(item.fact)
            original_fact = original_by_normalized.get(normalized, item.fact)

            try:
                conn.execute(
                    """
                    INSERT INTO facts (category, fact_text, normalized_text, text_hash, title, description, hashtags, created_at, status)
                    VALUES (?, ?, ?, ?, '', '', '[]', datetime('now'), 'historical')
                    """,
                    (category, original_fact, normalized, text_hash(normalized)),
                )
                inserted += 1
            except Exception:
                skipped_duplicate += 1

        conn.commit()
    finally:
        conn.close()

    print(f"\nTamamlandı: {inserted} eklendi, {skipped_invalid} geçersiz kategori, {skipped_duplicate} zaten mevcuttu.")
    print(f"(Toplam girdi: {len(facts)}, sınıflandırılan: {len(classified)})")


if __name__ == "__main__":
    main()
