"""Módulo RAG (Retrieval-Augmented Generation) para xzonas-supervisor.

Utiliza Qdrant como vector store y LlamaIndex como framework de indexación y retrieval.
El modelo de embeddings es BAAI/bge-small-en-v1.5 (CPU, ~130 MB, multilingüe, gratuito).

Activación: establece RAG_ENABLED=true y QDRANT_URL=http://qdrant:6333 en .env.
Si Qdrant no está disponible o RAG_ENABLED=false, el módulo devuelve contexto vacío
de forma silenciosa sin romper el flujo del supervisor.

Uso en el orchestrator:
    from .rag import rag_store
    context = await rag_store.query("rendimiento del chatbot PHP")
    # → str con parágrafos relevantes o "" si no hay contexto
"""
from __future__ import annotations

import asyncio
import logging
import os
from pathlib import Path
from typing import Any

logger = logging.getLogger(__name__)

# Directorio de docs del proyecto (montado de solo lectura en Docker)
_DOCS_DIR = Path(__file__).parent.parent / "docs"


class RagStore:
    """Wrapper fail-safe around LlamaIndex + Qdrant.

    Instancia única (singleton) gestionada en `setup()` al arrancar el supervisor.
    Si la importación de las librerías falla o Qdrant no está levantado, todas las
    operaciones devuelven resultados vacíos sin levantar excepciones.
    """

    def __init__(self) -> None:
        self._ready = False
        self._index: Any = None
        self._qdrant_client: Any = None
        self._collection: str = "xzonas_docs"
        self._embed_model_name: str = "BAAI/bge-small-en-v1.5"
        self._embed_model: Any = None
        self._ingested_dirs: list[str] = []

    # ------------------------------------------------------------------ #
    # Ciclo de vida                                                         #
    # ------------------------------------------------------------------ #

    def setup(
        self,
        qdrant_url: str,
        collection: str = "xzonas_docs",
        embed_model: str = "BAAI/bge-small-en-v1.5",
    ) -> None:
        """Inicializa el cliente Qdrant y el índice LlamaIndex.

        Llamada desde `startup_event()` en main.py cuando RAG_ENABLED=true.
        Si algo falla, registra el error pero no relanza la excepción.
        """
        if not qdrant_url:
            logger.info("rag.disabled: QDRANT_URL no configurado")
            return
        self._collection = collection
        self._embed_model_name = embed_model
        try:
            self._init_sync(qdrant_url)
            logger.info(
                "rag.ready",
                extra={"collection": collection, "embed_model": embed_model, "qdrant_url": qdrant_url},
            )
        except Exception as exc:  # noqa: BLE001
            logger.warning("rag.setup.failed: %s", exc)

    def _init_sync(self, qdrant_url: str) -> None:
        """Importa y configura las librerías (puede tardar por descarga del modelo)."""
        from llama_index.core import Settings as LlamaSettings, StorageContext, VectorStoreIndex  # type: ignore[import-not-found]
        from llama_index.embeddings.huggingface import HuggingFaceEmbedding  # type: ignore[import-not-found]
        from llama_index.vector_stores.qdrant import QdrantVectorStore  # type: ignore[import-not-found]
        from qdrant_client import QdrantClient  # type: ignore[import-not-found]
        from qdrant_client.models import Distance, VectorParams  # type: ignore[import-not-found]

        # Cliente Qdrant
        self._qdrant_client = QdrantClient(url=qdrant_url, timeout=10)

        # Crear colección si no existe (dim=384 para bge-small)
        existing = {c.name for c in self._qdrant_client.get_collections().collections}
        if self._collection not in existing:
            self._qdrant_client.create_collection(
                collection_name=self._collection,
                vectors_config=VectorParams(size=384, distance=Distance.COSINE),
            )

        # Modelo de embeddings HuggingFace (descargado en caché local)
        self._embed_model = HuggingFaceEmbedding(model_name=self._embed_model_name)
        LlamaSettings.embed_model = self._embed_model
        LlamaSettings.llm = None  # no usamos LLM en el retrieval; lo hace el orchestrator

        # Índice sobre la colección Qdrant
        vector_store = QdrantVectorStore(
            client=self._qdrant_client,
            collection_name=self._collection,
        )
        storage_context = StorageContext.from_defaults(vector_store=vector_store)
        self._index = VectorStoreIndex.from_vector_store(
            vector_store=vector_store,
            storage_context=storage_context,
        )
        self._ready = True

    # ------------------------------------------------------------------ #
    # Ingesta de documentos                                                #
    # ------------------------------------------------------------------ #

    async def ingest_directory(self, path: str | Path | None = None) -> dict[str, Any]:
        """Indexa todos los archivos .md del directorio indicado (default: docs/).

        Se ejecuta en un hilo para no bloquear el event loop (LlamaIndex es síncrono).
        """
        if not self._ready:
            return {"status": "disabled", "indexed": 0}
        docs_path = Path(path) if path else _DOCS_DIR
        if not docs_path.is_dir():
            return {"status": "error", "detail": f"Directorio no encontrado: {docs_path}"}
        try:
            result = await asyncio.to_thread(self._ingest_dir_sync, docs_path)
            self._ingested_dirs.append(str(docs_path))
            return result
        except Exception as exc:  # noqa: BLE001
            logger.warning("rag.ingest.error: %s", exc)
            return {"status": "error", "detail": str(exc)}

    def _ingest_dir_sync(self, docs_path: Path) -> dict[str, Any]:
        from llama_index.core import SimpleDirectoryReader  # type: ignore[import-not-found]

        docs = SimpleDirectoryReader(
            input_dir=str(docs_path),
            required_exts=[".md", ".txt"],
            recursive=True,
        ).load_data()

        if not docs:
            return {"status": "ok", "indexed": 0, "detail": "No se encontraron archivos .md o .txt"}

        for doc in docs:
            self._index.insert(doc)

        return {"status": "ok", "indexed": len(docs), "path": str(docs_path)}

    async def ingest_texts(self, texts: list[dict[str, str]]) -> dict[str, Any]:
        """Indexa textos arbitrarios. Cada item: {"text": "...", "id": "...", "source": "..."}.

        Útil para indexar propuestas aprobadas, memorias del supervisor o estudios.
        """
        if not self._ready:
            return {"status": "disabled", "indexed": 0}
        if not texts:
            return {"status": "ok", "indexed": 0}
        try:
            result = await asyncio.to_thread(self._ingest_texts_sync, texts)
            return result
        except Exception as exc:  # noqa: BLE001
            logger.warning("rag.ingest_texts.error: %s", exc)
            return {"status": "error", "detail": str(exc)}

    def _ingest_texts_sync(self, texts: list[dict[str, str]]) -> dict[str, Any]:
        from llama_index.core import Document as LlamaDoc  # type: ignore[import-not-found]

        docs = [
            LlamaDoc(
                text=item["text"],
                doc_id=item.get("id", ""),
                metadata={"source": item.get("source", "manual"), "type": item.get("type", "text")},
            )
            for item in texts
            if item.get("text")
        ]
        for doc in docs:
            self._index.insert(doc)
        return {"status": "ok", "indexed": len(docs)}

    # ------------------------------------------------------------------ #
    # Retrieval                                                             #
    # ------------------------------------------------------------------ #

    async def query(self, text: str, top_k: int = 5) -> str:
        """Recupera los fragmentos más relevantes para `text`.

        Devuelve una cadena con los fragmentos concatenados, o "" si RAG no está listo.
        Diseñado para inyectarse en el system prompt del orchestrator.
        """
        if not self._ready or not text.strip():
            return ""
        try:
            return await asyncio.to_thread(self._query_sync, text, top_k)
        except Exception as exc:  # noqa: BLE001
            logger.warning("rag.query.error: %s", exc)
            return ""

    def _query_sync(self, text: str, top_k: int) -> str:
        retriever = self._index.as_retriever(similarity_top_k=top_k)
        nodes = retriever.retrieve(text)
        if not nodes:
            return ""
        fragments = []
        for node in nodes:
            source = node.metadata.get("source", "doc")
            score = f"{node.score:.3f}" if node.score is not None else "n/a"
            fragments.append(f"[{source} | score={score}]\n{node.get_content()[:600]}")
        return "\n\n---\n\n".join(fragments)

    # ------------------------------------------------------------------ #
    # Estado / health                                                       #
    # ------------------------------------------------------------------ #

    def status(self) -> dict[str, Any]:
        if not self._ready:
            return {"ready": False, "collection": self._collection}
        try:
            info = self._qdrant_client.get_collection(self._collection)
            count = info.points_count or 0
        except Exception:  # noqa: BLE001
            count = -1
        return {
            "ready": True,
            "collection": self._collection,
            "embed_model": self._embed_model_name,
            "vectors_count": count,
            "ingested_dirs": self._ingested_dirs,
        }


# Singleton global accesible desde main.py y orchestrator.py
rag_store = RagStore()
