"""Resolución determinística de asociaciones canónicas con evidencia."""

from __future__ import annotations

import math
import unicodedata
from collections.abc import Sequence
from typing import Any

from app.clinical_pipeline.domain.models import (
    CanonicalAssociationCandidate,
    CanonicalAssociationDecision,
    CanonicalCatalogEntry,
    CanonicalOccurrence,
)


MIN_SIMILARITY_SCORE = 82.0
MIN_SIMILARITY_MARGIN = 5.0
RESOLVER_VERSION = "v1"


def _text(value: Any) -> str:
    return str(value or "").strip()


def _normalized(value: Any) -> str:
    decomposed = unicodedata.normalize("NFKD", _text(value))
    return "".join(char for char in decomposed if not unicodedata.combining(char)).casefold()


def _candidate_values(occurrence: CanonicalOccurrence) -> list[dict[str, Any]]:
    raw = occurrence.automatic.get("candidates")
    if not isinstance(raw, list):
        return [
            {
                "canonical_id": occurrence.canonical_id,
                "canonical_term": occurrence.canonical_term,
                "score": occurrence.match_score,
                "margin": occurrence.match_margin,
                "method": occurrence.match_method,
            }
        ]
    values = [dict(item) for item in raw if isinstance(item, dict) and _text(item.get("canonical_id"))]
    return values or [
        {
            "canonical_id": occurrence.canonical_id,
            "canonical_term": occurrence.canonical_term,
            "score": occurrence.match_score,
            "margin": occurrence.match_margin,
            "method": occurrence.match_method,
        }
    ]


def _evidence_for(occurrence: CanonicalOccurrence) -> list[dict[str, Any]]:
    evidence = occurrence.to_document()
    return [
        {
            key: evidence[key]
            for key in (
                "occurrence_id",
                "document_uid",
                "document_id",
                "source_hash",
                "filename",
                "page",
                "section",
                "excerpt",
                "document_type",
                "artifact_type",
            )
            if evidence.get(key) not in (None, "")
        }
    ]


def _has_attributable_evidence(occurrence: CanonicalOccurrence) -> bool:
    document = any(
        _text(getattr(occurrence, field))
        for field in ("document_uid", "document_id", "source_hash", "filename")
    )
    anchor = bool(occurrence.page or _text(occurrence.section) or _text(occurrence.excerpt))
    return document and anchor


def _candidate_model(
    raw: dict[str, Any], entry: CanonicalCatalogEntry | None, occurrence: CanonicalOccurrence
) -> CanonicalAssociationCandidate:
    return CanonicalAssociationCandidate(
        canonical_id=_text(raw.get("canonical_id") or (entry.canonical_id if entry else "")),
        canonical_term=_text(raw.get("canonical_term") or (entry.canonical_term if entry else "")),
        artifact_type=_text(raw.get("artifact_type") or (entry.artifact_types[0] if entry and entry.artifact_types else occurrence.artifact_type)),
        score=raw.get("score", occurrence.match_score),
        margin=raw.get("margin", occurrence.match_margin),
        method=_text(raw.get("method") or occurrence.match_method),
    )


class DeterministicCanonicalAssociationResolver:
    """Combines similarity, catalog/context rules and documentary evidence.

    Scores are similarities in the ``0..100`` range.  This resolver deliberately
    has no probability field or calibration step; a future probabilistic adapter
    can implement the same port without changing this contract.
    """

    version = RESOLVER_VERSION

    def __init__(
        self,
        *,
        min_score: float = MIN_SIMILARITY_SCORE,
        min_margin: float = MIN_SIMILARITY_MARGIN,
    ) -> None:
        self.min_score = float(min_score)
        self.min_margin = float(min_margin)

    def resolve(
        self,
        occurrence: CanonicalOccurrence,
        candidates: Sequence[CanonicalCatalogEntry],
        context: dict[str, Any] | None = None,
    ) -> CanonicalAssociationDecision:
        context = dict(context or {})
        entries = {entry.canonical_id: entry for entry in candidates}
        raw_candidates = _candidate_values(occurrence)
        candidate_models = [
            _candidate_model(raw, entries.get(_text(raw.get("canonical_id"))), occurrence)
            for raw in raw_candidates
        ]
        selected = entries.get(occurrence.canonical_id)
        reasons: list[str] = []
        conflicts = list(context.get("conflicts") or ())
        if context.get("description_conflict") or context.get("external_blocked"):
            conflicts.append("context_conflict")
        catalog_available = context.get("catalog_available", True)
        catalog = _text(context.get("catalog")) or "canonical"

        if selected is None:
            reasons.append("candidate_not_present_in_catalog")
        elif occurrence.artifact_type and selected.artifact_types and occurrence.artifact_type not in selected.artifact_types:
            reasons.append("artifact_type_incompatible")
        elif not selected.is_leaf:
            reasons.append("candidate_not_leaf")
        if not catalog_available:
            reasons.append("catalog_unavailable")
        expected_catalog = _text(context.get("expected_catalog"))
        if expected_catalog and catalog != expected_catalog:
            reasons.append("catalog_incompatible")
        available_catalogs = context.get("available_catalogs")
        if available_catalogs and catalog not in set(available_catalogs):
            reasons.append("catalog_unavailable")
        expected_catalog_version = _text(context.get("catalog_version"))
        if expected_catalog_version and selected is not None and selected.version != expected_catalog_version:
            reasons.append("catalog_version_mismatch")
        if selected is not None:
            available_context = _normalized(
                " ".join(
                    (
                        occurrence.original_term,
                        occurrence.excerpt,
                        occurrence.section,
                        _text(context.get("context_text")),
                    )
                )
            )
            missing_context = [
                item
                for item in selected.required_context
                if _normalized(item) and _normalized(item) not in available_context
            ]
            if missing_context:
                reasons.append("required_context_missing")

            allowed_documents = context.get("compatible_document_types")
            if isinstance(allowed_documents, dict):
                allowed_documents = allowed_documents.get(occurrence.artifact_type)
            if allowed_documents and occurrence.document_type not in set(allowed_documents):
                reasons.append("document_type_incompatible")

        if not _has_attributable_evidence(occurrence):
            reasons.append("evidence_not_attributable")

        score = occurrence.match_score
        margin = occurrence.match_margin
        if occurrence.match_method == "fuzzy_candidate":
            if score is None or not math.isfinite(float(score)) or float(score) < self.min_score:
                reasons.append("similarity_below_threshold")
            if margin is None or not math.isfinite(float(margin)) or float(margin) < self.min_margin:
                reasons.append("similarity_margin_insufficient")
        if occurrence.match_method == "ambiguous" or len(candidate_models) > 1:
            reasons.append("multiple_plausible_candidates")
        if occurrence.match_method == "ocr_required":
            reasons.append("ocr_required_before_association")
        if occurrence.match_method == "not_found":
            reasons.append("candidate_not_found")

        reasons.extend(_text(value) for value in conflicts if _text(value))
        unique_reasons = list(dict.fromkeys(reasons))
        status = "validado_automaticamente"
        if conflicts:
            status = "conflicto"
        elif unique_reasons:
            status = "pendiente_revision"
        reason = unique_reasons[0] if unique_reasons else "evidence_and_context_sufficient"
        return CanonicalAssociationDecision(
            status=status,
            selected_candidate_id=occurrence.canonical_id,
            selected_candidate_term=occurrence.canonical_term,
            candidates=candidate_models,
            method=occurrence.match_method,
            score=score,
            margin=margin,
            reasons=unique_reasons,
            reason=reason,
            evidences=_evidence_for(occurrence),
            catalog=catalog,
            catalog_version=selected.version if selected is not None else "",
            normalization_version=occurrence.normalization_version,
            resolver_version=self.version,
        )


__all__ = [
    "DeterministicCanonicalAssociationResolver",
    "MIN_SIMILARITY_MARGIN",
    "MIN_SIMILARITY_SCORE",
    "RESOLVER_VERSION",
]
