"""Implementaciones intercambiables de localización terminológica."""

from __future__ import annotations

import hashlib
import re
import unicodedata
from collections.abc import Callable, Sequence
from datetime import UTC, datetime
from difflib import SequenceMatcher
from typing import Any


try:  # RapidFuzz es la implementación productiva; el fallback mantiene imports locales testeables.
    from rapidfuzz import fuzz
except ImportError:  # pragma: no cover - solo para entornos sin dependencias sincronizadas
    fuzz = None

from app.clinical_pipeline.domain.models import (
    CanonicalCatalogEntry,
    CanonicalMatchMethod,
    CanonicalOccurrence,
    PdfPage,
)


NORMALIZATION_VERSION = "v2"
STRATEGY_VERSION = "v2"
FUZZY_MIN_SCORE = 82.0
FUZZY_MIN_MARGIN = 5.0


def normalize_canonical_text(value: Any) -> str:
    """Normaliza Unicode, acentos, puntuación y espacios sin alterar la fuente."""

    decomposed = unicodedata.normalize("NFKD", str(value or ""))
    without_marks = "".join(char for char in decomposed if not unicodedata.combining(char))
    return re.sub(r"[^\w]+", " ", without_marks, flags=re.UNICODE).casefold().strip()


def _fuzzy_score(left: str, right: str) -> float:
    if fuzz is not None:
        return float(fuzz.ratio(left, right))
    return round(SequenceMatcher(None, left, right).ratio() * 100, 4)


def _entry(value: CanonicalCatalogEntry | dict[str, Any] | str) -> CanonicalCatalogEntry:
    if isinstance(value, CanonicalCatalogEntry):
        return value
    if isinstance(value, str):
        return CanonicalCatalogEntry(
            canonical_id=value,
            canonical_term=value,
            aliases=[value],
            artifact_types=[],
        )
    return CanonicalCatalogEntry.model_validate(value)


def _page(value: Any, index: int) -> PdfPage:
    if isinstance(value, PdfPage):
        return value
    if isinstance(value, str):
        return PdfPage(
            page_number=index,
            text=value,
            extraction_method="legacy",
            has_text=bool(value.strip()),
            character_count=len(value),
        )
    payload = dict(value or {})
    page_number = int(payload.get("page_number") or payload.get("page") or index)
    text = str(payload.get("text") or "")
    return PdfPage(
        page_number=page_number,
        text=text,
        extraction_method=str(payload.get("extraction_method") or ""),
        has_text=bool(payload.get("has_text", bool(text.strip()))),
        character_count=int(payload.get("character_count") or len(text)),
    )


def _occurrence_id(
    *,
    document_uid: str,
    artifact_type: str,
    canonical_id: str,
    page: int | None,
    position: int,
) -> str:
    raw = "|".join(
        (document_uid, artifact_type, canonical_id, str(page or ""), str(max(position, 0)))
    )
    return f"occ:{hashlib.sha256(raw.encode('utf-8')).hexdigest()[:24]}"


def _excerpt(text: str, start: int, end: int, limit: int = 500) -> str:
    left = max(0, start - 100)
    right = min(len(text), max(end, start) + 180)
    return re.sub(r"\s+", " ", text[left:right]).strip()[:limit]


def _context_satisfies(entry: CanonicalCatalogEntry, line: str, context: dict[str, Any]) -> bool:
    artifact_type = str(context.get("artifact_type") or "").strip()
    if artifact_type and entry.artifact_types and artifact_type not in entry.artifact_types:
        return False
    if entry.ambiguous_aliases:
        return True
    required = [normalize_canonical_text(item) for item in entry.required_context if str(item).strip()]
    context_text = normalize_canonical_text(f"{line} {context.get('section', '')}")
    return all(item in context_text for item in required)


class RapidFuzzCanonicalMatcher:
    """Matcher local por página con umbral y margen revisables."""

    strategy = "canonical"
    strategy_version = STRATEGY_VERSION

    def __init__(
        self,
        *,
        min_score: float = FUZZY_MIN_SCORE,
        min_margin: float = FUZZY_MIN_MARGIN,
        score_fn: Callable[[str, str], float] | None = None,
    ) -> None:
        self.min_score = float(min_score)
        self.min_margin = float(min_margin)
        self.score_fn = score_fn or _fuzzy_score

    def find_occurrences(
        self,
        document_pages: Sequence[Any],
        candidates: Sequence[CanonicalCatalogEntry | dict[str, Any] | str],
        context: dict[str, Any] | None = None,
    ) -> list[CanonicalOccurrence]:
        context = dict(context or {})
        entries = [_entry(item) for item in candidates]
        pages = [_page(item, index) for index, item in enumerate(document_pages, start=1)]
        occurrences: list[CanonicalOccurrence] = []
        found_candidate_ids: set[str] = set()
        text_pages = [page for page in pages if page.has_text and page.text.strip()]

        for page in text_pages:
            lines = list(re.finditer(r"[^\n]+", page.text)) or [re.match(r".*", page.text)]
            for line_match in lines:
                if line_match is None:
                    continue
                line = line_match.group(0)
                line_offset = line_match.start()
                exact_for_line: set[str] = set()
                for entry in entries:
                    if not entry.active or not _context_satisfies(entry, line, context):
                        continue
                    aliases = list(
                        dict.fromkeys([entry.canonical_term, *entry.aliases, *entry.ambiguous_aliases])
                    )
                    normalized_aliases = sorted(
                        ((normalize_canonical_text(alias), alias) for alias in aliases if str(alias).strip()),
                        key=lambda item: len(item[0]),
                        reverse=True,
                    )
                    normalized_line = normalize_canonical_text(line)
                    for normalized_alias, _alias in normalized_aliases:
                        if not normalized_alias:
                            continue
                        if self._is_shadowed_exact_alias(
                            normalized_line,
                            normalized_alias,
                            entry,
                            entries,
                            context,
                        ):
                            continue
                        boundary_pattern = re.compile(
                            rf"(?<!\w){re.escape(normalized_alias)}(?!\w)"
                        )
                        boundary_match = boundary_pattern.search(normalized_line)
                        start = boundary_match.start() if boundary_match else -1
                        while start >= 0:
                            original_match = self._original_match(line, normalized_alias, start)
                            original_term = original_match.group(0) if original_match else normalized_alias
                            absolute_position = line_offset + (original_match.start() if original_match else start)
                            is_ambiguous = any(
                                normalize_canonical_text(item) == normalized_alias
                                for item in entry.ambiguous_aliases
                            ) and not self._has_ambiguous_context(entry, line, context)
                            raw_alias_match = re.search(
                                re.escape(original_term), line, flags=re.IGNORECASE
                            )
                            method: CanonicalMatchMethod = (
                                "ambiguous"
                                if is_ambiguous
                                else "exact_alias"
                                if raw_alias_match
                                else "normalized_exact"
                            )
                            occurrence = self._build_occurrence(
                                entry=entry,
                                context=context,
                                page=page,
                                original_term=original_term,
                                excerpt=_excerpt(line, max(0, start), max(0, start) + len(original_term)),
                                method=method,
                                score=None if is_ambiguous else 100.0,
                                margin=None,
                                position=absolute_position,
                            )
                            occurrence.automatic["candidates"] = [
                                {
                                    "canonical_id": entry.canonical_id,
                                    "canonical_term": entry.canonical_term,
                                    "artifact_type": context.get("artifact_type") or "",
                                    "score": occurrence.match_score,
                                    "method": method,
                                }
                            ]
                            occurrences.append(occurrence)
                            exact_for_line.add(entry.canonical_id)
                            found_candidate_ids.add(entry.canonical_id)
                            next_match = boundary_pattern.search(normalized_line, start + 1)
                            start = next_match.start() if next_match else -1

                if exact_for_line:
                    continue
                fuzzy_matches = self._fuzzy_line_matches(line, entries, context)
                if not fuzzy_matches:
                    continue
                first = fuzzy_matches[0]
                second_score = fuzzy_matches[1][1] if len(fuzzy_matches) > 1 else 0.0
                score = first[1]
                # Ambiguous aliases may be reviewed when matched literally, but
                # they must not bypass the fuzzy threshold.
                if score < self.min_score:
                    continue
                method: CanonicalMatchMethod = (
                    "fuzzy_candidate"
                    if score >= self.min_score and score - second_score >= self.min_margin
                    else "ambiguous"
                )
                fuzzy_original = self._original_span(line, first[2], first[3])
                occurrence = self._build_occurrence(
                    entry=first[0],
                    context=context,
                    page=page,
                    original_term=fuzzy_original,
                    excerpt=_excerpt(line, first[3], first[3] + len(fuzzy_original)),
                    method=method,
                    score=score,
                    margin=score - second_score if len(fuzzy_matches) > 1 else score,
                    position=line_offset + first[3],
                )
                occurrence.automatic["candidates"] = [
                    {
                        "canonical_id": candidate.canonical_id,
                        "canonical_term": candidate.canonical_term,
                        "artifact_type": context.get("artifact_type") or "",
                        "score": round(candidate_score, 4),
                        "method": "fuzzy_candidate",
                    }
                    for candidate, candidate_score, _window, _position in fuzzy_matches
                    if candidate_score >= self.min_score
                ]
                occurrences.append(occurrence)
                found_candidate_ids.add(first[0].canonical_id)

        blank_pages = [page for page in pages if not page.has_text or not page.text.strip()]
        for page in blank_pages:
            for entry in entries:
                context_available = not entry.required_context or self._has_ambiguous_context(entry, "", context)
                if entry.active and context_available and _context_satisfies(entry, "", context):
                    occurrences.append(
                        self._build_occurrence(
                            entry=entry,
                            context=context,
                            page=page,
                            original_term="",
                            excerpt="",
                            method="ocr_required",
                            score=None,
                            margin=None,
                            position=0,
                        )
                    )

        if not text_pages and not blank_pages:
            for entry in entries:
                occurrences.append(
                    self._build_occurrence(
                        entry=entry,
                        context=context,
                        page=None,
                        original_term="",
                        excerpt="",
                        method="not_found",
                        score=None,
                        margin=None,
                        position=0,
                    )
                )
        else:
            for entry in entries:
                if entry.canonical_id not in found_candidate_ids and not blank_pages:
                    occurrences.append(
                        self._build_occurrence(
                            entry=entry,
                            context=context,
                            page=None,
                            original_term="",
                            excerpt="",
                            method="not_found",
                            score=None,
                            margin=None,
                            position=0,
                        )
                    )
        return self._deduplicate(occurrences)

    def _fuzzy_line_matches(
        self,
        line: str,
        entries: list[CanonicalCatalogEntry],
        context: dict[str, Any],
    ) -> list[tuple[CanonicalCatalogEntry, float, str, int]]:
        normalized_line = normalize_canonical_text(line)
        words = normalized_line.split()
        if not words:
            return []
        matches: list[tuple[CanonicalCatalogEntry, float, str, int]] = []
        for entry in entries:
            if not entry.active or not _context_satisfies(entry, line, context):
                continue
            target = normalize_canonical_text(entry.canonical_term)
            target_words = target.split()
            if not target_words:
                continue
            best = (0.0, "", 0)
            for size in range(max(1, len(target_words) - 1), min(len(words), len(target_words) + 1) + 1):
                for start in range(0, len(words) - size + 1):
                    window = " ".join(words[start : start + size])
                    score = self.score_fn(window, target)
                    if score > best[0]:
                        best = (score, window, start)
            if best[0] > 0:
                original = best[1]
                original_position = normalize_canonical_text(line).find(best[1])
                matches.append((entry, best[0], original, max(original_position, 0)))
        return sorted(matches, key=lambda item: item[1], reverse=True)

    @staticmethod
    def _is_shadowed_exact_alias(
        normalized_line: str,
        normalized_alias: str,
        entry: CanonicalCatalogEntry,
        entries: list[CanonicalCatalogEntry],
        context: dict[str, Any],
    ) -> bool:
        """Avoid a generic exact alias duplicating a more specific catalog match."""
        for other in entries:
            if other.canonical_id == entry.canonical_id or not other.active:
                continue
            if not _context_satisfies(other, normalized_line, context):
                continue
            aliases = [other.canonical_term, *other.aliases, *other.ambiguous_aliases]
            for alias in aliases:
                normalized_other = normalize_canonical_text(alias)
                if len(normalized_other) <= len(normalized_alias):
                    continue
                if not re.search(rf"(?<!\w){re.escape(normalized_other)}(?!\w)", normalized_line):
                    continue
                if re.search(rf"(?<!\w){re.escape(normalized_alias)}(?!\w)", normalized_other):
                    return True
        return False

    @staticmethod
    def _original_span(line: str, normalized_term: str, normalized_start: int) -> str:
        match = RapidFuzzCanonicalMatcher._original_match(line, normalized_term, normalized_start)
        if match:
            return match.group(0)
        normalized = normalize_canonical_text(line)
        return normalized[normalized_start : normalized_start + len(normalized_term)]

    @staticmethod
    def _original_match(line: str, normalized_term: str, normalized_start: int) -> re.Match[str] | None:
        parts = [part for part in normalized_term.split() if part]
        if not parts:
            return None
        matches = list(
            re.finditer(r"[\W_]+".join(re.escape(part) for part in parts), line, re.IGNORECASE)
        )
        if not matches:
            return None
        return min(
            matches,
            key=lambda candidate: abs(
                len(normalize_canonical_text(line[: candidate.start()])) - normalized_start
            ),
        )

    @staticmethod
    def _has_ambiguous_context(
        entry: CanonicalCatalogEntry,
        line: str,
        context: dict[str, Any],
    ) -> bool:
        required = [normalize_canonical_text(item) for item in entry.required_context if str(item).strip()]
        available = normalize_canonical_text(f"{line} {context.get('section', '')}")
        return bool(required) and all(item in available for item in required)

    def _build_occurrence(
        self,
        *,
        entry: CanonicalCatalogEntry,
        context: dict[str, Any],
        page: PdfPage | None,
        original_term: str,
        excerpt: str,
        method: CanonicalMatchMethod,
        score: float | None,
        margin: float | None,
        position: int,
    ) -> CanonicalOccurrence:
        incident = method in {"fuzzy_candidate", "ambiguous", "ocr_required"}
        return CanonicalOccurrence(
            occurrence_id=_occurrence_id(
                document_uid=str(
                    context.get("document_uid")
                    or context.get("document_id")
                    or context.get("source_hash")
                    or ""
                ),
                artifact_type=str(context.get("artifact_type") or ""),
                canonical_id=entry.canonical_id,
                page=page.page_number if page else None,
                position=position,
            ),
            case_key=str(context.get("case_key") or ""),
            document_uid=str(context.get("document_uid") or ""),
            document_id=str(context.get("document_id") or ""),
            source_hash=str(context.get("source_hash") or ""),
            document_type=str(context.get("document_type") or ""),
            filename=str(context.get("filename") or ""),
            artifact_type=str(context.get("artifact_type") or ""),
            page=page.page_number if page else None,
            section=str(context.get("section") or ""),
            excerpt=excerpt,
            original_term=original_term,
            canonical_term=entry.canonical_term,
            canonical_id=entry.canonical_id,
            match_method=method,
            match_score=round(score, 4) if score is not None else None,
            match_margin=round(margin, 4) if margin is not None else None,
            normalization_version=NORMALIZATION_VERSION,
            extraction_method=page.extraction_method if page else "",
            status=(
                "pending_canonical_review"
                if incident
                else "not_found"
                if method == "not_found"
                else "confirmed"
            ),
            automatic={
                "applied": False,
                "strategy": self.strategy,
                "strategy_version": self.strategy_version,
                "timestamp": datetime.now(UTC).isoformat(),
                "reason": method,
                "suggested_value": entry.canonical_term,
                "related_occurrences": [],
            },
        )

    @staticmethod
    def _deduplicate(values: list[CanonicalOccurrence]) -> list[CanonicalOccurrence]:
        seen: set[str] = set()
        result: list[CanonicalOccurrence] = []
        for value in values:
            if value.occurrence_id in seen:
                continue
            seen.add(value.occurrence_id)
            result.append(value)
        return result


class LegacyCanonicalMatcher(RapidFuzzCanonicalMatcher):
    """Adaptador de compatibilidad: conserva búsqueda literal y no aplica fuzzy."""

    strategy = "legacy"

    def __init__(self, **_kwargs: Any) -> None:
        super().__init__(min_score=101, min_margin=0, score_fn=lambda _left, _right: 0.0)
