from __future__ import annotations

import json
import re
from typing import Any

from app.case_epicrisis.application.audit_traceability import project_auditoria_traceability
from app.case_epicrisis.domain.chronology import (
    build_clinical_trajectories,
    project_effective_date,
    sort_chronological_items,
)
from app.case_epicrisis.domain.models import AyudaDiagnostica
from app.case_epicrisis.domain.publication import evaluate_publication_eligibility
from app.services.clinical_document_projection import (
    get_structured_document_model,
    render_document_analysis_html,
)

from .common import (
    _clean_unique_text_list,
    _normalize_ascii,
    _normalize_bool,
    _normalize_whitespace,
    strip_html_tags,
)
from .factura import (
    _extract_html_field,
    _imagenologia_factura_display,
    _radiologia_interpretacion,
    _servicios_factura,
)


_AID_TYPES = {"imagen", "laboratorio", "procedimiento_diagnostico"}

_AID_TEXT_TERMS = (
    "rx",
    "rayos x",
    "radiografia",
    "radiografía",
    "radiografico",
    "radiográfico",
    "tac",
    "tomografia",
    "tomografía",
    "rm",
    "rnm",
    "rmn",
    "resonancia",
    "ecografia",
    "ecografía",
    "ultrasonido",
    "fluoroscopia",
    "fluoroscopía",
    "intensificador de imagen",
    "laboratorio",
    "laboratorios",
    "hemograma",
    "creatinina",
    "uroanalisis",
    "uroanálisis",
    "paraclinico",
    "paraclínico",
    "paraclinicos",
    "paraclínicos",
    "prequirurgicos",
    "prequirúrgicos",
    "electrocardiograma",
    "ekg",
    "ecg",
    "ayuda diagnostica",
    "ayuda diagnóstica",
    "estudio diagnostico",
    "estudio diagnóstico",
)

_AID_TEXT_PATTERN = re.compile(
    r"\b("
    r"rx|rayos\s+x|radiograf[ií]a|radiogr[aá]fico|tac|tomograf[ií]a|rm|rnm|rmn|resonancia|"
    r"ecograf[ií]a|ultrasonido|fluoroscop[ií]a|intensificador(?:es)?\s+de\s+imagen(?:es)?|"
    r"laboratorio(?:s)?|hemograma|creatinina|uroan[aá]lisis|paracl[ií]nico(?:s)?|"
    r"prequir[uú]rgico(?:s)?|electrocardiograma|ekg|ecg|ayuda\s+diagn[oó]stica|"
    r"estudio\s+diagn[oó]stico"
    r")\b",
    flags=re.IGNORECASE,
)

_AID_UNINTERPRETED_PLACEHOLDERS = {
    "",
    "no interpretado",
    "no interpretada",
    "sin interpretacion",
    "sin interpretación",
    "pendiente",
}

_AID_STATE_PRIORITY = {
    "no_interpretado": 0,
    "interpretado": 1,
    "pendiente_revision": 2,
    "excluido": 3,
}
_AID_REVIEW_ALERTS = {
    "pending_canonical_review",
    "pendiente_revision",
    "conflicto",
    "asociacion_canonica_no_confirmada",
    "asociacion_canonica_pendiente_revision",
    "evidencia_contradictoria",
    "contradiccion",
    "contradictorio",
}
_CLINICAL_AID_SOURCES = frozenset(
    {
        "historia_clinica",
        "radiologia",
        "laboratorio",
        "quirurgico",
        "generico",
        "agente_hallazgos_qx",
    }
)
_INVOICE_SUPPORT_ALERT = "factura_sin_soporte_clinico_asociado"
_INVOICE_CONTRADICTION_ALERT = "evidencia_contradictoria"
_INVOICE_SUPPORT_INCIDENT = "factura_sin_soporte_clinico_asociado"
_DIAGNOSTIC_ARTIFACT_TYPE = "ayuda_diagnostica"
DIAGNOSTIC_AID_CONSOLIDATION_VERSION = "v3"
DIAGNOSTIC_AID_RULES_VERSION = "v3"
_REJECTED_ARTIFACT_TYPES = {
    "medicamento",
    "medicamentos",
    "insumo",
    "insumos",
    "suministro",
    "suministros",
    "administrativo",
    "narrativa_clinica",
}
_NARRATIVE_STOP_WORDS = {
    "a",
    "al",
    "con",
    "como",
    "de",
    "del",
    "el",
    "en",
    "e",
    "la",
    "las",
    "los",
    "o",
    "para",
    "por",
    "se",
    "su",
    "un",
    "una",
    "y",
    "ya",
}
_NARRATIVE_BOUNDARY_WORDS = {
    "administra",
    "administran",
    "administrado",
    "aplica",
    "aplicaron",
    "continua",
    "continúa",
    "consume",
    "consumió",
    "formula",
    "formulan",
    "formulado",
    "indica",
    "indican",
    "inicia",
    "inició",
    "recibe",
    "recibió",
    "solicita",
    "solicitó",
    "toma",
    "tomó",
    "antes",
    "después",
    "posterior",
    "previo",
    "previa",
}
_MEDICATION_OR_SUPPLY_MARKERS = re.compile(
    r"\b(?:medicamentos?|f[aá]rmacos?|insumos?|suministros?|material(?:es)?|dosis|posolog[ií]a|"
    r"cantidad|unidades?|tabletas?|ampollas?|c[aá]psulas?|mg|ml|cc|administrad[oa]s?)\b",
    re.IGNORECASE,
)
_ADMINISTRATIVE_MARKERS = re.compile(
    r"\b(?:subtotal|total|iva|copago|autorizaci[oó]n|factura|valor unitario|precio|contrato)\b",
    re.IGNORECASE,
)
_EVIDENCE_RESULT_MARKERS = re.compile(
    r"\b(resultado|resultados|hallazgo|hallazgos|conclusion|conclusión|interpretacion|interpretación|"
    r"sin\s+(lesion|lesión|alteraciones|evidencia)|descartad[oa]|fractura|consolidacion|consolidación)\b",
    re.IGNORECASE,
)
_EVIDENCE_ORDER_MARKERS = re.compile(
    r"\b(se\s+solicita|solicitud|ordenad[oa]|pendiente|toma\s+de|para\s+descartar)\b",
    re.IGNORECASE,
)
_EVIDENCE_NUMBER_MARKER = re.compile(r"\b\d+(?:[.,]\d+)?\b")
_EXCLUDED_BLOCK_PREFIX = re.compile(
    r"^\s*(?:medicamentos?|f[aá]rmacos?|insumos?|suministros?|material(?:es)?|dosis|posolog[ií]a|"
    r"cantidad|unidades?|subtotal|total|iva|copago|autorizaci[oó]n|factura|valor unitario|precio)\s*:",
    re.IGNORECASE,
)


def _evidence_role(
    *,
    source: Any,
    excerpt: Any = "",
    structured: bool = False,
    occurrence: dict[str, Any] | None = None,
) -> str:
    """Classify evidence without turning a mention into a clinical result."""
    raw_occurrence = occurrence if isinstance(occurrence, dict) else {}
    explicit_stage = _normalize_ascii(
        raw_occurrence.get("clinical_stage")
        or raw_occurrence.get("stage")
        or raw_occurrence.get("etapa_clinica")
        or ""
    ).casefold()
    if explicit_stage:
        if any(token in explicit_stage for token in ("orden", "solic", "pendiente")):
            return "orden"
        if any(token in explicit_stage for token in ("realiz", "tomad", "ejecut")):
            return "realizado"
        if any(token in explicit_stage for token in ("result", "report", "interpret", "hallaz")):
            return "resultado_narrativo"
        if any(token in explicit_stage for token in ("admin", "factur", "registro")):
            return "registro_administrativo"

    source_text = _normalize_ascii(source).casefold()
    excerpt_text = _normalize_whitespace(excerpt)
    normalized_excerpt = _normalize_ascii(excerpt_text).casefold()
    if source_text == "factura" or _ADMINISTRATIVE_MARKERS.search(excerpt_text):
        return "registro_administrativo"
    if re.match(r"^\s*\d{5,6}\s*[-:]", excerpt_text):
        return "registro_administrativo"
    if _EVIDENCE_ORDER_MARKERS.search(excerpt_text):
        return "orden"
    if structured:
        return "resultado_narrativo"
    if _EVIDENCE_RESULT_MARKERS.search(excerpt_text) or _EVIDENCE_NUMBER_MARKER.search(excerpt_text):
        return "resultado_narrativo"
    if normalized_excerpt:
        return "realizado"
    return "resultado_narrativo"


def _normalized_ayuda_tipo(value: Any) -> str:
    normalized = _normalize_ascii(value).lower().replace("-", "_").replace(" ", "_")
    aliases = {
        "imagenes": "imagen",
        "imagenologia": "imagen",
        "radiologia": "imagen",
        "radiologico": "imagen",
        "radiografico": "imagen",
        "rx": "imagen",
        "lab": "laboratorio",
        "laboratorios": "laboratorio",
        "procedimiento": "procedimiento_diagnostico",
        "procedimiento_diagnóstico": "procedimiento_diagnostico",
        "electrocardiograma": "procedimiento_diagnostico",
        "ekg": "procedimiento_diagnostico",
        "ecg": "procedimiento_diagnostico",
    }
    normalized = aliases.get(normalized, normalized)
    return normalized if normalized in _AID_TYPES else ""


def _normalize_ayuda_tipo(value: Any) -> str:
    """Return a safe legacy type while retaining unknown-type metadata separately."""
    return _normalized_ayuda_tipo(value) or "imagen"


def _normalize_ayuda_key_text(value: Any) -> str:
    text = _normalize_ascii(value).lower()
    text = re.sub(r"[^a-z0-9]+", "-", text)
    return text.strip("-")


def _normalize_key_for_terms(value: Any) -> str:
    return re.sub(r"\s+", " ", _normalize_ascii(value).lower()).strip()


def _positive_page(value: Any) -> int | None:
    try:
        page = int(value)
    except (TypeError, ValueError):
        return None
    return page if page >= 1 else None


def _diagnostic_evidence_excerpt(
    documento: dict[str, Any],
    *,
    original_term: str,
    fuente: str,
) -> tuple[str, str]:
    """Build a bounded clinical excerpt without sending the whole document to the LLM."""
    structured = get_structured_document_model(documento)
    structured_payload: dict[str, Any] = {}
    if structured is not None:
        structured_payload = structured.model_dump(mode="json", exclude_none=True)

    raw_parts = [
        str(documento.get("descripcion") or ""),
        str(documento.get("texto_extraido") or ""),
        strip_html_tags(str(documento.get("analisis_html") or documento.get("analisis") or "")),
    ]
    if structured_payload:
        raw_parts.insert(0, json.dumps(structured_payload, ensure_ascii=False, separators=(",", ":")))
    source_text = _normalize_whitespace(" ".join(part for part in raw_parts if _normalize_whitespace(part)))
    if not source_text:
        return "", "reporte_estructurado" if structured_payload else "referencia_narrativa"

    normalized_source = _normalize_ascii(source_text).casefold()
    normalized_term = _normalize_ascii(original_term).casefold()
    position = normalized_source.find(normalized_term) if normalized_term else -1
    if not structured_payload:
        # A narrative reference is evidence of a mention, not a clinical report.
        # Keep the source anchor useful for audit while preventing nearby orders,
        # medications, or a complete clinical summary from becoming the aid.
        fragments = re.split(r"[\n.;]+", source_text)
        for fragment in fragments:
            fragment = _normalize_whitespace(fragment)
            if not fragment:
                continue
            fragment_normalized = _normalize_ascii(fragment).casefold()
            if normalized_term and normalized_term in fragment_normalized:
                fragment_position = fragment_normalized.find(normalized_term)
                before = fragment[:fragment_position].strip(" ,:-")[-90:]
                after = fragment[fragment_position + len(original_term) :].strip(" ,:-")
                after = re.split(
                    r"\b(?:y|e|pero|adem[aá]s)\s+(?:se\s+)?(?:administra|recibe|toma|formula|indica)\b",
                    after,
                    maxsplit=1,
                    flags=re.IGNORECASE,
                )[0]
                after = re.split(
                    r"\s+(?:medicamentos?|insumos?|suministros?)\s*:", after, maxsplit=1, flags=re.IGNORECASE
                )[0]
                after = re.split(
                    r"\b(?:medicamentos?|insumos?|suministros?|dosis|cantidad|unidades?|mg|ml|cc)\b",
                    after,
                    maxsplit=1,
                    flags=re.IGNORECASE,
                )[0]
                bounded = _normalize_whitespace(
                    " ".join(part for part in (before, original_term, after[:100]) if part)
                )
                return bounded[:280], "referencia_narrativa"
        if position < 0:
            return source_text[:280], "referencia_narrativa"
    if position >= 0:
        start = max(0, position - 500)
        excerpt = source_text[start : start + 1400]
    else:
        excerpt = source_text[:1400]
    evidence_kind = "reporte_estructurado" if structured_payload else "referencia_narrativa"
    if _normalize_ascii(fuente).casefold() == "factura":
        evidence_kind = "factura"
    return _normalize_whitespace(excerpt), evidence_kind


def _build_document_evidence(
    *,
    documento: dict[str, Any],
    fuente: str,
    original_term: str,
    interpretacion_estructurada: bool,
    candidate_id: str = "",
    artifact_type: str = _DIAGNOSTIC_ARTIFACT_TYPE,
) -> dict[str, Any] | None:
    documento_id = _normalize_whitespace(documento.get("_id") or documento.get("id"))
    document_uid = _normalize_whitespace(documento.get("document_uid"))
    documento_nombre = _normalize_whitespace(
        documento.get("nombre_archivo") or documento.get("filename") or documento.get("archivo")
    )
    page = _positive_page(documento.get("pagina") or documento.get("page"))
    section = _normalize_whitespace(documento.get("seccion") or documento.get("section"))
    if not any((fuente, documento_id, document_uid, documento_nombre, original_term)):
        return None
    evidence_scope = document_uid or documento_id or documento_nombre or fuente or "desconocida"
    excerpt, evidence_kind = _diagnostic_evidence_excerpt(
        documento,
        original_term=original_term,
        fuente=fuente,
    )
    stable_scope = ":".join(
        value
        for value in (
            _normalize_ayuda_key_text(fuente),
            _normalize_ayuda_key_text(evidence_scope),
            _normalize_ayuda_key_text(original_term),
            str(page or ""),
            _normalize_ayuda_key_text(section),
        )
        if value
    )
    return {
        "evidencia_id": f"documento:{stable_scope}",
        "artifact_type": artifact_type,
        "fuente": _normalize_whitespace(fuente),
        "documento_id": documento_id,
        "document_uid": document_uid,
        "documento_nombre": documento_nombre,
        "pagina": page,
        "seccion": section,
        "termino_original": _normalize_whitespace(original_term),
        "document_type": _normalize_whitespace(fuente),
        "filename": documento_nombre,
        "page": page,
        "section": section,
        "razon_asociacion": (
            "Soporte documental de la ayuda diagnóstica."
            if artifact_type == _DIAGNOSTIC_ARTIFACT_TYPE
            else "Evidencia conservada para explicar la exclusión de la ayuda diagnóstica."
        ),
        "extracto": excerpt,
        "tipo_evidencia": evidence_kind,
        "evidence_role": _evidence_role(
            source=fuente,
            excerpt=excerpt,
            structured=interpretacion_estructurada,
        ),
        "rol_evidencia": _evidence_role(
            source=fuente,
            excerpt=excerpt,
            structured=interpretacion_estructurada,
        ),
        "candidate_id": candidate_id,
        "interpretacion_estructurada": bool(interpretacion_estructurada),
    }


def _dump_ayuda_model(ayuda: dict[str, Any]) -> dict[str, Any]:
    """Validate the common contract while returning the legacy dictionary shape."""
    validated = AyudaDiagnostica.model_validate(ayuda)
    serialized = validated.model_dump(mode="json", exclude_none=False)
    # Canonical occurrences historically use English matcher keys. Keep those
    # keys in the read model while the domain validation accepts both dialects.
    serialized["evidencias"] = [
        dict(item) if isinstance(item, dict) else item for item in (ayuda.get("evidencias") or [])
    ]
    return serialized


def _build_ayuda_key(
    *,
    tipo: str,
    nombre: str,
    fuente: str,
    documento_id: str = "",
) -> str:
    base = _normalize_ayuda_key_text(nombre) or "sin-nombre"
    scope = _normalize_ayuda_key_text(documento_id or fuente) or "sin-fuente"
    return f"ayuda:{tipo}:{base}:{scope}"


def evaluar_ayuda_diagnostica_exportable(ayuda: dict[str, Any], excluded_keys: Any = None) -> bool:
    excluded = set(_clean_unique_text_list(excluded_keys))
    key = _normalize_whitespace(ayuda.get("key"))
    if key and key in excluded:
        return False
    if _normalize_bool(ayuda.get("glosado")):
        return False
    incidents = [incident for incident in ayuda.get("incidencias") or [] if isinstance(incident, dict)]
    blocking_incidents = [
        incident for incident in incidents if incident.get("codigo") != _INVOICE_SUPPORT_INCIDENT
    ]
    if blocking_incidents:
        return False
    return _normalize_ascii(ayuda.get("estado_interpretacion")).lower() != "excluido"


def _build_ayuda_diagnostica(
    *,
    tipo: str,
    nombre: Any,
    concepto: Any = "",
    fuente: str,
    documento: dict[str, Any] | None = None,
    ordenado: bool = False,
    interpretado: bool = False,
    facturado: bool = False,
    glosado: bool = False,
    estado_interpretacion: str = "",
    razon_inclusion: str = "",
    alertas: Any = None,
    clinical_date: Any = None,
    interpretacion_estructurada: bool = False,
    candidate_id: str = "",
) -> dict[str, Any] | None:
    tipo_original = _normalize_whitespace(tipo)
    tipo_normalizado = _normalized_ayuda_tipo(tipo_original) or "imagen"
    classification_status = "accepted" if _normalized_ayuda_tipo(tipo_original) else "unclassified"
    classification_reason = (
        "Referencia clasificada como ayuda diagnóstica."
        if classification_status == "accepted"
        else "El tipo original de la ayuda diagnóstica no se pudo mapear de forma confiable."
    )
    nombre_texto = _normalize_whitespace(nombre)
    concepto_texto = _normalize_whitespace(concepto)
    if not nombre_texto:
        return None

    documento = documento if isinstance(documento, dict) else {}
    effective_date = project_effective_date(documento, clinical_date=clinical_date)
    documento_id = _normalize_whitespace(documento.get("_id") or documento.get("id"))
    documento_nombre = _normalize_whitespace(
        documento.get("nombre_archivo") or documento.get("filename") or documento.get("archivo")
    )
    estado = _normalize_ascii(estado_interpretacion).lower() if estado_interpretacion else ""
    if not estado:
        estado = "interpretado" if interpretado else "no_interpretado"
    if estado not in {"interpretado", "no_interpretado", "pendiente_revision", "excluido"}:
        estado = "pendiente_revision"
    interpretado = estado == "interpretado" or bool(interpretado)
    if estado == "no_interpretado" and not concepto_texto:
        concepto_texto = "No interpretado"

    normalized_alerts = _clean_unique_text_list(alertas)
    if estado == "no_interpretado" and "ayuda_no_interpretada" not in normalized_alerts:
        normalized_alerts.append("ayuda_no_interpretada")
    if (
        _normalize_whitespace(documento.get("texto_no_extraible"))
        and "texto_no_extraible" not in normalized_alerts
    ):
        normalized_alerts.append("texto_no_extraible")

    ayuda = {
        "key": _build_ayuda_key(
            tipo=tipo_normalizado,
            nombre=nombre_texto,
            fuente=fuente,
            documento_id=documento_id,
        ),
        "artifact_type": _DIAGNOSTIC_ARTIFACT_TYPE,
        "candidate_id": candidate_id
        or f"aid-candidate:{_normalize_ayuda_key_text(fuente)}:{_normalize_ayuda_key_text(documento.get('document_uid') or documento_id or documento_nombre)}:{_normalize_ayuda_key_text(nombre_texto)}",
        "classification_status": classification_status,
        "classification_reason": classification_reason,
        "tipo_original": tipo_original,
        "tipo": tipo_normalizado,
        "nombre": nombre_texto,
        "concepto": concepto_texto or ("No interpretado" if estado == "no_interpretado" else ""),
        "fuente": _normalize_whitespace(fuente) or "desconocida",
        "documento_id": documento_id,
        "documento_nombre": documento_nombre,
        "pagina": _positive_page(documento.get("pagina") or documento.get("page")),
        "seccion": _normalize_whitespace(documento.get("seccion") or documento.get("section")),
        "ordenado": bool(ordenado),
        "interpretado": bool(interpretado),
        "facturado": bool(facturado),
        "glosado": bool(glosado),
        "estado_interpretacion": estado,
        "interpretacion_estructurada": bool(interpretacion_estructurada),
        "razon_inclusion": _normalize_whitespace(razon_inclusion),
        "alertas": normalized_alerts,
        "case_key": _normalize_whitespace(documento.get("case_key")),
        "document_uid": _normalize_whitespace(documento.get("document_uid")),
        "evidencias": [],
        "normalizacion": {
            "strategy": (documento.get("canonical_processing") or {}).get("strategy") or "",
            "strategy_version": (documento.get("canonical_processing") or {}).get("strategy_version") or "",
            "normalization_version": (documento.get("canonical_processing") or {}).get(
                "normalization_version"
            )
            or "",
        },
        "incidencias": [],
        "effective_date": effective_date.effective_date,
        "date_source": effective_date.date_source.value,
        "date_precision": effective_date.date_precision.value,
        "date_warning": effective_date.warning,
    }
    base_evidence = _build_document_evidence(
        documento=documento,
        fuente=fuente,
        original_term=nombre_texto,
        interpretacion_estructurada=interpretacion_estructurada,
        candidate_id=ayuda["candidate_id"],
    )
    if base_evidence:
        ayuda["evidencias"].append(base_evidence)
    ayuda["termino_original"] = nombre_texto
    ayuda["terminos_originales"] = [nombre_texto]
    aid_text = _normalize_ascii(f"{nombre_texto} {concepto_texto}").lower()
    for occurrence in documento.get("canonical_occurrences") or []:
        if not isinstance(occurrence, dict) or occurrence.get("match_method") == "not_found":
            continue
        if (occurrence.get("artifact_type") or "") not in {"ayuda_diagnostica", "hallazgo_clinico", ""}:
            continue
        occurrence_terms = (occurrence.get("original_term"), occurrence.get("canonical_term"))
        if not any(
            normalized_term and normalized_term in aid_text
            for normalized_term in (_normalize_ascii(term).lower().strip() for term in occurrence_terms)
        ):
            # An occurrence belongs to the document, not automatically to every
            # aid extracted from that document.
            continue
        occurrence_projection = dict(occurrence)
        occurrence_projection.setdefault("artifact_type", _DIAGNOSTIC_ARTIFACT_TYPE)
        occurrence_projection.setdefault("candidate_id", ayuda.get("candidate_id") or "")
        occurrence_projection.setdefault("fuente", fuente)
        occurrence_projection.setdefault("documento_id", documento_id)
        occurrence_projection.setdefault("documento_nombre", documento_nombre)
        occurrence_projection.setdefault("document_uid", ayuda.get("document_uid"))
        occurrence_projection.setdefault("document_type", fuente)
        occurrence_projection.setdefault("filename", documento_nombre)
        occurrence_projection.setdefault("page", ayuda.get("pagina"))
        occurrence_projection.setdefault("section", ayuda.get("seccion"))
        evidence_role = _evidence_role(
            source=occurrence_projection.get("fuente") or fuente,
            excerpt=occurrence_projection.get("excerpt") or occurrence_projection.get("extracto"),
            structured=bool(occurrence_projection.get("interpretacion_estructurada")),
            occurrence=occurrence_projection,
        )
        occurrence_projection.setdefault("evidence_role", evidence_role)
        occurrence_projection.setdefault("rol_evidencia", evidence_role)
        for field in ("effective_date", "date_source", "date_precision", "date_warning"):
            if field not in occurrence_projection:
                occurrence_projection[field] = ayuda.get(field)
        automatic = occurrence_projection.get("automatic")
        automatic = automatic if isinstance(automatic, dict) else {}
        association_reason = _normalize_whitespace(
            occurrence_projection.get("association_reason")
            or occurrence_projection.get("razon_asociacion")
            or automatic.get("reason")
        )
        if not association_reason:
            method = _normalize_whitespace(occurrence_projection.get("match_method"))
            status = _normalize_whitespace(occurrence_projection.get("status"))
            association_reason = (
                "Asociación canónica pendiente de revisión."
                if status in {"pending_canonical_review", "pendiente_revision", "conflicto"}
                else f"Asociación canónica confirmada mediante {method}."
                if method
                else "Asociación canónica documentada."
            )
        occurrence_projection.setdefault("association_reason", association_reason)
        occurrence_projection.setdefault("razon_asociacion", association_reason)
        ayuda["evidencias"].append(occurrence_projection)
        original_term = _normalize_whitespace(occurrence.get("original_term"))
        if original_term and _normalize_key_for_terms(original_term) not in {
            _normalize_key_for_terms(term) for term in ayuda["terminos_originales"]
        }:
            ayuda["terminos_originales"].append(original_term)
        ayuda["normalizacion"] = {
            "canonical_term": occurrence.get("canonical_term") or "",
            "normalization_version": occurrence.get("normalization_version") or "",
            "match_method": occurrence.get("match_method") or "",
        }
        if occurrence.get("status") in {"pending_canonical_review", "pendiente_revision", "conflicto"}:
            ayuda["incidencias"].append(dict(occurrence))
            alert = occurrence.get("status") or "pending_canonical_review"
            if alert not in normalized_alerts:
                normalized_alerts.append(alert)
    ayuda["alertas"] = normalized_alerts
    ayuda["classification_status"] = classification_status
    if ayuda["incidencias"]:
        ayuda["estado_interpretacion"] = "pendiente_revision"
    _refresh_ayuda_trajectory(ayuda)
    ayuda["exportable"] = evaluar_ayuda_diagnostica_exportable(ayuda)
    return _dump_ayuda_model(ayuda)


def _is_uninterpreted_placeholder(value: Any) -> bool:
    return _normalize_ascii(value).lower() in _AID_UNINTERPRETED_PLACEHOLDERS


def _format_factura_ayuda_nombre(nombre: Any, descripcion: Any) -> str:
    nombre_texto = _normalize_whitespace(nombre)
    descripcion_texto = _normalize_whitespace(descripcion)
    if not nombre_texto:
        return descripcion_texto
    if (
        descripcion_texto
        and descripcion_texto.lower() != nombre_texto.lower()
        and descripcion_texto.lower() not in nombre_texto.lower()
    ):
        return f"{nombre_texto} - {descripcion_texto}"
    return nombre_texto


def format_ayuda_diagnostica_nombre_exportable(item: Any) -> str:
    if not isinstance(item, dict):
        return _normalize_whitespace(item)
    nombre = _normalize_whitespace(item.get("nombre"))
    concepto = _normalize_whitespace(item.get("concepto") or item.get("interpretacion"))
    fuente = _normalize_ascii(item.get("fuente")).lower()
    if (
        fuente == "factura"
        and not _normalize_bool(item.get("interpretado"))
        and concepto
        and not _is_uninterpreted_placeholder(concepto)
        and concepto.lower() != nombre.lower()
        and concepto.lower() not in nombre.lower()
    ):
        return f"{nombre} - {concepto}" if nombre else concepto
    return nombre


def split_ayuda_diagnostica_codigo_nombre(value: Any) -> tuple[str, str]:
    """Separate a numeric presentation prefix from an aid name without mutating the source."""
    text = str(value or "").strip()
    match = re.match(r"^(?P<codigo>\d+)\s*-\s*(?P<nombre>\S(?:.*\S)?)$", text)
    if not match:
        return "", text
    return match.group("codigo"), match.group("nombre")


def format_ayuda_diagnostica_interpretacion_exportable(item: Any) -> str:
    if not isinstance(item, dict) or not _normalize_bool(item.get("interpretado")):
        return ""
    interpretacion = _normalize_whitespace(item.get("concepto") or item.get("interpretacion"))
    return "" if _is_uninterpreted_placeholder(interpretacion) else interpretacion


def project_ayuda_diagnostica(item: Any) -> dict[str, Any]:
    """Return the shared safe projection consumed by screen, PDF and Excel."""
    raw = dict(item) if isinstance(item, dict) else {}
    state = _normalize_ayuda_state(raw.get("estado_interpretacion"))
    interpreted = state == "interpretado" and _normalize_bool(raw.get("interpretado"))
    nombre_presentacion = format_ayuda_diagnostica_nombre_exportable(raw)
    codigo, nombre = split_ayuda_diagnostica_codigo_nombre(nombre_presentacion)
    concept = format_ayuda_diagnostica_interpretacion_exportable(raw) if interpreted else "No interpretado"
    raw_type = _normalize_whitespace(raw.get("tipo"))
    normalized_type = _normalized_ayuda_tipo(raw_type)
    classification_status = _normalize_whitespace(raw.get("classification_status")) or (
        "accepted" if normalized_type else "unclassified"
    )
    projection = {
        **raw,
        "tipo": normalized_type,
        "tipo_original": raw.get("tipo_original") or raw_type,
        "classification_status": classification_status,
        "classification_reason": raw.get("classification_reason")
        or (
            "Referencia clasificada como ayuda diagnóstica."
            if classification_status != "unclassified"
            else "El tipo original de la ayuda diagnóstica no se pudo mapear de forma confiable."
        ),
        "codigo": codigo or _normalize_whitespace(raw.get("codigo")),
        "nombre": nombre,
        "concepto": concept,
        "interpretacion": concept if interpreted else "",
        "interpretado": interpreted,
        "estado_interpretacion": state,
        "fuentes": _clean_unique_text_list(
            [
                raw.get("fuente"),
                *[
                    evidence.get("fuente") or evidence.get("document_type")
                    for evidence in raw.get("evidencias") or []
                    if isinstance(evidence, dict)
                ],
            ]
        ),
    }
    projection["auditoria"] = project_auditoria_traceability(
        projection,
        section="ayudas_diagnosticas",
    )
    publication = evaluate_publication_eligibility(raw, artifact_type="ayuda_diagnostica")
    projection["publicacion"] = publication.model_dump(
        mode="json",
        exclude={"evaluated_at"},
    )
    return projection


def _aid_merge_key(ayuda: dict[str, Any]) -> str:
    canonical_id = _normalize_whitespace(ayuda.get("canonical_id"))
    if canonical_id and not ayuda.get("trayectoria_requiere_revision"):
        return f"canonical:{canonical_id}"
    return _normalize_whitespace(ayuda.get("key"))


def _normalize_ayuda_state(value: Any) -> str:
    state = _normalize_ascii(value).lower()
    return state if state in _AID_STATE_PRIORITY else "pendiente_revision"


def _aid_requires_review(ayuda: dict[str, Any]) -> bool:
    alerts = {
        _normalize_ascii(value).lower()
        for value in ayuda.get("alertas") or []
        if _normalize_whitespace(value)
    }
    return bool(
        ayuda.get("incidencias")
        or ayuda.get("trayectoria_requiere_revision")
        or alerts.intersection(_AID_REVIEW_ALERTS)
    )


def _refresh_ayuda_trajectory(ayuda: dict[str, Any]) -> None:
    # Document-level evidence is part of the audit contract, but only canonical
    # occurrences can establish longitudinal identity or a merge.
    trajectory_evidence = [
        item for item in ayuda.get("evidencias") or [] if isinstance(item, dict) and item.get("occurrence_id")
    ]
    trajectories = build_clinical_trajectories(trajectory_evidence)
    ayuda["trayectorias"] = trajectories
    ayuda.pop("trayectoria", None)
    ayuda.pop("canonical_id", None)
    ayuda.pop("canonical_term", None)
    ayuda.pop("trayectoria_requiere_revision", None)
    ayuda.pop("confiabilidad_trayectoria", None)
    ayuda.pop("etapas_trayectoria", None)
    ayuda.pop("fuentes_trayectoria", None)

    strong = [item for item in trajectories if item.get("canonical_id") and not item.get("requires_review")]
    review_reasons = _clean_unique_text_list(
        [
            reason
            for trajectory in trajectories
            if trajectory.get("requires_review")
            for reason in trajectory.get("review_reasons") or []
        ]
    )
    if len(strong) == 1 and not review_reasons:
        trajectory = strong[0]
        ayuda["trayectoria"] = trajectory
        ayuda["canonical_id"] = trajectory["canonical_id"]
        ayuda["canonical_term"] = trajectory.get("canonical_term") or ""
        ayuda["key"] = f"ayuda:{ayuda.get('tipo') or 'imagen'}:canonical:{trajectory['canonical_id']}"
        ayuda["confiabilidad_trayectoria"] = {
            "score_promedio": trajectory.get("confidence_score"),
            "evidencias": trajectory.get("evidence_count", 0),
            "fuentes": len(trajectory.get("source_ids") or []),
            "coherente": True,
        }
        ayuda["etapas_trayectoria"] = trajectory.get("clinical_stages") or []
        ayuda["fuentes_trayectoria"] = trajectory.get("source_ids") or []
        canonical_occurrence = next(
            (
                occurrence
                for occurrence in trajectory.get("occurrences") or []
                if isinstance(occurrence, dict)
            ),
            {},
        )
        ayuda["match_method"] = canonical_occurrence.get("match_method") or ""
        ayuda["match_score"] = canonical_occurrence.get("match_score")
        ayuda["match_margin"] = canonical_occurrence.get("match_margin")
        automatic = canonical_occurrence.get("automatic")
        automatic = automatic if isinstance(automatic, dict) else {}
        ayuda["catalog"] = canonical_occurrence.get("catalog") or automatic.get("catalog") or ""
        ayuda["catalog_version"] = (
            canonical_occurrence.get("catalog_version") or automatic.get("catalog_version") or ""
        )
    elif trajectories:
        ayuda["trayectoria_requiere_revision"] = True
        ayuda["confiabilidad_trayectoria"] = {
            "score_promedio": next(
                (
                    item.get("confidence_score")
                    for item in trajectories
                    if item.get("confidence_score") is not None
                ),
                None,
            ),
            "evidencias": sum(int(item.get("evidence_count") or 0) for item in trajectories),
            "fuentes": len(
                {source for item in trajectories for source in item.get("source_ids") or [] if source}
            ),
            "coherente": False,
        }

    trajectory_alert_prefix = "trayectoria:"
    alerts = [
        alert
        for alert in _clean_unique_text_list(ayuda.get("alertas"))
        if not alert.startswith(trajectory_alert_prefix)
    ]
    incidents = [
        item
        for item in ayuda.get("incidencias") or []
        if not isinstance(item, dict) or item.get("_trajectory_review") is not True
    ]
    for trajectory in trajectories:
        if not trajectory.get("requires_review"):
            continue
        for reason in trajectory.get("review_reasons") or []:
            alerts.append(f"{trajectory_alert_prefix}{reason}")
            if reason in {
                "asociacion_canonica_no_confirmada",
                "asociacion_canonica_pendiente_revision",
            }:
                # The canonical occurrence already carries this review item;
                # avoid duplicating it in the aid-level incident list.
                continue
            incidents.append(
                {
                    "_trajectory_review": True,
                    "trajectory_id": trajectory.get("trajectory_id", ""),
                    "reason": reason,
                }
            )
    ayuda["alertas"] = _clean_unique_text_list(alerts)
    ayuda["incidencias"] = incidents
    if review_reasons:
        ayuda["estado_interpretacion"] = "pendiente_revision"
        ayuda["interpretado"] = False


def _merge_ayuda(existing: dict[str, Any], incoming: dict[str, Any]) -> None:
    for field in ("ordenado", "interpretado", "facturado", "glosado"):
        existing[field] = bool(existing.get(field)) or bool(incoming.get(field))
    if incoming.get("concepto") and (
        not existing.get("concepto") or existing.get("concepto") == "No interpretado"
    ):
        existing["concepto"] = incoming.get("concepto", "")
    existing_state = _normalize_ayuda_state(existing.get("estado_interpretacion"))
    incoming_state = _normalize_ayuda_state(incoming.get("estado_interpretacion"))
    if "excluido" in {existing_state, incoming_state}:
        merged_state = "excluido"
    elif _aid_requires_review(existing) or _aid_requires_review(incoming):
        merged_state = "pendiente_revision"
    elif "interpretado" in {existing_state, incoming_state}:
        merged_state = "interpretado"
    else:
        merged_state = "no_interpretado"
    existing["estado_interpretacion"] = merged_state
    if existing["estado_interpretacion"] != "interpretado":
        existing["interpretado"] = False
    existing["interpretacion_estructurada"] = bool(
        existing.get("interpretacion_estructurada") or incoming.get("interpretacion_estructurada")
    )
    if incoming.get("interpretacion_asistida"):
        existing["interpretacion_asistida"] = incoming["interpretacion_asistida"]
    if incoming.get("classification_status") == "unclassified":
        existing["classification_status"] = "unclassified"
        existing["classification_reason"] = (
            incoming.get("classification_reason")
            or existing.get("classification_reason")
            or "El tipo original de la ayuda diagnóstica no se pudo mapear de forma confiable."
        )
    elif not existing.get("classification_status"):
        existing["classification_status"] = incoming.get("classification_status") or "accepted"
    if not existing.get("tipo_original") and incoming.get("tipo_original"):
        existing["tipo_original"] = incoming["tipo_original"]
    if existing["estado_interpretacion"] != "interpretado":
        existing["concepto"] = "No interpretado"
    for field in ("fuente", "documento_id", "documento_nombre", "razon_inclusion"):
        if not existing.get(field) and incoming.get(field):
            existing[field] = incoming[field]
    existing_alerts = _clean_unique_text_list(existing.get("alertas"))
    for alerta in _clean_unique_text_list(incoming.get("alertas")):
        if alerta not in existing_alerts:
            existing_alerts.append(alerta)
    existing["alertas"] = existing_alerts
    existing_terms = _clean_unique_text_list(
        [
            existing.get("termino_original"),
            *(existing.get("terminos_originales") or []),
            incoming.get("termino_original"),
            *(incoming.get("terminos_originales") or []),
        ]
    )
    existing["terminos_originales"] = existing_terms
    if not existing.get("termino_original") and existing_terms:
        existing["termino_original"] = existing_terms[0]
    evidence_by_key: dict[str, dict[str, Any]] = {}
    for item in [*existing.get("evidencias", []), *incoming.get("evidencias", [])]:
        if not isinstance(item, dict):
            continue
        evidence_key = str(
            item.get("occurrence_id")
            or item.get("evidencia_id")
            or ":".join(
                str(item.get(key) or "")
                for key in ("document_uid", "documento_id", "fuente", "termino_original")
            )
        )
        if evidence_key.strip(":"):
            evidence_by_key[evidence_key] = item
    existing["evidencias"] = list(evidence_by_key.values())
    existing["incidencias"] = list(
        {
            str(item.get("occurrence_id") or item.get("trajectory_id") or item.get("reason")): item
            for item in [*existing.get("incidencias", []), *incoming.get("incidencias", [])]
            if isinstance(item, dict)
            and (item.get("occurrence_id") or item.get("trajectory_id") or item.get("reason"))
        }.values()
    )
    existing_normalization = dict(existing.get("normalizacion") or {})
    for key, value in dict(incoming.get("normalizacion") or {}).items():
        if value and not existing_normalization.get(key):
            existing_normalization[key] = value
    existing["normalizacion"] = existing_normalization
    if incoming.get("effective_date") and (
        not existing.get("effective_date")
        or sort_chronological_items(
            [
                {
                    "key": existing.get("key"),
                    "effective_date": existing.get("effective_date"),
                    "date_source": existing.get("date_source"),
                    "date_precision": existing.get("date_precision"),
                },
                {
                    "key": incoming.get("key"),
                    "effective_date": incoming.get("effective_date"),
                    "date_source": incoming.get("date_source"),
                    "date_precision": incoming.get("date_precision"),
                },
            ]
        )[0].get("effective_date")
        == incoming.get("effective_date")
    ):
        for field in ("effective_date", "date_source", "date_precision", "date_warning"):
            existing[field] = incoming.get(field)
    _refresh_ayuda_trajectory(existing)
    existing["exportable"] = evaluar_ayuda_diagnostica_exportable(existing)


def _append_ayuda(target: list[dict[str, Any]], ayuda: dict[str, Any] | None) -> None:
    if ayuda is None:
        return
    for existing in target:
        if _aid_merge_key(existing) == _aid_merge_key(ayuda):
            _merge_ayuda(existing, ayuda)
            return
    target.append(ayuda)


def _aid_evidence_sources(ayuda: dict[str, Any]) -> set[str]:
    sources = {_normalize_ascii(ayuda.get("fuente")).lower()}
    sources.update(
        _normalize_ascii(evidence.get("fuente") or evidence.get("document_type")).lower()
        for evidence in ayuda.get("evidencias") or []
        if isinstance(evidence, dict)
    )
    return {source for source in sources if source}


def _confirmed_aid_canonical_id(ayuda: dict[str, Any]) -> str:
    canonical_id = _normalize_whitespace(ayuda.get("canonical_id"))
    return canonical_id if canonical_id and not ayuda.get("trayectoria_requiere_revision") else ""


def _invoice_has_clinical_support(
    factura: dict[str, Any],
    supported_names: set[tuple[str, str]],
    supported_canonical_ids: set[str],
) -> bool:
    sources = _aid_evidence_sources(factura)
    if sources.intersection(_CLINICAL_AID_SOURCES):
        return True
    signature = (_normalize_ayuda_tipo(factura.get("tipo")), _normalize_ayuda_key_text(factura.get("nombre")))
    if signature in supported_names:
        return True
    canonical_id = _confirmed_aid_canonical_id(factura)
    return bool(canonical_id and canonical_id in supported_canonical_ids)


def _apply_invoice_support_reviews(ayudas: list[dict[str, Any]]) -> None:
    """Keep invoice-only aids traceable without presenting them as clinical support."""
    supported_names = {
        (_normalize_ayuda_tipo(item.get("tipo")), _normalize_ayuda_key_text(item.get("nombre")))
        for item in ayudas
        if _normalize_ascii(item.get("fuente")).lower() != "factura"
    }
    supported_canonical_ids = {
        canonical_id
        for item in ayudas
        if _normalize_ascii(item.get("fuente")).lower() != "factura"
        and (canonical_id := _confirmed_aid_canonical_id(item))
    }
    for item in ayudas:
        if _normalize_ascii(item.get("fuente")).lower() != "factura":
            continue
        if _invoice_has_clinical_support(item, supported_names, supported_canonical_ids):
            continue
        alerts = _clean_unique_text_list(item.get("alertas"))
        for alert in (_INVOICE_SUPPORT_ALERT, _INVOICE_CONTRADICTION_ALERT):
            if alert not in alerts:
                alerts.append(alert)
        item["alertas"] = alerts
        incidents = [incident for incident in item.get("incidencias") or [] if isinstance(incident, dict)]
        if not any(incident.get("codigo") == _INVOICE_SUPPORT_INCIDENT for incident in incidents):
            incidents.append(
                {
                    "incidencia_id": f"factura-soporte:{item.get('key') or item.get('documento_id') or item.get('nombre')}",
                    "tipo": "evidencia_contradictoria",
                    "codigo": _INVOICE_SUPPORT_INCIDENT,
                    "razon": "La factura registra la ayuda, pero no se identificó soporte clínico asociado.",
                    "documento_id": item.get("documento_id") or "",
                    "document_uid": item.get("document_uid") or "",
                    "fuente": "factura",
                    "revisable": True,
                }
            )
        item["incidencias"] = incidents
        item["estado_interpretacion"] = "pendiente_revision"
        item["interpretado"] = False
        item["concepto"] = "No interpretado"


def _factura_ayudas_diagnosticas(factura_doc: dict[str, Any] | None) -> list[dict[str, Any]]:
    if not isinstance(factura_doc, dict):
        return []
    servicios = _servicios_factura(factura_doc)
    ayudas: list[dict[str, Any]] = []
    for item in servicios.get("imagenologia") or []:
        nombre = (
            _normalize_whitespace(item.get("estudio") or item.get("prueba") or item.get("concepto"))
            if isinstance(item, dict)
            else _imagenologia_factura_display(item)
        )
        descripcion = _normalize_whitespace(item.get("descripcion")) if isinstance(item, dict) else ""
        _append_ayuda(
            ayudas,
            _build_ayuda_diagnostica(
                tipo=_tipo_ayuda_desde_texto(f"{nombre} {descripcion}"),
                nombre=_format_factura_ayuda_nombre(nombre, descripcion),
                concepto="",
                fuente="factura",
                documento=factura_doc,
                clinical_date=item.get("fecha_servicio") if isinstance(item, dict) else None,
                ordenado=True,
                facturado=True,
                estado_interpretacion="no_interpretado",
                razon_inclusion="Detectada en imagenología facturada.",
            ),
        )
    for item in servicios.get("examenes_laboratorio") or []:
        nombre = (
            _normalize_whitespace(item.get("estudio") or item.get("prueba") or item.get("concepto"))
            if isinstance(item, dict)
            else _imagenologia_factura_display(item)
        )
        descripcion = _normalize_whitespace(item.get("descripcion")) if isinstance(item, dict) else ""
        _append_ayuda(
            ayudas,
            _build_ayuda_diagnostica(
                tipo="laboratorio",
                nombre=_format_factura_ayuda_nombre(nombre, descripcion),
                concepto="",
                fuente="factura",
                documento=factura_doc,
                clinical_date=item.get("fecha_servicio") if isinstance(item, dict) else None,
                ordenado=True,
                facturado=True,
                estado_interpretacion="no_interpretado",
                razon_inclusion="Detectada en exámenes de laboratorio facturados.",
            ),
        )
    return ayudas


def _radiologia_ayuda_diagnostica(doc: dict[str, Any]) -> dict[str, Any] | None:
    structured = get_structured_document_model(doc)
    html = str(doc.get("analisis_html") or doc.get("analisis") or render_document_analysis_html(doc) or "")
    nombre = ""
    if structured is not None and hasattr(structured, "tipo_estudio"):
        nombre = _normalize_whitespace(getattr(structured, "tipo_estudio", "") or "")
    if not nombre:
        nombre = _normalize_whitespace(doc.get("nombre_archivo") or doc.get("filename"))
    if not nombre:
        nombre = _extract_html_field(html, ("Tipo de estudio", "Estudio", "Examen")) or "Reporte radiológico"
    concepto = _radiologia_interpretacion(doc)
    structured_conclusion = bool(
        structured is not None
        and _normalize_whitespace(getattr(structured, "conclusion", "") or "")
        and _normalize_ascii(getattr(structured, "conclusion", "") or "").lower()
        not in {"...", "no especificado", "no disponible"}
    )
    interpretado = structured_conclusion
    return _build_ayuda_diagnostica(
        tipo="imagen",
        nombre=nombre,
        concepto=concepto,
        fuente="radiologia",
        documento=doc,
        ordenado=True,
        interpretado=interpretado,
        estado_interpretacion="interpretado" if interpretado else "no_interpretado",
        interpretacion_estructurada=structured_conclusion,
        razon_inclusion="Documento radiológico asociado al caso.",
    )


def _laboratorio_ayuda_diagnostica(doc: dict[str, Any]) -> dict[str, Any] | None:
    structured = get_structured_document_model(doc)
    html = str(doc.get("analisis_html") or doc.get("analisis") or render_document_analysis_html(doc) or "")
    nombre = ""
    concepto = ""
    structured_interpretation = False
    if structured is not None and hasattr(structured, "tipo_examen"):
        nombre = _normalize_whitespace(getattr(structured, "tipo_examen", "") or "")
        concepto = _normalize_whitespace(
            getattr(structured, "interpretacion", "") or getattr(structured, "observaciones", "") or ""
        )
        structured_interpretation = bool(concepto)
    if not nombre:
        nombre = (
            _extract_html_field(html, ("Tipo de examen", "Examen", "Prueba", "Nombre del examen"))
            or _normalize_whitespace(doc.get("nombre_archivo") or doc.get("filename"))
            or "Reporte de laboratorio"
        )
    if not concepto:
        concepto = _extract_html_field(
            html,
            (
                "Interpretación",
                "Interpretacion",
                "Conclusión",
                "Conclusion",
                "Observaciones",
                "Resumen",
            ),
        )
    interpretado = structured_interpretation
    return _build_ayuda_diagnostica(
        tipo="laboratorio",
        nombre=nombre,
        concepto=concepto or "No interpretado",
        fuente="laboratorio",
        documento=doc,
        ordenado=True,
        interpretado=interpretado,
        estado_interpretacion="interpretado" if interpretado else "no_interpretado",
        interpretacion_estructurada=structured_interpretation,
        razon_inclusion="Documento de laboratorio asociado al caso.",
    )


def _tipo_ayuda_desde_texto(texto: str) -> str:
    normalized = _normalize_ascii(texto).lower()
    if any(term in normalized for term in ("endoscopia", "electrocardiograma", "ecg", "ekg")):
        return "procedimiento_diagnostico"
    if any(
        term in normalized
        for term in (
            "laboratorio",
            "hemograma",
            "creatinina",
            "uroanalisis",
            "paraclinico",
            "prequirurgico",
        )
    ):
        return "laboratorio"
    return "imagen"


def _extract_textual_aid_names(sentence: str) -> list[str]:
    candidates: list[str] = []
    for match in _AID_TEXT_PATTERN.finditer(sentence):
        if _EXCLUDED_BLOCK_PREFIX.match(sentence[: match.start()] if match.start() else sentence):
            continue
        term = _normalize_whitespace(match.group(0)).strip(" -.;:")
        if not term:
            continue
        tail_parts: list[str] = []
        remainder = sentence[match.end() :]
        for token in re.findall(r"[\wÁÉÍÓÚÜÑáéíóúüñ0-9/-]+", remainder):
            normalized_token = _normalize_ascii(token).casefold()
            if _MEDICATION_OR_SUPPLY_MARKERS.fullmatch(token) or _ADMINISTRATIVE_MARKERS.fullmatch(token):
                break
            if normalized_token in _NARRATIVE_BOUNDARY_WORDS or normalized_token in {"y", "e", "o", "pero"}:
                break
            if tail_parts and normalized_token in _NARRATIVE_STOP_WORDS:
                break
            tail_parts.append(token)
            if len(tail_parts) >= 6:
                break
        reference = _normalize_whitespace(" ".join([term, *tail_parts]))
        code_match = re.search(r"\b\d{5,6}\s*$", sentence[: match.start()])
        if code_match:
            reference = f"{code_match.group(0).strip()} - {reference}"
        candidates.append(reference[:140])
    return _clean_unique_text_list(candidates)


def _sentencias_con_ayudas(texto: Any) -> list[str]:
    raw = str(texto or "")
    if not _normalize_whitespace(raw):
        return []
    sentences = re.split(r"[\n.;]+", raw)
    matches: list[str] = []
    for sentence in sentences:
        sentence = _normalize_whitespace(sentence).strip(" -.;:")
        if not sentence or len(sentence) < 4:
            continue
        if _AID_TEXT_PATTERN.search(sentence):
            matches.extend(_extract_textual_aid_names(sentence))
    return _clean_unique_text_list(matches)


def _narrative_rejected_candidate(
    *,
    document: dict[str, Any],
    source: str,
    segment: str,
    artifact_type: str,
    reason: str,
) -> dict[str, Any]:
    document_id = _normalize_whitespace(document.get("_id") or document.get("id"))
    document_uid = _normalize_whitespace(document.get("document_uid"))
    document_name = _normalize_whitespace(
        document.get("nombre_archivo") or document.get("filename") or document.get("archivo")
    )
    scope = document_uid or document_id or document_name or source or "desconocido"
    candidate_id = "rejected-candidate:" + ":".join(
        value
        for value in (
            _normalize_ayuda_key_text(source),
            _normalize_ayuda_key_text(scope),
            _normalize_ayuda_key_text(segment),
            artifact_type,
        )
        if value
    )
    evidence = _build_document_evidence(
        documento=document,
        fuente=source,
        original_term=segment,
        interpretacion_estructurada=False,
        candidate_id=candidate_id,
        artifact_type=artifact_type,
    )
    return {
        "artifact_type": artifact_type,
        "candidate_id": candidate_id,
        "classification_status": "rejected",
        "classification_reason": reason,
        "status": "excluido",
        "estado": "excluido",
        "nombre": _normalize_whitespace(segment)[:180],
        "texto_original": _normalize_whitespace(segment)[:280],
        "fuente": source,
        "document_uid": document_uid,
        "documento_id": document_id,
        "razon_exclusion": reason,
        "razon_revision": reason,
        "evidencias": [evidence] if evidence else [],
        "incidencias": [{"codigo": "clasificacion_excluida", "razon": reason, "revisable": True}],
    }


def classify_diagnostic_aid_candidates(
    document: Any,
    fuente: str,
) -> dict[str, list[dict[str, Any]]]:
    """Classify one document before any canonical matching or aid consolidation.

    Structured references are authoritative. Narrative text can only create a
    bounded mention candidate; medication, supply, and administrative segments
    are routed to the audit queue and never become ``AyudaDiagnostica`` values.
    """
    if not isinstance(document, dict):
        return {"accepted": [], "rejected": []}
    structured = get_structured_document_model(document)
    accepted: list[dict[str, Any]] = []
    rejected: list[dict[str, Any]] = []
    if structured is not None and hasattr(structured, "referencias_diagnosticas"):
        for item in getattr(structured, "referencias_diagnosticas", []) or []:
            name = _normalize_whitespace(getattr(item, "nombre", ""))
            if not name:
                continue
            raw_type = _normalize_ascii(getattr(item, "tipo", "") or "").casefold()
            if raw_type in _REJECTED_ARTIFACT_TYPES:
                rejected.append(
                    _narrative_rejected_candidate(
                        document=document,
                        source=fuente,
                        segment=name,
                        artifact_type=raw_type,
                        reason="La referencia estructurada pertenece a medicamentos o insumos y no a ayudas diagnósticas.",
                    )
                )
                continue
            accepted.append(
                {
                    "tipo": getattr(item, "tipo", "imagen"),
                    "nombre": name,
                    "concepto": _normalize_whitespace(getattr(item, "concepto", "")),
                    "structured": True,
                }
            )
        if accepted:
            return {"accepted": accepted, "rejected": rejected}

    text = "\n".join(
        filter(
            None,
            [
                str(document.get("descripcion") or ""),
                str(document.get("texto_extraido") or ""),
                strip_html_tags(
                    str(document.get("analisis_html") or render_document_analysis_html(document) or "")
                ),
            ],
        )
    )
    for raw_segment in re.split(r"[\n.;]+", text):
        segment = _normalize_whitespace(raw_segment).strip(" -.;:")
        if not segment:
            continue
        references = _extract_textual_aid_names(segment)
        if references:
            accepted.extend({"tipo": _tipo_ayuda_desde_texto(item), "nombre": item} for item in references)
            continue
        if _MEDICATION_OR_SUPPLY_MARKERS.search(segment):
            marker = _MEDICATION_OR_SUPPLY_MARKERS.search(segment)
            marker_text = _normalize_ascii(marker.group(0) if marker else "").casefold()
            artifact_type = (
                "insumo"
                if marker_text in {"insumo", "insumos", "suministro", "suministros"}
                else "medicamento"
            )
            rejected.append(
                _narrative_rejected_candidate(
                    document=document,
                    source=fuente,
                    segment=segment,
                    artifact_type=artifact_type,
                    reason="El fragmento pertenece a medicamentos o insumos y no a ayudas diagnósticas.",
                )
            )
        elif _ADMINISTRATIVE_MARKERS.search(segment):
            rejected.append(
                _narrative_rejected_candidate(
                    document=document,
                    source=fuente,
                    segment=segment,
                    artifact_type="administrativo",
                    reason="El fragmento pertenece a información administrativa y no a una ayuda diagnóstica.",
                )
            )
    return {"accepted": accepted, "rejected": rejected}


def _referencias_ayudas_desde_documento(doc: Any, fuente: str) -> list[dict[str, Any]]:
    if not isinstance(doc, dict):
        return []
    classification = classify_diagnostic_aid_candidates(doc, fuente)
    structured_candidates = classification["accepted"]
    if structured_candidates and all(item.get("structured") for item in structured_candidates):
        ayudas: list[dict[str, Any]] = []
        for item in structured_candidates:
            _append_ayuda(
                ayudas,
                _build_ayuda_diagnostica(
                    tipo=item.get("tipo", ""),
                    nombre=item.get("nombre", ""),
                    concepto=item.get("concepto", ""),
                    fuente=fuente,
                    documento=doc,
                    ordenado=True,
                    interpretado=False,
                    estado_interpretacion="pendiente_revision",
                    interpretacion_estructurada=False,
                    razon_inclusion="Referencia diagnóstica estructurada asociada al documento.",
                ),
            )
        if ayudas:
            return ayudas
    text = "\n".join(
        filter(
            None,
            [
                str(doc.get("descripcion") or ""),
                strip_html_tags(str(doc.get("analisis_html") or render_document_analysis_html(doc) or "")),
            ],
        )
    )
    ayudas: list[dict[str, Any]] = []
    if not _normalize_whitespace(text):
        _append_ayuda(
            ayudas,
            _build_ayuda_diagnostica(
                tipo="procedimiento_diagnostico",
                nombre=_normalize_whitespace(doc.get("nombre_archivo")) or "Documento sin texto extraíble",
                concepto="No interpretado",
                fuente=fuente,
                documento={**doc, "texto_no_extraible": True},
                estado_interpretacion="pendiente_revision",
                razon_inclusion="El documento no contiene texto extraíble por ReadPdf.read.",
                alertas=["texto_no_extraible"],
            ),
        )
        return ayudas
    for candidate in classification["accepted"]:
        _append_ayuda(
            ayudas,
            _build_ayuda_diagnostica(
                tipo=candidate.get("tipo") or "imagen",
                nombre=candidate.get("nombre"),
                concepto="Referencia textual sin interpretación estructurada.",
                fuente=fuente,
                documento=doc,
                ordenado=True,
                estado_interpretacion="pendiente_revision",
                razon_inclusion="Referencia textual conservadora en documento clínico.",
                alertas=["referencia_textual_pendiente_revision"],
            ),
        )
    return ayudas


def _referencias_ayudas_desde_qx(context: dict[str, Any]) -> list[dict[str, Any]]:
    text = "\n".join(
        [
            str(context.get("hallazgos_quirurgicos") or ""),
            str(context.get("descripcion_procedimiento") or ""),
        ]
    )
    ayudas: list[dict[str, Any]] = []
    for sentence in _sentencias_con_ayudas(text):
        _append_ayuda(
            ayudas,
            _build_ayuda_diagnostica(
                tipo=_tipo_ayuda_desde_texto(sentence),
                nombre=sentence,
                concepto="Referencia desde hallazgos/descripción QX pendiente de revisión.",
                fuente="agente_hallazgos_qx",
                estado_interpretacion="pendiente_revision",
                razon_inclusion="Referencia diagnóstica mencionada en contexto quirúrgico.",
                alertas=["referencia_qx_pendiente_revision"],
            ),
        )
    return ayudas


def _coerce_manual_ayuda_item(item: Any) -> dict[str, Any] | None:
    if isinstance(item, dict):
        raw_artifact_type = _normalize_ascii(item.get("artifact_type") or item.get("tipo")).lower()
        if raw_artifact_type in _REJECTED_ARTIFACT_TYPES:
            return None
        nombre = item.get("nombre") or item.get("texto") or item.get("descripcion")
        ayuda = _build_ayuda_diagnostica(
            tipo=item.get("tipo") or "imagen",
            nombre=nombre,
            concepto=item.get("concepto") or item.get("interpretacion") or "",
            fuente=item.get("fuente") or "manual_pdf",
            documento=None,
            ordenado=_normalize_bool(item.get("ordenado")),
            interpretado=_normalize_bool(item.get("interpretado")),
            facturado=_normalize_bool(item.get("facturado")),
            glosado=_normalize_bool(item.get("glosado")),
            estado_interpretacion=item.get("estado_interpretacion") or "pendiente_revision",
            razon_inclusion=item.get("razon_inclusion") or "Agregada manualmente para exportación.",
            alertas=item.get("alertas"),
        )
        if ayuda and _normalize_whitespace(item.get("key")):
            ayuda["key"] = _normalize_whitespace(item.get("key"))
        return ayuda

    text = _normalize_whitespace(item)
    if not text:
        return None
    parts = [part.strip() for part in text.split("|")]
    if len(parts) >= 2:
        tipo = parts[0]
        if _normalize_ascii(tipo).lower() in _REJECTED_ARTIFACT_TYPES:
            return None
        nombre = parts[1]
        concepto = " | ".join(parts[2:]).strip()
    else:
        tipo = _tipo_ayuda_desde_texto(text)
        nombre = text
        concepto = ""
    return _build_ayuda_diagnostica(
        tipo=tipo,
        nombre=nombre,
        concepto=concepto or "Agregada manualmente.",
        fuente="manual_pdf",
        estado_interpretacion="pendiente_revision",
        razon_inclusion="Agregada manualmente para exportación.",
    )


def coerce_ayudas_diagnosticas(values: Any) -> list[dict[str, Any]]:
    result: list[dict[str, Any]] = []
    for item in values or []:
        if not isinstance(item, dict):
            ayuda = _coerce_manual_ayuda_item(item)
        else:
            raw_artifact_type = _normalize_ascii(item.get("artifact_type") or item.get("tipo")).lower()
            if raw_artifact_type in _REJECTED_ARTIFACT_TYPES:
                continue
            ayuda = _build_ayuda_diagnostica(
                tipo=item.get("tipo") or "imagen",
                nombre=item.get("nombre") or item.get("texto") or item.get("descripcion"),
                concepto=item.get("concepto") or item.get("interpretacion") or "",
                fuente=item.get("fuente") or "desconocida",
                documento={
                    "_id": item.get("documento_id"),
                    "nombre_archivo": item.get("documento_nombre"),
                },
                ordenado=_normalize_bool(item.get("ordenado")),
                interpretado=_normalize_bool(item.get("interpretado")),
                facturado=_normalize_bool(item.get("facturado")),
                glosado=_normalize_bool(item.get("glosado")),
                estado_interpretacion=item.get("estado_interpretacion") or "",
                razon_inclusion=item.get("razon_inclusion") or "",
                alertas=item.get("alertas"),
            )
            if ayuda and _normalize_whitespace(item.get("key")):
                ayuda["key"] = _normalize_whitespace(item.get("key"))
            if ayuda:
                for field in (
                    "item_id",
                    "artifact_id",
                    "status",
                    "decisions",
                    "reason_review",
                    "candidate_id",
                    "classification_status",
                    "classification_reason",
                    "case_key",
                    "document_uid",
                    "evidencias",
                    "incidencias",
                    "normalizacion",
                    "termino_original",
                    "terminos_originales",
                    "canonical_id",
                    "canonical_term",
                    "match_method",
                    "match_score",
                    "match_margin",
                    "catalog",
                    "catalog_version",
                    "trayectoria",
                    "trayectorias",
                    "trayectoria_requiere_revision",
                    "confiabilidad_trayectoria",
                    "etapas_trayectoria",
                    "fuentes_trayectoria",
                    "effective_date",
                    "date_source",
                    "date_precision",
                    "date_warning",
                    "interpretacion_asistida",
                ):
                    if item.get(field):
                        ayuda[field] = item[field]
            if ayuda and isinstance(item.get("costo_facturado"), dict):
                ayuda["costo_facturado"] = dict(item["costo_facturado"])
            if ayuda:
                if _aid_requires_review(ayuda):
                    ayuda["estado_interpretacion"] = "pendiente_revision"
                    ayuda["interpretado"] = False
                    ayuda["concepto"] = "No interpretado"
                ayuda["exportable"] = evaluar_ayuda_diagnostica_exportable(ayuda)
                if item.get("exportable") is False:
                    ayuda["exportable"] = False
        _append_ayuda(result, ayuda)
    return result


def format_ayuda_diagnostica_presentacion(item: Any) -> str:
    if not isinstance(item, dict):
        return _normalize_whitespace(item)
    nombre = format_ayuda_diagnostica_nombre_exportable(item)
    concepto = _normalize_whitespace(item.get("concepto"))
    estado = _normalize_ayuda_state(item.get("estado_interpretacion"))
    fuente = _normalize_ascii(item.get("fuente")).lower()
    if estado != "interpretado" and fuente not in {"factura", "radiologia"}:
        return nombre
    if (
        concepto
        and not _is_uninterpreted_placeholder(concepto)
        and concepto.lower() != nombre.lower()
        and concepto.lower() not in nombre.lower()
    ):
        return f"{nombre} - {concepto}"
    return nombre


def build_diagnostic_aid_quality_report(
    context: dict[str, Any] | None,
    aids: list[dict[str, Any]] | None = None,
) -> dict[str, Any]:
    """Return repeatable, text-free quality metrics for the aid pipeline."""
    source = context if isinstance(context, dict) else {}
    values = aids if isinstance(aids, list) else build_ayudas_diagnosticas(dict(source))
    rows: dict[tuple[str, str, str, tuple[str, ...]], dict[str, Any]] = {}
    for aid in values:
        if not isinstance(aid, dict):
            continue
        evidences = [item for item in aid.get("evidencias") or [] if isinstance(item, dict)]
        methods = tuple(
            sorted(
                {
                    _normalize_whitespace(item.get("match_method")) or "structured_reference"
                    for item in evidences
                }
                or {"structured_reference"}
            )
        )
        source_name = _normalize_whitespace(aid.get("fuente")) or "desconocida"
        aid_type = _normalize_whitespace(aid.get("tipo")) or "desconocido"
        state = _normalize_ayuda_state(aid.get("estado_interpretacion"))
        key = (source_name, aid_type, state, methods)
        row = rows.setdefault(
            key,
            {
                "fuente": source_name,
                "tipo": aid_type,
                "estado": state,
                "metodos": list(methods),
                "artefactos": 0,
                "destinos": {
                    "deteccion": 0,
                    "consolidacion": 0,
                    "interpretacion": 0,
                    "auditoria": 0,
                    "publicacion": 0,
                },
                "identificadores": [],
                "referencias": [],
            },
        )
        row["artefactos"] += 1
        destination = row["destinos"]
        destination["deteccion"] += 1
        destination["consolidacion"] += 1
        if state == "interpretado":
            destination["interpretacion"] += 1
        eligibility = evaluate_publication_eligibility(aid, artifact_type=_DIAGNOSTIC_ARTIFACT_TYPE)
        if eligibility.requires_audit:
            destination["auditoria"] += 1
        if eligibility.publishable:
            destination["publicacion"] += 1
        identifier = _normalize_whitespace(
            aid.get("canonical_id") or aid.get("artifact_id") or aid.get("key") or aid.get("candidate_id")
        )
        if identifier and identifier not in row["identificadores"]:
            row["identificadores"].append(identifier)
        for evidence in evidences:
            reference = {
                key: evidence.get(key)
                for key in (
                    "occurrence_id",
                    "document_uid",
                    "document_id",
                    "source_hash",
                    "page",
                    "section",
                    "match_method",
                )
                if evidence.get(key) not in (None, "")
            }
            if reference and reference not in row["referencias"]:
                row["referencias"].append(reference)

    for row in rows.values():
        row["identificadores"] = sorted(row["identificadores"])
        row["referencias"] = sorted(
            row["referencias"],
            key=lambda reference: tuple(str(reference.get(key) or "") for key in ("document_uid", "page", "occurrence_id")),
        )
    report_rows = sorted(rows.values(), key=lambda item: (item["fuente"], item["tipo"], item["estado"]))
    totals = {
        destination: sum(row["destinos"][destination] for row in report_rows)
        for destination in ("deteccion", "consolidacion", "interpretacion", "auditoria", "publicacion")
    }
    catalog_versions = sorted(
        {
            _normalize_whitespace(item.get("catalog_version"))
            for aid in values
            if isinstance(aid, dict)
            for item in aid.get("evidencias") or []
            if isinstance(item, dict) and _normalize_whitespace(item.get("catalog_version"))
        }
    )
    return {
        "schema_version": "v1",
        "consolidation_version": DIAGNOSTIC_AID_CONSOLIDATION_VERSION,
        "rules_version": DIAGNOSTIC_AID_RULES_VERSION,
        "catalog_versions": catalog_versions,
        "case_key": _normalize_whitespace(source.get("case_key")),
        "rows": report_rows,
        "totals": totals,
    }


def consolidar_ayudas_diagnosticas(context: dict[str, Any]) -> list[dict[str, Any]]:
    context = context or {}
    ayudas: list[dict[str, Any]] = []
    for ayuda in _factura_ayudas_diagnosticas(context.get("factura")):
        _append_ayuda(ayudas, ayuda)
    for doc in context.get("radiologia") or []:
        _append_ayuda(ayudas, _radiologia_ayuda_diagnostica(doc))
    for doc in context.get("laboratorio") or []:
        _append_ayuda(ayudas, _laboratorio_ayuda_diagnostica(doc))
    history_documents = []
    seen_history_ids: set[str] = set()
    for doc in [context.get("historia"), *(context.get("historias_adicionales") or [])]:
        if not isinstance(doc, dict):
            continue
        doc_id = _normalize_whitespace(doc.get("_id") or doc.get("document_uid"))
        if doc_id and doc_id in seen_history_ids:
            continue
        if doc_id:
            seen_history_ids.add(doc_id)
        history_documents.append(doc)
    for doc_key, source in (("quirurgico", "quirurgico"),):
        for ayuda in _referencias_ayudas_desde_documento(context.get(doc_key), source):
            _append_ayuda(ayudas, ayuda)
    for doc in history_documents:
        for ayuda in _referencias_ayudas_desde_documento(doc, "historia_clinica"):
            _append_ayuda(ayudas, ayuda)
    for doc in context.get("generico") or []:
        for ayuda in _referencias_ayudas_desde_documento(doc, "generico"):
            _append_ayuda(ayudas, ayuda)
    for ayuda in _referencias_ayudas_desde_qx(context):
        _append_ayuda(ayudas, ayuda)
    for ayuda in coerce_ayudas_diagnosticas(
        (context.get("pdf_draft") or {}).get("manual_ayudas_diagnosticas")
    ):
        _append_ayuda(ayudas, ayuda)

    _apply_invoice_support_reviews(ayudas)

    excluded = (context.get("pdf_draft") or {}).get("excluded_ayudas_diagnosticas_keys")
    for item in ayudas:
        item["exportable"] = evaluar_ayuda_diagnostica_exportable(item, excluded)
    return ayudas


def build_ayudas_diagnosticas(context: dict[str, Any]) -> list[dict[str, Any]]:
    """Legacy facade for the common diagnostic-aid consolidation operation."""
    return consolidar_ayudas_diagnosticas(context)


def merge_soat_results(
    existing: list[dict[str, Any]] | None, nuevos: list[dict[str, Any]] | None
) -> list[dict[str, Any]]:
    merged: list[dict[str, Any]] = []
    seen: set[tuple[str, str]] = set()
    for item in (existing or []) + (nuevos or []):
        if not isinstance(item, dict):
            continue
        key = (
            str(item.get("codigo_soat") or "").strip().lower(),
            str(item.get("descripcion") or "").strip().lower(),
        )
        if key in seen:
            continue
        seen.add(key)
        merged.append(dict(item))
    return merged
