from __future__ import annotations

import re
import unicodedata
from typing import Any

from app.services.clinical_document_projection import (
    serialize_analysis_document,
)


def strip_html_tags(value: str) -> str:
    return re.sub(r"<[^>]+>", "", value or "").strip()


def serialize_doc(doc: dict[str, Any] | None) -> dict[str, Any] | None:
    return serialize_analysis_document(doc)


def _normalize_bool(value: Any) -> bool:
    if isinstance(value, bool):
        return value
    if isinstance(value, (int, float)):
        return bool(value)
    normalized = _normalize_ascii(value).lower()
    return normalized in {"1", "si", "sí", "true", "verdadero", "yes"}


def _clean_unique_text_list(values: Any) -> list[str]:
    if not isinstance(values, list):
        return []
    result: list[str] = []
    seen: set[str] = set()
    for item in values:
        text = str(item or "").strip()
        if not text:
            continue
        key = text.lower()
        if key in seen:
            continue
        seen.add(key)
        result.append(text)
    return result


def _normalize_whitespace(value: Any) -> str:
    return re.sub(r"\s+", " ", strip_html_tags(str(value or ""))).strip()


def _normalize_ascii(value: Any) -> str:
    text = _normalize_whitespace(value)
    normalized = unicodedata.normalize("NFKD", text)
    return "".join(char for char in normalized if not unicodedata.combining(char))


def _procedure_key(description: Any, cups: Any = "") -> str:
    code = re.sub(r"\D", "", str(cups or ""))
    text = _normalize_ascii(description).casefold()
    text = re.sub(r"[^a-z0-9]+", " ", text).strip()
    return f"{code}|" if code else f"|{text}"
