"""Normalisierung von Leerwerten, Kennzeichen, VIN und Spaltenüberschriften."""

from __future__ import annotations

import math
import re
import unicodedata
from datetime import date, datetime
from typing import Any

EMPTY_VALUES = {
    "",
    "-",
    "--",
    "---",
    "n/a",
    "na",
    "null",
    "none",
    "nan",
    "nicht vorhanden",
    "unbekannt",
}

_UMLAUT = str.maketrans(
    {
        "ä": "ae",
        "ö": "oe",
        "ü": "ue",
        "Ä": "ae",
        "Ö": "oe",
        "Ü": "ue",
        "ß": "ss",
    }
)


def _stringify(value: Any) -> str:
    if value is None:
        return ""
    if isinstance(value, float) and (math.isnan(value) or math.isinf(value)):
        return ""
    text = str(value).strip()
    if text.lower() == "nan":
        return ""
    return text


def is_empty(value: Any) -> bool:
    if value is None:
        return True
    if isinstance(value, float) and (math.isnan(value) or math.isinf(value)):
        return True
    if isinstance(value, datetime) or isinstance(value, date):
        return False
    text = _stringify(value)
    if not text:
        return True
    return text.lower() in EMPTY_VALUES


def clean_text(value: Any) -> str:
    if is_empty(value):
        return ""
    if isinstance(value, datetime):
        return value.strftime("%d.%m.%Y")
    if isinstance(value, date):
        return value.strftime("%d.%m.%Y")
    if isinstance(value, float) and value.is_integer():
        return str(int(value))
    if isinstance(value, int):
        return str(value)
    return _stringify(value)


def normalize_header(value: Any) -> str:
    text = _stringify(value).translate(_UMLAUT).lower()
    text = unicodedata.normalize("NFKD", text)
    text = "".join(ch for ch in text if not unicodedata.combining(ch))
    return re.sub(r"[^a-z0-9]", "", text)


def header_tokens(value: Any) -> set[str]:
    raw = _stringify(value)
    tokens = set()
    if not raw:
        return tokens
    tokens.add(normalize_header(raw))
    for part in re.split(r"[=:|/]", raw):
        normalized = normalize_header(part)
        if normalized:
            tokens.add(normalized)
    return {t for t in tokens if t}


def normalize_plate(value: Any) -> str:
    if is_empty(value):
        return ""
    text = _stringify(value).upper()
    return re.sub(r"[^A-Z0-9]", "", text)


def normalize_vin(value: Any) -> str:
    if is_empty(value):
        return ""
    text = _stringify(value).upper()
    return re.sub(r"[^A-Z0-9]", "", text)


def normalize_name(value: Any) -> str:
    if is_empty(value):
        return ""
    text = _stringify(value).translate(_UMLAUT).lower()
    text = unicodedata.normalize("NFKD", text)
    text = "".join(ch for ch in text if not unicodedata.combining(ch))
    text = re.sub(r"[^a-z0-9]+", " ", text)
    return re.sub(r"\s+", " ", text).strip()


def format_zip(value: Any) -> str:
    if is_empty(value):
        return ""
    if isinstance(value, float) and value.is_integer():
        number = str(int(value))
        return number.zfill(5) if number.isdigit() and len(number) <= 5 else number
    if isinstance(value, int):
        number = str(value)
        return number.zfill(5) if len(number) <= 5 else number
    text = _stringify(value)
    if text.endswith(".0") and text[:-2].isdigit():
        text = text[:-2]
    if text.isdigit() and len(text) <= 5:
        return text.zfill(5)
    return text


def combine_street(street: Any, house_no: Any) -> str:
    street_text = clean_text(street)
    house_text = clean_text(house_no)
    if not house_text:
        return street_text
    if not street_text:
        return house_text
    if house_text.lower() in street_text.lower():
        return street_text
    return f"{street_text} {house_text}".strip()


_DATE_FORMATS = (
    "%d.%m.%Y",
    "%d.%m.%y",
    "%Y-%m-%d",
    "%d/%m/%Y",
    "%d/%m/%y",
    "%Y/%m/%d",
    "%d-%m-%Y",
    "%m/%d/%Y",
    "%Y%m%d",
)


def parse_date(value: Any) -> date | None:
    if is_empty(value):
        return None
    if isinstance(value, datetime):
        return value.date()
    if isinstance(value, date):
        return value
    if hasattr(value, "to_pydatetime"):
        try:
            return value.to_pydatetime().date()
        except Exception:
            return None
    if isinstance(value, (int, float)) and not isinstance(value, bool):
        if isinstance(value, float) and value.is_integer():
            value = int(value)
        if isinstance(value, int) and 30000 < value < 80000:
            try:
                from datetime import timedelta

                return (datetime(1899, 12, 30) + timedelta(days=int(value))).date()
            except Exception:
                return None
    text = _stringify(value)
    if text.endswith(".0") and text[:-2].isdigit():
        text = text[:-2]
    for fmt in _DATE_FORMATS:
        try:
            return datetime.strptime(text, fmt).date()
        except ValueError:
            continue
    return None
