from __future__ import annotations

import re
from typing import Any


EMAIL_RE = re.compile(r"\b[A-Z0-9._%+\-]+@[A-Z0-9.\-]+\.[A-Z]{2,}\b", re.I)
# Avoid matching CSS hex / versions as phones loosely
PHONE_RE = re.compile(
    r"(?:\+|00)?(?:\d[\s().\-]?){8,15}\d",
)
LINKEDIN_RE = re.compile(r"https?://(?:www\.)?linkedin\.com/(?:in|company|pub)/[^\s)\"'<>]+", re.I)
NAME_HINT_RE = re.compile(
    r"(?:contact(?:o)?|recruiter|hiring manager|responsable|publicado por|posted by)\s*[:\-]\s*([A-ZÁÉÍÓÚÑ][\wÁÉÍÓÚÑáéíóúñ.'\-]+(?:\s+[A-ZÁÉÍÓÚÑ][\wÁÉÍÓÚÑáéíóúñ.'\-]+){0,3})",
    re.I,
)

NOISE_EMAIL = {
    "example@example.com",
    "email@email.com",
    "noreply@",
    "no-reply@",
    "donotreply@",
}


def _clean_phone(raw: str) -> str | None:
    digits = re.sub(r"[^\d+]", "", raw)
    if len(re.sub(r"\D", "", digits)) < 9:
        return None
    # discard likely IDs / salaries
    if len(re.sub(r"\D", "", digits)) > 15:
        return None
    return raw.strip()


def extract_contacts(text: str, extra: dict[str, Any] | None = None) -> dict[str, Any]:
    blob = text or ""
    if extra:
        blob += "\n" + " ".join(str(v) for v in extra.values() if v)

    emails = []
    for m in EMAIL_RE.findall(blob):
        low = m.lower()
        if any(n in low for n in NOISE_EMAIL):
            continue
        if low.endswith((".png", ".jpg", ".gif", ".svg")):
            continue
        emails.append(m)

    phones = []
    for m in PHONE_RE.findall(blob):
        cleaned = _clean_phone(m)
        if cleaned:
            phones.append(cleaned)

    linkedins = LINKEDIN_RE.findall(blob)
    name = None
    nm = NAME_HINT_RE.search(blob)
    if nm:
        name = nm.group(1).strip()

    # Deduplicate preserving order
    def uniq(items: list[str]) -> list[str]:
        seen: set[str] = set()
        out: list[str] = []
        for i in items:
            key = i.lower().strip()
            if key in seen:
                continue
            seen.add(key)
            out.append(i.strip())
        return out

    emails_u = uniq(emails)
    phones_u = uniq(phones)
    linked_u = uniq(linkedins)

    return {
        "contact_email": emails_u[0] if emails_u else None,
        "contact_phone": phones_u[0] if phones_u else None,
        "contact_name": name,
        "contact_linkedin": linked_u[0] if linked_u else None,
        "contact_emails": emails_u[:8],
        "contact_phones": phones_u[:5],
        "contact_links": linked_u[:5],
    }
