from __future__ import annotations

import hashlib
import re
from dataclasses import dataclass, field
from datetime import datetime
from typing import Any


@dataclass
class JobDTO:
    external_id: str
    source: str
    title: str
    company: str = ""
    location: str = ""
    remote: bool = False
    salary_min: int | None = None
    salary_max: int | None = None
    salary_currency: str | None = None
    url: str = ""
    description: str = ""
    lang: str = "en"
    tags: list[str] = field(default_factory=list)
    requires_english_fluent: bool = False
    hire_from_spain_ok: bool | None = None
    posted_at: datetime | None = None
    raw: dict[str, Any] = field(default_factory=dict)

    def fingerprint(self) -> str:
        key = f"{self.title.lower().strip()}|{self.company.lower().strip()}|{self.url.split('?')[0]}"
        return hashlib.sha1(key.encode("utf-8")).hexdigest()


class BaseScraper:
    key: str = "base"
    name: str = "Base"

    def __init__(self, config: dict[str, Any] | None = None, user_agent: str = "JobsWorld/1.0"):
        self.config = config or {}
        self.user_agent = user_agent

    def fetch(self) -> list[JobDTO]:
        raise NotImplementedError

    @staticmethod
    def detect_english_fluent(text: str) -> bool:
        patterns = [
            r"fluent english",
            r"native english",
            r"english\s*(?:fluency|native|proficient)",
            r"c1\s*english",
            r"c2\s*english",
            r"ingl[eé]s\s*(?:fluido|nativo|avanzado)",
        ]
        low = text.lower()
        return any(re.search(p, low) for p in patterns)

    @staticmethod
    def detect_remote(text: str, location: str = "") -> bool:
        blob = f"{text} {location}".lower()
        return any(
            w in blob
            for w in ["remote", "remoto", "work from home", "wfh", "distributed", "anywhere", "worldwide"]
        )

    @staticmethod
    def detect_lang(text: str) -> str:
        es_markers = ["experiencia", "requisitos", "jornada", "contrato", "salario", "desarrollador"]
        hits = sum(1 for m in es_markers if m in text.lower())
        return "es" if hits >= 2 else "en"

    @staticmethod
    def parse_salary_usd(text: str) -> tuple[int | None, int | None, str | None]:
        # $80k-$120k or 80000-120000 USD
        m = re.search(
            r"\$?\s*(\d{2,3})\s*[kK]\s*[-–to]+\s*\$?\s*(\d{2,3})\s*[kK]",
            text,
        )
        if m:
            return int(m.group(1)) * 1000, int(m.group(2)) * 1000, "USD"
        m = re.search(r"\$?\s*(\d{2,3})\s*[kK]", text)
        if m:
            v = int(m.group(1)) * 1000
            return v, v, "USD"
        m = re.search(r"(\d{4,6})\s*[-–]\s*(\d{4,6})\s*(USD|EUR|€|\$)?", text, re.I)
        if m:
            cur = (m.group(3) or "USD").upper().replace("€", "EUR").replace("$", "USD")
            return int(m.group(1)), int(m.group(2)), cur
        return None, None, None
