from __future__ import annotations

from typing import Any
from urllib.parse import quote_plus, urljoin

import httpx
from bs4 import BeautifulSoup

from app.scrapers.base import BaseScraper, JobDTO
from app.scrapers.html_scrapers import hashlib_id
from app.services.job_urls import href_looks_usable


class LinkedInJobsScraper(BaseScraper):
    """LinkedIn Jobs via guest search HTML (rate-limited; may fall back to search bridges)."""

    key = "linkedin"
    name = "LinkedIn Empleos"

    def fetch(self) -> list[JobDTO]:
        queries = self.config.get("queries") or [
            "programador remoto",
            "desarrollador remote Spain",
            "devops remoto teletrabajo",
            "software engineer remote Spanish speaking",
            "full stack remote Europe Spanish",
        ]
        jobs: list[JobDTO] = []
        for q in queries:
            jobs.extend(self._search(q))
        return self._dedupe(jobs)[:120]

    def _search(self, query: str) -> list[JobDTO]:
        # f_WT=2 = solo remoto. Para búsquedas locales (Valencia…) no forzar remoto.
        location = self.config.get("location") or "Europe"
        remote_only = bool(self.config.get("remote_only", True))
        q_lower = query.lower()
        if "valencia" in q_lower:
            location = "Valencia, Comunidad Valenciana, España"
            remote_only = False
        elif any(x in q_lower for x in ("madrid", "barcelona", "españa", "spain")):
            remote_only = False
            if "madrid" in q_lower:
                location = "Madrid, Comunidad de Madrid, España"
            elif "barcelona" in q_lower:
                location = "Barcelona, Cataluña, España"
            elif self.config.get("location") in (None, "", "Europe"):
                location = "Spain"
        remote_qs = "&f_WT=2" if remote_only else ""
        url = (
            "https://www.linkedin.com/jobs-guest/jobs/api/seeMoreJobPostings/search"
            f"?keywords={quote_plus(query)}&location={quote_plus(location)}{remote_qs}&start=0"
        )
        headers = {
            "User-Agent": self.user_agent,
            "Accept-Language": "es-ES,es;q=0.9,en;q=0.5",
            "Accept": "text/html,application/xhtml+xml",
        }
        try:
            with httpx.Client(timeout=45.0, headers=headers, follow_redirects=True) as client:
                resp = client.get(url)
                if resp.status_code >= 400:
                    return self._bridge(query, location, f"HTTP {resp.status_code}", remote_only)
                soup = BeautifulSoup(resp.text, "lxml")
        except Exception as exc:  # noqa: BLE001
            return self._bridge(query, location, str(exc), remote_only)

        jobs: list[JobDTO] = []
        cards = soup.select("li") or soup.select("div.base-card, div.job-search-card")
        for card in cards:
            a = card.select_one("a.base-card__full-link, a[href*='/jobs/view/']")
            if not a:
                continue
            href = a.get("href") or ""
            title_el = card.select_one("h3, .base-search-card__title")
            company_el = card.select_one("h4, .base-search-card__subtitle")
            loc_el = card.select_one(".job-search-card__location, .base-search-card__metadata")
            title = (title_el.get_text(strip=True) if title_el else a.get_text(" ", strip=True))[:500]
            if not title or len(title) < 3:
                continue
            if not href_looks_usable(href):
                continue
            company = company_el.get_text(strip=True) if company_el else ""
            location = loc_el.get_text(strip=True) if loc_el else ("Remote" if remote_only else location)
            full_url = href.split("?")[0]
            if full_url.startswith("/"):
                full_url = urljoin("https://www.linkedin.com", full_url)
            text = card.get_text(" ", strip=True)
            is_remote = remote_only or self.detect_remote(text + " " + location)
            jobs.append(
                JobDTO(
                    external_id=hashlib_id(full_url),
                    source=self.key,
                    title=title,
                    company=company,
                    location=location,
                    remote=is_remote,
                    url=full_url,
                    description=text[:5000],
                    lang=self.detect_lang(text),
                    tags=["linkedin"] + (["remote"] if is_remote else ["local"]),
                    requires_english_fluent=self.detect_english_fluent(text),
                    hire_from_spain_ok=True if "spain" in location.lower() or "espa" in location.lower() or "valencia" in location.lower() else None,
                    raw={"query": query, "url": full_url},
                )
            )
        return jobs if jobs else self._bridge(query, location, "empty parse", remote_only)

    def _bridge(self, query: str, location: str, reason: str, remote_only: bool = True) -> list[JobDTO]:
        remote_qs = "&f_WT=2" if remote_only else ""
        search = (
            "https://www.linkedin.com/jobs/search/"
            f"?keywords={quote_plus(query)}&location={quote_plus(location)}{remote_qs}"
        )
        return [
            JobDTO(
                external_id=f"linkedin-bridge-{hashlib_id(query + location)[:10]}",
                source=self.key,
                title=f"[LinkedIn] {query}",
                company="LinkedIn",
                location=location,
                remote=remote_only,
                url=search,
                description=(
                    f"Búsqueda LinkedIn ({'remoto' if remote_only else 'local/híbrido'}; {reason}). "
                    "Abre el enlace para ver resultados. "
                    "También puedes pegar URLs de ofertas concretas en Fuentes → Manual."
                ),
                lang="es",
                tags=["linkedin", "manual-bridge"] + (["remote"] if remote_only else ["local"]),
                hire_from_spain_ok=True,
                raw={"fallback": True, "reason": reason},
            )
        ]

    @staticmethod
    def _dedupe(jobs: list[JobDTO]) -> list[JobDTO]:
        seen: set[str] = set()
        out: list[JobDTO] = []
        for j in jobs:
            if j.external_id in seen:
                continue
            seen.add(j.external_id)
            out.append(j)
        return out


class HimalayasScraper(BaseScraper):
    key = "himalayas"
    name = "Himalayas"

    def fetch(self) -> list[JobDTO]:
        url = "https://himalayas.app/jobs/api?limit=50"
        headers = {"User-Agent": self.user_agent, "Accept": "application/json"}
        try:
            with httpx.Client(timeout=45.0, headers=headers, follow_redirects=True) as client:
                resp = client.get(url)
                resp.raise_for_status()
                data = resp.json()
        except Exception as exc:  # noqa: BLE001
            return [
                JobDTO(
                    external_id="himalayas-bridge",
                    source=self.key,
                    title="[Himalayas] Remote tech jobs",
                    company="Himalayas",
                    location="Remote",
                    remote=True,
                    url="https://himalayas.app/jobs",
                    description=f"API no disponible ({exc}). Abre el portal.",
                    tags=["himalayas", "remote"],
                    raw={"fallback": True},
                )
            ]

        items = data if isinstance(data, list) else data.get("jobs") or data.get("data") or []
        jobs: list[JobDTO] = []
        for item in items:
            if not isinstance(item, dict):
                continue
            title = item.get("title") or item.get("jobTitle") or ""
            company = ""
            if isinstance(item.get("companyName"), str):
                company = item["companyName"]
            elif isinstance(item.get("company"), dict):
                company = item["company"].get("name") or ""
            desc = item.get("description") or item.get("excerpt") or ""
            link = item.get("applicationLink") or item.get("url") or item.get("guid") or ""
            if not link and item.get("slug"):
                link = f"https://himalayas.app/jobs/{item['slug']}"
            if not title or not link:
                continue
            loc = ", ".join(item.get("locationRestrictions") or []) or "Remote"
            smin, smax, scur = self.parse_salary_usd(str(item.get("minSalary") or "") + " " + desc)
            if item.get("minSalary") and not smin:
                try:
                    smin = int(item["minSalary"])
                    smax = int(item.get("maxSalary") or smin)
                    scur = item.get("currency") or "USD"
                except (TypeError, ValueError):
                    pass
            jobs.append(
                JobDTO(
                    external_id=str(item.get("id") or item.get("slug") or hashlib_id(link)),
                    source=self.key,
                    title=title,
                    company=company,
                    location=loc,
                    remote=True,
                    salary_min=smin,
                    salary_max=smax,
                    salary_currency=scur,
                    url=link,
                    description=desc,
                    lang=self.detect_lang(desc),
                    tags=[str(t).lower() for t in (item.get("categories") or item.get("skills") or [])][:20],
                    requires_english_fluent=self.detect_english_fluent(desc),
                    hire_from_spain_ok=None,
                    raw=item,
                )
            )
        return jobs


class JobicyScraper(BaseScraper):
    key = "jobicy"
    name = "Jobicy"

    def fetch(self) -> list[JobDTO]:
        url = "https://jobicy.com/api/v2/remote-jobs?count=50&tag=software"
        headers = {"User-Agent": self.user_agent}
        with httpx.Client(timeout=45.0, headers=headers, follow_redirects=True) as client:
            resp = client.get(url)
            resp.raise_for_status()
            payload = resp.json()

        jobs: list[JobDTO] = []
        for item in payload.get("jobs", []):
            desc = item.get("jobDescription") or ""
            smin, smax, scur = self.parse_salary_usd((item.get("jobSalary") or "") + " " + desc)
            jobs.append(
                JobDTO(
                    external_id=str(item.get("id")),
                    source=self.key,
                    title=item.get("jobTitle") or "Untitled",
                    company=item.get("companyName") or "",
                    location=item.get("jobGeo") or "Remote",
                    remote=True,
                    salary_min=smin,
                    salary_max=smax,
                    salary_currency=scur,
                    url=item.get("url") or "",
                    description=desc,
                    lang=self.detect_lang(desc),
                    tags=[str(t).lower() for t in (item.get("jobIndustry") or [])]
                    if isinstance(item.get("jobIndustry"), list)
                    else [],
                    requires_english_fluent=self.detect_english_fluent(desc),
                    hire_from_spain_ok=None,
                    raw=item,
                )
            )
        return jobs


class GetOnBoardScraper(BaseScraper):
    """Portal LATAM/ES — muchas ofertas en español, remoto."""

    key = "getonboard"
    name = "Get on Board"

    def fetch(self) -> list[JobDTO]:
        url = "https://www.getonbrd.com/api/v0/categories/programming/jobs?per_page=50"
        headers = {"User-Agent": self.user_agent, "Accept": "application/json"}
        try:
            with httpx.Client(timeout=45.0, headers=headers, follow_redirects=True) as client:
                resp = client.get(url)
                resp.raise_for_status()
                payload = resp.json()
        except Exception as exc:  # noqa: BLE001
            return [
                JobDTO(
                    external_id="getonboard-bridge",
                    source=self.key,
                    title="[Get on Board] Programación remoto",
                    company="Get on Board",
                    location="LATAM / Remote",
                    remote=True,
                    url="https://www.getonbrd.com/jobs/programming",
                    description=f"API no disponible ({exc}). Portal en español, útil sin inglés fluido.",
                    lang="es",
                    tags=["spanish", "remote", "latam"],
                    hire_from_spain_ok=True,
                    raw={"fallback": True},
                )
            ]

        jobs: list[JobDTO] = []
        for item in payload.get("jobs", payload.get("data", [])):
            attrs = item.get("attributes", item) if isinstance(item, dict) else {}
            title = attrs.get("title") or item.get("title") or ""
            company = ""
            if isinstance(attrs.get("company"), dict):
                company = attrs["company"].get("name") or ""
            desc = attrs.get("description") or attrs.get("preview") or ""
            slug = attrs.get("slug") or item.get("id")
            link = attrs.get("public_url") or (f"https://www.getonbrd.com/jobs/{slug}" if slug else "")
            if not title or not link:
                continue
            remote = bool(attrs.get("remote")) or self.detect_remote(desc)
            jobs.append(
                JobDTO(
                    external_id=str(item.get("id") or slug),
                    source=self.key,
                    title=title,
                    company=company,
                    location=attrs.get("cities_text") or ("Remoto" if remote else ""),
                    remote=remote or True,
                    url=link,
                    description=desc,
                    lang=self.detect_lang(desc) if desc else "es",
                    tags=["spanish", "getonboard"],
                    requires_english_fluent=self.detect_english_fluent(desc),
                    hire_from_spain_ok=True,
                    salary_currency="USD",
                    raw=item if isinstance(item, dict) else {},
                )
            )
        return jobs


class JoobleScraper(BaseScraper):
    key = "jooble"
    name = "Jooble"

    def fetch(self) -> list[JobDTO]:
        queries = list(self.config.get("queries") or [])
        main = self.config.get("query") or "programador remoto teletrabajo"
        if main and main not in queries:
            queries.insert(0, main)
        if not queries:
            queries = [main]
        jobs: list[JobDTO] = []
        for q in queries[:6]:
            jobs.extend(self._search_one(q))
        return self._dedupe(jobs)[:80]

    def _search_one(self, query: str) -> list[JobDTO]:
        loc = self.config.get("location") or ""
        url = f"https://es.jooble.org/SearchResult?ukw={quote_plus(query)}"
        if loc:
            url += f"&loc={quote_plus(loc)}"
        headers = {"User-Agent": self.user_agent, "Accept-Language": "es-ES,es;q=0.9"}
        try:
            with httpx.Client(timeout=45.0, headers=headers, follow_redirects=True) as client:
                resp = client.get(url)
                if resp.status_code >= 400:
                    return self._bridge(query, f"HTTP {resp.status_code}")
                soup = BeautifulSoup(resp.text, "lxml")
        except Exception as exc:  # noqa: BLE001
            return self._bridge(query, str(exc))

        jobs: list[JobDTO] = []
        default_loc = loc or "Remoto / Internacional"
        for a in soup.select("a[href*='/jdp/'], a.serp-item__title-link, h2 a"):
            href = a.get("href") or ""
            title = a.get_text(" ", strip=True)[:500]
            if not title or len(title) < 5:
                continue
            full = urljoin("https://es.jooble.org", href)
            parent = a.find_parent(["article", "div", "li"])
            text = parent.get_text(" ", strip=True) if parent else title
            jobs.append(
                JobDTO(
                    external_id=hashlib_id(full),
                    source=self.key,
                    title=title,
                    company="",
                    location=default_loc,
                    remote=self.detect_remote(text + " " + query),
                    url=full,
                    description=text[:5000],
                    lang="es",
                    tags=["jooble", "spanish"],
                    requires_english_fluent=self.detect_english_fluent(text),
                    hire_from_spain_ok=True,
                    salary_currency="EUR",
                    raw={"url": full, "query": query},
                )
            )
        return jobs if jobs else self._bridge(query, "empty")

    @staticmethod
    def _dedupe(jobs: list[JobDTO]) -> list[JobDTO]:
        seen: set[str] = set()
        out: list[JobDTO] = []
        for j in jobs:
            if j.external_id in seen:
                continue
            seen.add(j.external_id)
            out.append(j)
        return out

    def _bridge(self, query: str, reason: str) -> list[JobDTO]:
        return [
            JobDTO(
                external_id=f"jooble-bridge-{hashlib_id(query)[:8]}",
                source=self.key,
                title=f"[Jooble] {query}",
                company="Jooble",
                location="España / Remoto",
                remote=True,
                url=f"https://es.jooble.org/SearchResult?ukw={quote_plus(query)}",
                description=f"Agregador ES ({reason}). Útil para teletrabajo fuera de España.",
                lang="es",
                tags=["jooble", "spanish", "manual-bridge"],
                hire_from_spain_ok=True,
                raw={"fallback": True},
            )
        ]


class InfoempleoScraper(BaseScraper):
    key = "infoempleo"
    name = "Infoempleo"

    def fetch(self) -> list[JobDTO]:
        query = self.config.get("query") or "programador teletrabajo"
        url = f"https://www.infoempleo.com/trabajo/{quote_plus(query)}/"
        headers = {"User-Agent": self.user_agent, "Accept-Language": "es-ES,es;q=0.9"}
        try:
            with httpx.Client(timeout=45.0, headers=headers, follow_redirects=True) as client:
                resp = client.get(url)
                if resp.status_code >= 400:
                    return self._bridge(query, f"HTTP {resp.status_code}")
                soup = BeautifulSoup(resp.text, "lxml")
        except Exception as exc:  # noqa: BLE001
            return self._bridge(query, str(exc))

        jobs: list[JobDTO] = []
        for a in soup.select("a[href*='/ofertas/'], a[href*='oferta']"):
            href = a.get("href") or ""
            title = a.get_text(" ", strip=True)[:500]
            if not title or len(title) < 6:
                continue
            full = urljoin("https://www.infoempleo.com", href)
            text = a.parent.get_text(" ", strip=True) if a.parent else title
            jobs.append(
                JobDTO(
                    external_id=hashlib_id(full),
                    source=self.key,
                    title=title,
                    company="",
                    location="España / Internacional",
                    remote=self.detect_remote(text),
                    url=full,
                    description=text[:5000],
                    lang="es",
                    tags=["infoempleo", "spanish"],
                    requires_english_fluent=self.detect_english_fluent(text),
                    hire_from_spain_ok=True,
                    salary_currency="EUR",
                    raw={"url": full},
                )
            )
        seen: set[str] = set()
        out = []
        for j in jobs:
            if j.external_id in seen:
                continue
            seen.add(j.external_id)
            out.append(j)
        return out[:80] if out else self._bridge(query, "empty")

    def _bridge(self, query: str, reason: str) -> list[JobDTO]:
        return [
            JobDTO(
                external_id=f"infoempleo-bridge-{hashlib_id(query)[:8]}",
                source=self.key,
                title=f"[Infoempleo] {query}",
                company="Infoempleo",
                location="España",
                remote=True,
                url=f"https://www.infoempleo.com/trabajo/{quote_plus(query)}/",
                description=f"Portal ES ({reason}).",
                lang="es",
                tags=["infoempleo", "spanish", "manual-bridge"],
                hire_from_spain_ok=True,
                raw={"fallback": True},
            )
        ]


class WorkingNomadsScraper(BaseScraper):
    key = "workingnomads"
    name = "Working Nomads"

    def fetch(self) -> list[JobDTO]:
        url = "https://www.workingnomads.com/jobsapi.json?category=development"
        headers = {"User-Agent": self.user_agent}
        try:
            with httpx.Client(timeout=45.0, headers=headers, follow_redirects=True) as client:
                resp = client.get(url)
                resp.raise_for_status()
                data = resp.json()
        except Exception:
            # alternate endpoint
            url = "https://www.workingnomads.co/api/exposed_jobs/"
            with httpx.Client(timeout=45.0, headers=headers, follow_redirects=True) as client:
                resp = client.get(url)
                if resp.status_code >= 400:
                    return [
                        JobDTO(
                            external_id="wn-bridge",
                            source=self.key,
                            title="[Working Nomads] Development remote",
                            company="Working Nomads",
                            location="Remote",
                            remote=True,
                            url="https://www.workingnomads.com/jobs?category=development",
                            description="API no disponible. Abre el portal.",
                            tags=["remote", "nomad"],
                            raw={"fallback": True},
                        )
                    ]
                data = resp.json()

        items = data if isinstance(data, list) else data.get("jobs") or []
        jobs: list[JobDTO] = []
        for item in items:
            title = item.get("title") or ""
            link = item.get("url") or item.get("apply_url") or ""
            if not title or not link:
                continue
            desc = item.get("description") or ""
            jobs.append(
                JobDTO(
                    external_id=str(item.get("id") or hashlib_id(link)),
                    source=self.key,
                    title=title,
                    company=item.get("company_name") or item.get("company") or "",
                    location=item.get("location") or "Remote",
                    remote=True,
                    url=link,
                    description=desc,
                    lang=self.detect_lang(desc),
                    tags=["remote", "nomad"],
                    requires_english_fluent=self.detect_english_fluent(desc),
                    raw=item,
                )
            )
        return jobs[:100]


class LandingJobsScraper(BaseScraper):
    """Tech jobs EU/PT — a menudo remoto y abierto a España."""

    key = "landingjobs"
    name = "Landing.jobs"

    def fetch(self) -> list[JobDTO]:
        url = "https://landing.jobs/api/v1/jobs?limit=50"
        headers = {"User-Agent": self.user_agent, "Accept": "application/json"}
        try:
            with httpx.Client(timeout=45.0, headers=headers, follow_redirects=True) as client:
                resp = client.get(url)
                if resp.status_code >= 400:
                    return self._bridge(f"HTTP {resp.status_code}")
                payload = resp.json()
        except Exception as exc:  # noqa: BLE001
            return self._bridge(str(exc))

        items = payload if isinstance(payload, list) else payload.get("jobs") or payload.get("data") or []
        jobs: list[JobDTO] = []
        for item in items:
            title = item.get("title") or ""
            link = item.get("url") or ""
            if item.get("slug") and not link:
                link = f"https://landing.jobs/at/{item['slug']}"
            if not title:
                continue
            if not link:
                link = "https://landing.jobs/"
            company = ""
            if isinstance(item.get("company"), dict):
                company = item["company"].get("name") or ""
            else:
                company = item.get("company_name") or ""
            desc = item.get("description") or ""
            remote = bool(item.get("remote")) or self.detect_remote(desc)
            jobs.append(
                JobDTO(
                    external_id=str(item.get("id") or hashlib_id(link + title)),
                    source=self.key,
                    title=title,
                    company=company,
                    location=item.get("city") or ("Remote" if remote else "EU"),
                    remote=remote,
                    url=link,
                    description=desc,
                    lang=self.detect_lang(desc),
                    tags=["eu", "landingjobs"],
                    requires_english_fluent=self.detect_english_fluent(desc),
                    hire_from_spain_ok=True,
                    raw=item if isinstance(item, dict) else {},
                )
            )
        return jobs if jobs else self._bridge("empty")

    def _bridge(self, reason: str) -> list[JobDTO]:
        return [
            JobDTO(
                external_id="landingjobs-bridge",
                source=self.key,
                title="[Landing.jobs] Tech EU / remote",
                company="Landing.jobs",
                location="EU",
                remote=True,
                url="https://landing.jobs/jobs",
                description=f"Portal tech europeo ({reason}). Interesante desde España.",
                tags=["eu", "manual-bridge"],
                hire_from_spain_ok=True,
                raw={"fallback": True},
            )
        ]
