from __future__ import annotations

import re
from typing import Any
from urllib.parse import quote_plus, urljoin

import httpx
from bs4 import BeautifulSoup

from app.scrapers.base import BaseScraper, JobDTO
from app.services.job_urls import href_looks_usable


class WeWorkRemotelyScraper(BaseScraper):
    key = "weworkremotely"
    name = "We Work Remotely"

    def fetch(self) -> list[JobDTO]:
        # Varias categorías RSS (no solo programming)
        feeds = self.config.get("feeds") or [
            "https://weworkremotely.com/categories/remote-programming-jobs.rss",
            "https://weworkremotely.com/categories/remote-customer-support-jobs.rss",
            "https://weworkremotely.com/categories/remote-sales-and-marketing-jobs.rss",
            "https://weworkremotely.com/categories/remote-design-jobs.rss",
            "https://weworkremotely.com/categories/remote-product-jobs.rss",
            "https://weworkremotely.com/categories/remote-devops-sysadmin-jobs.rss",
            "https://weworkremotely.com/remote-jobs.rss",
        ]
        headers = {"User-Agent": self.user_agent, "Accept": "application/rss+xml, application/xml, text/xml, */*"}
        jobs: list[JobDTO] = []
        seen: set[str] = set()
        with httpx.Client(timeout=45.0, headers=headers, follow_redirects=True) as client:
            for url in feeds[:7]:
                try:
                    resp = client.get(url)
                    resp.raise_for_status()
                    soup = BeautifulSoup(resp.text, "lxml-xml")
                except Exception:
                    continue
                for item in soup.select("item"):
                    title = (item.title.get_text(strip=True) if item.title else "")[:500]
                    link = item.link.get_text(strip=True) if item.link else ""
                    desc = item.description.get_text(" ", strip=True) if item.description else title
                    if not title or not link or link in seen:
                        continue
                    seen.add(link)
                    company = ""
                    role = title
                    if ":" in title:
                        company, role = [p.strip() for p in title.split(":", 1)]
                    external_id = link.rstrip("/").split("/")[-1] or hashlib_id(link)
                    jobs.append(
                        JobDTO(
                            external_id=external_id,
                            source=self.key,
                            title=role or title,
                            company=company,
                            location="Remote",
                            remote=True,
                            url=link,
                            description=desc[:8000],
                            lang="en",
                            tags=["remote", "wwr"],
                            requires_english_fluent=self.detect_english_fluent(desc),
                            hire_from_spain_ok=None,
                            raw={"title": title, "link": link, "feed": url},
                        )
                    )
        return jobs[:150]


class InfoJobsScraper(BaseScraper):
    key = "infojobs"
    name = "InfoJobs"

    def fetch(self) -> list[JobDTO]:
        query = self.config.get("query") or "empleo"
        url = f"https://www.infojobs.net/jobsearch/search-results/list.xhtml?keyword={quote_plus(query)}&segmentId="
        headers = {
            "User-Agent": self.user_agent,
            "Accept-Language": "es-ES,es;q=0.9",
        }
        jobs: list[JobDTO] = []
        try:
            with httpx.Client(timeout=45.0, headers=headers, follow_redirects=True) as client:
                resp = client.get(url)
                if resp.status_code >= 400:
                    return self._fallback_sample(query, "infojobs blocked or unavailable")
                soup = BeautifulSoup(resp.text, "lxml")
        except Exception as exc:  # noqa: BLE001
            return self._fallback_sample(query, str(exc))

        for card in soup.select("div.ij-OfferCardContent, article, div.offer"):
            a = card.select_one("a[href*='/ofertas-trabajo/'], a[href*='oferta']")
            if not a:
                continue
            href = a.get("href") or ""
            title = a.get_text(" ", strip=True)[:500]
            if not title or not href_looks_usable(href):
                continue
            full_url = urljoin("https://www.infojobs.net", href)
            company_el = card.select_one(".ij-OfferCardContent-description-title-link, .company")
            location_el = card.select_one(".ij-OfferCardContent-description-list-item, .location")
            company = company_el.get_text(strip=True) if company_el else ""
            location = location_el.get_text(strip=True) if location_el else "España"
            text = card.get_text(" ", strip=True)
            remote = self.detect_remote(text, location)
            smin, smax, scur = self.parse_salary_usd(text)
            if not scur and ("€" in text or "EUR" in text.upper()):
                scur = "EUR"
            jobs.append(
                JobDTO(
                    external_id=hashlib_id(full_url),
                    source=self.key,
                    title=title,
                    company=company,
                    location=location,
                    remote=remote,
                    salary_min=smin,
                    salary_max=smax,
                    salary_currency=scur or "EUR",
                    url=full_url,
                    description=text[:5000],
                    lang="es",
                    tags=["spain", "infojobs"],
                    requires_english_fluent=self.detect_english_fluent(text),
                    hire_from_spain_ok=True,
                    raw={"url": full_url},
                )
            )
        return jobs[:80] if jobs else self._fallback_sample(query, "no cards parsed")

    def _fallback_sample(self, query: str, reason: str) -> list[JobDTO]:
        # Soft placeholder so pipeline stays testable when portal blocks bots
        return [
            JobDTO(
                external_id=f"infojobs-placeholder-{hashlib_id(query)[:8]}",
                source=self.key,
                title=f"[InfoJobs] Buscar: {query}",
                company="InfoJobs",
                location="España",
                remote=True,
                url=f"https://www.infojobs.net/jobsearch/search-results/list.xhtml?keyword={quote_plus(query)}",
                description=f"Portal protegido o estructura cambiada ({reason}). Abre el enlace para buscar manualmente.",
                lang="es",
                tags=["spain", "manual-bridge"],
                hire_from_spain_ok=True,
                raw={"fallback": True, "reason": reason},
            )
        ]


class TecnoempleoScraper(BaseScraper):
    key = "tecnoempleo"
    name = "Tecnoempleo"

    def fetch(self) -> list[JobDTO]:
        query = self.config.get("query") or "desarrollador"
        url = f"https://www.tecnoempleo.com/ofertas-trabajo/{quote_plus(query)}"
        headers = {"User-Agent": self.user_agent, "Accept-Language": "es-ES,es;q=0.9"}
        try:
            with httpx.Client(timeout=45.0, headers=headers, follow_redirects=True) as client:
                resp = client.get(url)
                if resp.status_code >= 400:
                    return self._fallback(query, f"HTTP {resp.status_code}")
                soup = BeautifulSoup(resp.text, "lxml")
        except Exception as exc:  # noqa: BLE001
            return self._fallback(query, str(exc))

        jobs: list[JobDTO] = []
        for a in soup.select("a[href*='/oferta-empleo/'], a[href*='oferta']"):
            href = a.get("href") or ""
            title = a.get_text(" ", strip=True)
            if not title or len(title) < 5 or not href_looks_usable(href):
                continue
            full_url = urljoin("https://www.tecnoempleo.com", href)
            parent_text = a.parent.get_text(" ", strip=True) if a.parent else title
            jobs.append(
                JobDTO(
                    external_id=hashlib_id(full_url),
                    source=self.key,
                    title=title[:500],
                    company="",
                    location="España",
                    remote=self.detect_remote(parent_text),
                    url=full_url,
                    description=parent_text[:5000],
                    lang="es",
                    tags=["spain", "tecnoempleo"],
                    requires_english_fluent=self.detect_english_fluent(parent_text),
                    hire_from_spain_ok=True,
                    salary_currency="EUR",
                    raw={"url": full_url},
                )
            )
        # dedupe
        seen: set[str] = set()
        out: list[JobDTO] = []
        for j in jobs:
            if j.external_id in seen:
                continue
            seen.add(j.external_id)
            out.append(j)
        return out[:80] if out else self._fallback(query, "no links")

    def _fallback(self, query: str, reason: str) -> list[JobDTO]:
        return [
            JobDTO(
                external_id=f"tecnoempleo-placeholder-{hashlib_id(query)[:8]}",
                source=self.key,
                title=f"[Tecnoempleo] Buscar: {query}",
                company="Tecnoempleo",
                location="España",
                remote=True,
                url=f"https://www.tecnoempleo.com/ofertas-trabajo/{quote_plus(query)}",
                description=f"Portal no scrapeable ahora ({reason}). Usa el enlace.",
                lang="es",
                tags=["spain", "manual-bridge"],
                hire_from_spain_ok=True,
                raw={"fallback": True, "reason": reason},
            )
        ]


class IndeedScraper(BaseScraper):
    key = "indeed"
    name = "Indeed"

    def fetch(self) -> list[JobDTO]:
        query = self.config.get("query") or "remote software engineer"
        location = self.config.get("location") or "remote"
        url = f"https://www.indeed.com/jobs?q={quote_plus(query)}&l={quote_plus(location)}"
        headers = {
            "User-Agent": self.user_agent,
            "Accept-Language": "en-US,en;q=0.9",
        }
        try:
            with httpx.Client(timeout=45.0, headers=headers, follow_redirects=True) as client:
                resp = client.get(url)
                if resp.status_code >= 400 or "captcha" in resp.text.lower():
                    return self._fallback(query, location, f"HTTP {resp.status_code} or captcha")
                soup = BeautifulSoup(resp.text, "lxml")
        except Exception as exc:  # noqa: BLE001
            return self._fallback(query, location, str(exc))

        jobs: list[JobDTO] = []
        for card in soup.select("div.job_seen_beacon, div.cardOutline, a.jcs-JobTitle"):
            a = card if card.name == "a" else card.select_one("a[href*='/rc/clk'], a[href*='/viewjob'], h2 a")
            if not a:
                continue
            href = a.get("href") or ""
            title = a.get_text(" ", strip=True)
            if not title or not href_looks_usable(href):
                continue
            full_url = urljoin("https://www.indeed.com", href)
            text = card.get_text(" ", strip=True) if card.name != "a" else title
            company_m = re.search(r"([A-Z][\w&.\- ]{2,40})", text)
            smin, smax, scur = self.parse_salary_usd(text)
            jobs.append(
                JobDTO(
                    external_id=hashlib_id(full_url),
                    source=self.key,
                    title=title[:500],
                    company=company_m.group(1).strip() if company_m else "",
                    location=location,
                    remote=self.detect_remote(text, location),
                    salary_min=smin,
                    salary_max=smax,
                    salary_currency=scur or "USD",
                    url=full_url,
                    description=text[:5000],
                    lang=self.detect_lang(text),
                    tags=["indeed"],
                    requires_english_fluent=self.detect_english_fluent(text),
                    hire_from_spain_ok=None,
                    raw={"url": full_url},
                )
            )
        seen: set[str] = set()
        out: list[JobDTO] = []
        for j in jobs:
            if j.external_id in seen:
                continue
            seen.add(j.external_id)
            out.append(j)
        return out[:60] if out else self._fallback(query, location, "parse empty")

    def _fallback(self, query: str, location: str, reason: str) -> list[JobDTO]:
        return [
            JobDTO(
                external_id=f"indeed-bridge-{hashlib_id(query+location)[:8]}",
                source=self.key,
                title=f"[Indeed] {query}",
                company="Indeed",
                location=location,
                remote=True,
                url=f"https://www.indeed.com/jobs?q={quote_plus(query)}&l={quote_plus(location)}",
                description=f"Indeed suele bloquear bots ({reason}). Abre la búsqueda en el navegador.",
                lang="en",
                tags=["indeed", "manual-bridge"],
                raw={"fallback": True, "reason": reason},
            )
        ]


def hashlib_id(value: str) -> str:
    import hashlib

    return hashlib.sha1(value.encode("utf-8")).hexdigest()
