import os
import requests
import re
from urllib.parse import urlparse, urljoin
from typing import Optional, Tuple, Dict, Any, List
from scrapers.proxy_manager import get_rotating_proxy_dict
from scrapers.ats_ingestion import clean_html_text, determine_department, determine_experience_level

COMMON_CAREER_PATHS = [
    "/careers",
    "/jobs",
    "/careers/jobs",
    "/careers/job-search",
    "/open-positions"
]

NON_US_REGEX = re.compile(
    r'(?:london|uk|united kingdom|england|germany|berlin|munich|france|paris|canada|toronto|vancouver|montreal|india|bengaluru|bangalore|delhi|singapore|australia|sydney|melbourne|tokyo|japan|brazil|sao paulo|amsterdam|netherlands|emea|apac|latam|poland|warsaw|romania|spain|madrid|barcelona|ireland|dublin|switzerland|zurich)',
    re.IGNORECASE
)

def is_valid_us_location(location_name: str, title: str = "") -> bool:
    combined = f"{location_name} {title}"
    return not bool(NON_US_REGEX.search(combined))

NEGATIVE_REMOTE_REGEX = re.compile(
    r'\b(?:not\s+remote|non-remote|no\s+remote|on-site|onsite|in-office|in\s+office|hybrid|office\s+only|relocation\s+required|must\s+report\s+to\s+office)\b',
    re.IGNORECASE
)

POSITIVE_REMOTE_REGEX = re.compile(
    r'\b(?:100%\s+remote|fully\s+remote|remote\s+only|strictly\s+remote|anywhere\s+in\s+(?:the\s+)?(?:us|usa|united states))\b',
    re.IGNORECASE
)

def parse_is_us_remote(location_name: str, title: str = "", description: str = "") -> bool:
    loc = (location_name or "").strip()
    tit = (title or "").strip()
    desc = (description or "").strip()

    if not is_valid_us_location(loc, tit):
        return False

    loc_lower = loc.lower()
    tit_lower = tit.lower()
    combined_header = f"{loc_lower} {tit_lower}"

    # Negative indicator overrides on location/title unless explicitly 100% / fully remote
    if NEGATIVE_REMOTE_REGEX.search(combined_header):
        if not POSITIVE_REMOTE_REGEX.search(combined_header):
            return False

    # Check first 1500 chars of description for explicit on-site or hybrid mandate
    if desc:
        desc_start = desc[:1500].lower()
        if re.search(r'\b(?:this\s+position\s+is\s+not\s+remote|not\s+a\s+remote\s+position|must\s+be\s+willing\s+to\s+work\s+on-site|requires\s+working\s+on-site|on-site\s+attendance\s+is\s+required|hybrid\s+work\s+schedule|in-person\s+attendance\s+required|must\s+commute\s+to\s+the\s+office)\b', desc_start):
            return False

    is_remote_mention = bool(re.search(r'\b(?:remote|telecommute|work\s+from\s+home|virtual|anywhere)\b', combined_header))
    has_us_indicator = any(u in loc_lower for u in ["us", "usa", "united states", "americas", "ct", "connecticut", "ny", "new york", "ca", "texas", "tx", "fl", "florida", "various", "nationwide"])

    return is_remote_mention and (has_us_indicator or "remote" in loc_lower or "anywhere" in loc_lower)

class SmartCareersCrawler:
    """
    Crawls company domains, finds their /careers or /jobs pages,
    and automatically detects and ingests from Greenhouse, Lever, Ashby, or Workday.
    Supports optional SOCKS5 / HTTP proxies via PROXY_URL or SOCKS5_PROXY.
    """
    def __init__(self, headers: Optional[Dict[str, str]] = None):
        self.headers = headers or {
            "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
            "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
        }
        self.proxies = get_rotating_proxy_dict()

    def _safe_get(self, url: str, timeout: int = 4, allow_redirects: bool = True) -> Optional[requests.Response]:
        """
        Attempts to fetch directly or via rotating Privado proxy with strict timeout.
        """
        proxies = get_rotating_proxy_dict()
        if proxies:
            try:
                return requests.get(url, headers=self.headers, proxies=proxies, timeout=timeout, allow_redirects=allow_redirects)
            except Exception:
                pass

        try:
            return requests.get(url, headers=self.headers, timeout=timeout, allow_redirects=allow_redirects)
        except Exception:
            return None

    def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]:
        parsed = urlparse(url)
        host = parsed.netloc.lower()
        path = parsed.path.strip("/")

        # 1. Greenhouse
        if "boards.greenhouse.io" in host or "job-boards.greenhouse.io" in host:
            parts = [p for p in path.split("/") if p and p != "embed"]
            if parts:
                return "greenhouse", parts[0]

        # 2. Lever
        if "jobs.lever.co" in host:
            parts = [p for p in path.split("/") if p]
            if parts:
                return "lever", parts[0]

        # 3. Ashby
        if "jobs.ashbyhq.com" in host:
            parts = [p for p in path.split("/") if p]
            if parts:
                return "ashby", parts[0]

        # 4. Workday
        if "myworkdayjobs.com" in host:
            match = re.search(r"([a-zA-Z0-9_\-]+)\.(wd[0-9]+)\.myworkdayjobs\.com\/(?:en-US\/)?([a-zA-Z0-9_\-]+)", url, re.IGNORECASE)
            if match:
                tenant, wd_num, site_slug = match.group(1), match.group(2), match.group(3)
                return "workday", f"{tenant}:{wd_num}:{site_slug}"

        if not html_text:
            return None, None

        # Regex fallback on HTML body
        gh_match = re.search(r"boards\.greenhouse\.io\/(?:embed\/job_board\?for=|)([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
        if gh_match:
            return "greenhouse", gh_match.group(1)

        lever_match = re.search(r"jobs\.lever\.co\/([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
        if lever_match:
            return "lever", lever_match.group(1)

        ashby_match = re.search(r"jobs\.ashbyhq\.com\/([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
        if ashby_match:
            return "ashby", ashby_match.group(1)

        wd_match = re.search(r"([a-zA-Z0-9_\-]+)\.(wd[0-9]+)\.myworkdayjobs\.com\/(?:en-US\/)?([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
        if wd_match:
            return "workday", f"{wd_match.group(1)}:{wd_match.group(2)}:{wd_match.group(3)}"

        return None, None

    def probe_domain_for_careers(self, domain: str) -> Tuple[Optional[str], Optional[str], Optional[str]]:
        clean_domain = domain.replace("https://", "").replace("http://", "").split("/")[0].strip().lower()
        base_url = f"https://{clean_domain}"
        found_careers_url = None

        # Probe standard paths with fast timeout
        for path in COMMON_CAREER_PATHS:
            test_url = f"{base_url}{path}"
            r = self._safe_get(test_url, timeout=3, allow_redirects=True)
            if r is not None and r.status_code == 200:
                final_url = r.url
                if not found_careers_url:
                    found_careers_url = final_url
                provider, slug = self.detect_ats_from_url_or_html(final_url, r.text)
                if provider and slug:
                    return provider, slug, final_url

        # Check home page for outbound career / ATS links
        r = self._safe_get(base_url, timeout=3, allow_redirects=True)
        if r is not None and r.status_code == 200:
            provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
            if provider and slug:
                return provider, slug, r.url

            links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby|myworkdayjobs)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE)
            for link in links[:5]:
                target = link if link.startswith("http") else urljoin(base_url, link)
                provider, slug = self.detect_ats_from_url_or_html(target)
                if provider and slug:
                    return provider, slug, target
                cr = self._safe_get(target, timeout=3, allow_redirects=True)
                if cr is not None and cr.status_code == 200:
                    provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
                    if provider and slug:
                        return provider, slug, cr.url
                    if not found_careers_url:
                        found_careers_url = cr.url

        return None, None, found_careers_url

    def fetch_greenhouse_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
        url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true"
        collected = []
        try:
            r = self._safe_get(url, timeout=6)
            if r is None or r.status_code != 200:
                return collected
            data = r.json()
            jobs = data.get("jobs", [])
            for j in jobs:
                title = clean_html_text(j.get("title", ""))
                job_url = j.get("absolute_url", "")
                if not title or not job_url:
                    continue

                location_obj = j.get("location", {})
                location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj)

                if not is_valid_us_location(location_name, title):
                    continue

                desc_clean = clean_html_text(j.get("content", "") or "")
                is_remote = parse_is_us_remote(location_name, title, desc_clean)
                departments = j.get("departments", [])
                dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""

                dept = determine_department(title, dept_text)
                exp_level = determine_experience_level(title)

                collected.append({
                    "title": title,
                    "company": company_name,
                    "location": location_name if location_name != "Remote" else "Remote, USA",
                    "is_remote": is_remote,
                    "department": dept,
                    "experience_level": exp_level,
                    "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.",
                    "salary_min": None,
                    "salary_max": None,
                    "job_url": job_url,
                    "source": "greenhouse"
                })
        except Exception as e:
            print(f"[Greenhouse Warning] {company_name} ({slug}) error: {e}")
        return collected

    def fetch_lever_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
        url = f"https://api.lever.co/v0/postings/{slug}?mode=json"
        collected = []
        try:
            r = self._safe_get(url, timeout=6)
            if r is None or r.status_code != 200:
                return collected
            jobs = r.json()
            for j in jobs:
                title = clean_html_text(j.get("text", ""))
                job_url = j.get("hostedUrl", "")
                if not title or not job_url:
                    continue

                categories = j.get("categories", {})
                location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA"
                workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else ""

                if not is_valid_us_location(location_name, title):
                    continue

                description_plain = j.get("descriptionPlain", "") or j.get("description", "")
                desc_clean = clean_html_text(description_plain)
                is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title, desc_clean)
                dept_text = categories.get("department", "") if isinstance(categories, dict) else ""
                dept = determine_department(title, dept_text)
                exp_level = determine_experience_level(title)

                collected.append({
                    "title": title,
                    "company": company_name,
                    "location": location_name if location_name != "Remote" else "Remote, USA",
                    "is_remote": is_remote,
                    "department": dept,
                    "experience_level": exp_level,
                    "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.",
                    "salary_min": None,
                    "salary_max": None,
                    "job_url": job_url,
                    "source": "lever"
                })
        except Exception as e:
            print(f"[Lever Warning] {company_name} ({slug}) error: {e}")
        return collected

    def fetch_ashby_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
        url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
        collected = []
        try:
            r = self._safe_get(url, timeout=6)
            if r is None or r.status_code != 200:
                return collected
            data = r.json()
            jobs = data.get("jobs", [])
            for j in jobs:
                title = clean_html_text(j.get("title", ""))
                job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{slug}/{j.get('id', '')}"
                if not title or not job_url:
                    continue

                location_name = j.get("location", "Remote, USA") or "Remote, USA"
                if not is_valid_us_location(location_name, title):
                    continue

                description = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "")
                if bool(j.get("isRemote", False)) and not NEGATIVE_REMOTE_REGEX.search(f"{location_name} {title}".lower()):
                    is_remote = True
                else:
                    is_remote = parse_is_us_remote(location_name, title, description)

                dept_text = j.get("department", "")
                dept = determine_department(title, dept_text)
                exp_level = determine_experience_level(title)
                
                salary_min = None
                salary_max = None
                comp = j.get("compensation")
                if isinstance(comp, dict):
                    comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min")
                    comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max")
                    if comp_min:
                        try:
                            salary_min = float(comp_min)
                        except (ValueError, TypeError):
                            pass
                    if comp_max:
                        try:
                            salary_max = float(comp_max)
                        except (ValueError, TypeError):
                            pass

                collected.append({
                    "title": title,
                    "company": company_name,
                    "location": location_name,
                    "is_remote": is_remote,
                    "department": dept,
                    "experience_level": exp_level,
                    "description": description[:2500] or f"Direct posting at {company_name}. Apply via Ashby.",
                    "salary_min": salary_min,
                    "salary_max": salary_max,
                    "job_url": job_url,
                    "source": "ashby"
                })
        except Exception as e:
            print(f"[Ashby Warning] {company_name} ({slug}) error: {e}")
        return collected

    def fetch_workday_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
        """
        Fetches public postings via Workday CXS JSON API.
        slug format: tenant:wd_num:site_slug (e.g. target:wd5:targetcareers)
        """
        collected = []
        try:
            parts = slug.split(":")
            if len(parts) != 3:
                return collected
            tenant, wd_num, site_slug = parts[0], parts[1], parts[2]
            api_url = f"https://{tenant}.{wd_num}.myworkdayjobs.com/wday/cxs/{tenant}/{site_slug}/jobs"
            payload = {"appliedFacets": {}, "limit": 20, "offset": 0, "searchText": ""}
            
            headers = {
                "User-Agent": self.headers["User-Agent"],
                "Accept": "application/json",
                "Content-Type": "application/json"
            }
            r = requests.post(api_url, json=payload, headers=headers, timeout=6)
            if r.status_code != 200:
                return collected
            
            data = r.json()
            postings = data.get("jobPostings", [])
            for p in postings:
                title = clean_html_text(p.get("title", ""))
                ext_path = p.get("externalPath", "")
                if not title or not ext_path:
                    continue

                location_name = p.get("locationsText", "USA") or "USA"
                if not is_valid_us_location(location_name, title):
                    continue

                is_remote = parse_is_us_remote(location_name, title)
                dept = determine_department(title, "")
                exp_level = determine_experience_level(title)
                canonical_url = f"https://{tenant}.{wd_num}.myworkdayjobs.com/en-US/{site_slug}{ext_path}"

                collected.append({
                    "title": title,
                    "company": company_name,
                    "location": location_name,
                    "is_remote": is_remote,
                    "department": dept,
                    "experience_level": exp_level,
                    "description": f"Verified active position at {company_name}. Direct applications, specifications, and full qualifications available on official Workday portal.",
                    "salary_min": None,
                    "salary_max": None,
                    "job_url": canonical_url,
                    "source": "workday"
                })
        except Exception as e:
            print(f"[Workday Warning] {company_name} ({slug}) error: {e}")
        return collected

    def extract_schema_org_job_postings(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]:
        import json
        from bs4 import BeautifulSoup
        jobs = []
        if not html_text:
            return jobs

        soup = BeautifulSoup(html_text, "html.parser")
        scripts = soup.find_all("script", type=re.compile(r"application/ld\+json", re.I))

        for s in scripts:
            try:
                data = json.loads(s.string or "{}")
            except Exception:
                continue

            items = data if isinstance(data, list) else [data]
            for item in items:
                if not isinstance(item, dict):
                    continue
                graph = item.get("@graph", [])
                candidates = graph if isinstance(graph, list) and graph else [item]

                for c in candidates:
                    if not isinstance(c, dict):
                        continue
                    item_type = str(c.get("@type", "")).lower()
                    if "jobposting" in item_type:
                        title = clean_html_text(c.get("title", ""))
                        if not title:
                            continue

                        hiring_org = c.get("hiringOrganization", {})
                        company = default_company
                        if isinstance(hiring_org, dict):
                            company = hiring_org.get("name") or default_company
                        elif isinstance(hiring_org, str):
                            company = hiring_org

                        job_loc = c.get("jobLocation", {})
                        location_name = "Remote, USA"
                        if isinstance(job_loc, dict):
                            addr = job_loc.get("address", {})
                            if isinstance(addr, dict):
                                city = addr.get("addressLocality", "")
                                region = addr.get("addressRegion", "")
                                country = addr.get("addressCountry", "")
                                parts = [p for p in [city, region, country] if p]
                                location_name = ", ".join(parts) or "USA"
                            elif isinstance(addr, str):
                                location_name = addr

                        desc_raw = c.get("description", "")
                        desc_clean = clean_html_text(desc_raw)

                        work_location_type = str(c.get("jobLocationType", "")).upper()
                        if "TELECOMMUTE" in work_location_type and not NEGATIVE_REMOTE_REGEX.search(f"{location_name} {title}".lower()):
                            is_remote = True
                        else:
                            is_remote = parse_is_us_remote(location_name, title, desc_clean)

                        if not is_valid_us_location(location_name, title):
                            continue

                        salary_min = None
                        salary_max = None
                        base_salary = c.get("baseSalary", {})
                        if isinstance(base_salary, dict):
                            val = base_salary.get("value", {})
                            if isinstance(val, dict):
                                s_min = val.get("minValue") or val.get("value")
                                s_max = val.get("maxValue") or val.get("value")
                                if s_min:
                                    try: salary_min = float(s_min)
                                    except: pass
                                if s_max:
                                    try: salary_max = float(s_max)
                                    except: pass

                        job_url = c.get("url") or page_url
                        if job_url.startswith("/"):
                            job_url = urljoin(page_url, job_url)

                        dept = determine_department(title, "")
                        exp_level = determine_experience_level(title)

                        jobs.append({
                            "title": title,
                            "company": company,
                            "location": location_name,
                            "is_remote": is_remote,
                            "department": dept,
                            "experience_level": exp_level,
                            "description": desc_clean[:2500] or f"Job position at {company}. Apply direct on careers portal.",
                            "salary_min": salary_min,
                            "salary_max": salary_max,
                            "job_url": job_url,
                            "source": "web_direct"
                        })
        return jobs

    def extract_html_job_links(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]:
        from bs4 import BeautifulSoup
        jobs = []
        if not html_text:
            return jobs

        soup = BeautifulSoup(html_text, "html.parser")
        anchors = soup.find_all("a", href=True)
        seen_links = set()

        non_job_slugs = [
            "alert", "search", "faq", "culture", "event", "privacy", "terms", "story",
            "benefit", "leadership", "community", "report", "sustainability", "map", "login"
        ]

        for a in anchors:
            href = a.get("href", "").strip()
            text = clean_html_text(a.get_text(strip=True))
            if not href or not text or len(text) < 6 or len(text) > 85:
                continue

            low_href = href.lower()
            low_text = text.lower()

            is_job_href = any(k in low_href for k in ["/job/", "/position/", "/opening/", "gh_jid=", "requisition"])
            if not is_job_href:
                continue

            if any(s in low_href for s in non_job_slugs) or any(s in low_text for s in non_job_slugs):
                continue

            full_url = href if href.startswith("http") else urljoin(page_url, href)
            if full_url in seen_links or full_url.rstrip("/") == page_url.rstrip("/"):
                continue
            seen_links.add(full_url)

            dept = determine_department(text, "")
            exp_level = determine_experience_level(text)
            is_remote = parse_is_us_remote("United States", text)

            jobs.append({
                "title": text,
                "company": default_company,
                "location": "Remote, USA" if is_remote else "United States",
                "is_remote": is_remote,
                "department": dept,
                "experience_level": exp_level,
                "description": f"Direct opening for {text} at {default_company}. Application details, requirements, and qualifications available at {full_url}",
                "salary_min": None,
                "salary_max": None,
                "job_url": full_url,
                "source": "web_direct"
            })
            if len(jobs) >= 50:
                break

        return jobs

    def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]:
        try:
            r = self._safe_get(url, timeout=5, allow_redirects=True)
            if r is None or r.status_code != 200:
                return []
            html_text = r.text
            # 1. Try Schema.org JSON-LD first (highest quality structured jobs)
            jobs = self.extract_schema_org_job_postings(html_text, r.url, default_company)
            if jobs:
                print(f"[Smart Crawler] Extracted {len(jobs)} Schema.org JobPosting records from {url}")
                return jobs
            # 2. Try HTML job links
            jobs = self.extract_html_job_links(html_text, r.url, default_company)
            if jobs:
                print(f"[Smart Crawler] Extracted {len(jobs)} HTML career links from {url}")
                return jobs
        except Exception as e:
            print(f"[Smart Crawler Warning] Scraping {url} failed: {e}")
        return []
