import html
import json
import re
import time
import logging
import threading
import random
from curl_cffi.requests import Session as CurlSession
from urllib.parse import quote, quote_plus

from .models import SearchResponse
from .utils import current_epoch_ms

USER_AGENTS = [
    {
        "ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/138.0.0.0 Safari/537.36",
        "sec_ch_ua": '"Chromium";v="138", "Not/A)Brand";v="24", "Google Chrome";v="138"',
        "platform": '"Windows"',
    },
    {
        "ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/137.0.0.0 Safari/537.36 Edg/137.0.0.0",
        "sec_ch_ua": '"Chromium";v="137", "Not/A)Brand";v="24", "Microsoft Edge";v="137"',
        "platform": '"Windows"',
    },
    {
        "ua": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/138.0.0.0 Safari/537.36",
        "sec_ch_ua": '"Chromium";v="138", "Not/A)Brand";v="24", "Google Chrome";v="138"',
        "platform": '"macOS"',
    },
    {
        "ua": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/18.5 Safari/605.1.15",
        "sec_ch_ua": '"Not/A)Brand";v="24", "Safari";v="18"',
        "platform": '"macOS"',
    },
    {
        "ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:140.0) Gecko/20100101 Firefox/140.0",
        "sec_ch_ua": "",
        "platform": '"Windows"',
    },
    {
        "ua": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10.15; rv:140.0) Gecko/20100101 Firefox/140.0",
        "sec_ch_ua": "",
        "platform": '"macOS"',
    },
    {
        "ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/136.0.0.0 Safari/537.36 OPR/120.0.0.0",
        "sec_ch_ua": '"Chromium";v="136", "Not/A)Brand";v="24", "Opera";v="120"',
        "platform": '"Windows"',
    },
    {
        "ua": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/138.0.0.0 Safari/537.36",
        "sec_ch_ua": '"Chromium";v="138", "Not/A)Brand";v="24", "Google Chrome";v="138"',
        "platform": '"Linux"',
    },
    {
        "ua": "Mozilla/5.0 (X11; Linux x86_64; rv:140.0) Gecko/20100101 Firefox/140.0",
        "sec_ch_ua": "",
        "platform": '"Linux"',
    },
    {
        "ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/135.0.0.0 Safari/537.36 Edg/135.0.0.0",
        "sec_ch_ua": '"Chromium";v="135", "Not/A)Brand";v="24", "Microsoft Edge";v="135"',
        "platform": '"Windows"',
    },
    {
        "ua": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/136.0.0.0 Safari/537.36 Edg/136.0.0.0",
        "sec_ch_ua": '"Chromium";v="136", "Not/A)Brand";v="24", "Microsoft Edge";v="136"',
        "platform": '"macOS"',
    },
    {
        "ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/138.0.0.0 Safari/537.36 Vivaldi/7.4",
        "sec_ch_ua": '"Chromium";v="138", "Not/A)Brand";v="24", "Vivaldi";v="7"',
        "platform": '"Windows"',
    },
]

IMPERSONATE_PROFILES = ["chrome", "chrome110", "chrome116", "chrome120", "safari15_5", "edge99"]

_proxy_lock = threading.Lock()
_shared_proxy_index = 0


def _next_proxy_url(proxy_list: list):
    global _shared_proxy_index
    if not proxy_list:
        return None
    with _proxy_lock:
        proxy_url = proxy_list[_shared_proxy_index % len(proxy_list)]
        _shared_proxy_index += 1
    return proxy_url


class Pinterest:
    BASE_URL = "https://in.pinterest.com"

    def __init__(self, user_agent: str = "", proxies: dict = None, proxy_list: list = None,
                 sleep_time=None, max_retries: int = 3,
                 warmup: bool = True, max_requests_per_session: int = 50):
        self.errors = []
        self.max_retries = max_retries
        self.sleep_time = sleep_time if sleep_time is not None else (1.5, 4.0)
        self.proxies = proxies or {}
        self.proxy_list = proxy_list or []
        self._fixed_ua = bool(user_agent)

        self.user_agent = user_agent or USER_AGENTS[0]["ua"]

        self.warmup = warmup
        self.max_requests_per_session = max(1, int(max_requests_per_session))
        self._request_count = 0

        self.BASE_HEADERS = {
            'Host': self.BASE_URL.replace('https://', ''),
            'Sec-Ch-Ua-Platform': '"Windows"',
            'Sec-Ch-Ua': '"Chromium";v="137", "Not/A)Brand";v="24"',
            'Sec-Ch-Ua-Model': '""',
            'Sec-Ch-Ua-Mobile': '?0',
            'X-Requested-With': 'XMLHttpRequest',
            'Accept': 'application/json, text/javascript, */*, q=0.01',
            'X-Pinterest-Source-Url': '',
            'X-Pinterest-Appstate': 'active',
            'Accept-Language': 'en-US,en;q=0.9',
            'Screen-Dpr': '1',
            'X-Pinterest-Pws-Handler': 'www/search/[scope].js',
            'User-Agent': self.user_agent,
            'Sec-Ch-Ua-Platform-Version': '""',
            'Sec-Fetch-Site': 'same-origin',
            'Sec-Fetch-Mode': 'cors',
            'Sec-Fetch-Dest': 'empty',
            'Referer': f'{self.BASE_URL}/',
            'Priority': 'u=1, i',
        }

        self._init_session()

    # -----------------------------
    # Session management (curl_cffi)
    # One session per worker, one pinned proxy per session.
    # Reset every max_requests_per_session requests and on 403/429 so the
    # next session picks the next proxy from the pool (new datacenter IP).
    # -----------------------------

    def _init_session(self):
        profile = random.choice(IMPERSONATE_PROFILES)
        self.session = CurlSession(impersonate=profile)
        self._current_profile = profile
        self._request_count = 0
        self._pinned_proxy = self._next_proxy_from_pool()

    def reset_session(self):
        try:
            self.session.close()
        except Exception:
            pass
        self._init_session()

    # -----------------------------
    # Proxy rotation
    # -----------------------------

    def _next_proxy_from_pool(self) -> dict:
        if not self.proxy_list:
            return self.proxies
        proxy_url = _next_proxy_url(self.proxy_list)
        return {"http": proxy_url, "https": proxy_url}

    def _get_next_proxy(self) -> dict:
        return self._pinned_proxy

    # -----------------------------
    # User-Agent rotation
    # -----------------------------

    def _rotate_headers(self) -> dict:
        headers = self.BASE_HEADERS.copy()
        if not self._fixed_ua:
            profile = random.choice(USER_AGENTS)
            headers["User-Agent"] = profile["ua"]
            headers["Sec-Ch-Ua-Platform"] = profile["platform"]
            if profile["sec_ch_ua"]:
                headers["Sec-Ch-Ua"] = profile["sec_ch_ua"]
            else:
                headers.pop("Sec-Ch-Ua", None)
        return headers

    # -----------------------------
    # Sleep with jitter
    # -----------------------------

    def _jitter_sleep(self):
        if isinstance(self.sleep_time, (tuple, list)):
            min_s, max_s = self.sleep_time
            time.sleep(random.uniform(min_s, max_s))
        elif self.sleep_time:
            time.sleep(self.sleep_time + random.uniform(0, 1.5))

    # -----------------------------
    # Request with exponential backoff
    # -----------------------------

    def _request_with_backoff(self, url: str, headers: dict):
        for attempt in range(self.max_retries):
            try:
                response = self.session.get(url, headers=headers, proxies=self._pinned_proxy, timeout=15)
            except Exception as e:
                if attempt < self.max_retries - 1:
                    wait = (2 ** attempt) + random.uniform(1, 3)
                    logging.warning(f"Request error: {e}, retrying in {wait:.1f}s ({attempt+1}/{self.max_retries})")
                    time.sleep(wait)
                    self.reset_session()
                    headers = self._rotate_headers()
                    continue
                raise

            self._request_count += 1
            if self._request_count >= self.max_requests_per_session:
                logging.debug(f"Session limit ({self.max_requests_per_session}) reached, rotating session/proxy")
                self.reset_session()

            if response.status_code in (429, 403):
                if attempt < self.max_retries - 1:
                    wait = (2 ** attempt) + random.uniform(1, 5)
                    logging.warning(
                        f"Rate limited ({response.status_code}), backing off {wait:.1f}s "
                        f"({attempt+1}/{self.max_retries})"
                    )
                    time.sleep(wait)
                    self.reset_session()
                    headers = self._rotate_headers()
                    continue
                return response

            return response

        return None

    # -----------------------------
    # Pinterest API calls
    # -----------------------------

    def _do_warmup(self, query: str):
        source_url = f"/search/pins/?q={quote(query)}&rs=typed"
        try:
            self._request_with_backoff(f"{self.BASE_URL}{source_url}", self._rotate_headers())
        except Exception:
            pass

    def _call_search_results(self, query: str, page_size: int):
        source_url = f"/search/pins/?q={quote(query)}&rs=typed"

        payload = {
            "options": {
                "applied_unified_filters": None,
                "appliedProductFilters": "---",
                "article": None,
                "auto_correction_disabled": False,
                "corpus": None,
                "customized_rerank_type": None,
                "domains": None,
                "filters": None,
                "journey_depth": None,
                "page_size": f"{page_size}",
                "price_max": None,
                "price_min": None,
                "query_pin_sigs": None,
                "query": quote(query),
                "redux_normalize_feed": True,
                "request_params": None,
                "rs": "typed",
                "scope": "pins",
                "selected_one_bar_modules": None,
                "source_id": None,
                "source_module_id": None,
                "seoDrawerEnabled": False,
                "source_url": quote_plus(source_url),
                "top_pin_id": None,
                "top_pin_ids": None
            },
            "context": {}
        }

        encoded = quote_plus(json.dumps(payload).replace(" ", ""))
        encoded = (
            encoded.replace("%2520", "%20")
            .replace("%252F", "%2F")
            .replace("%253F", "%3F")
            .replace("%252520", "%2520")
            .replace("%253D", "%3D")
            .replace("%2526", "%26")
        )

        epoch = int(time.time() * 1000)
        url = (
            f"{self.BASE_URL}/resource/BaseSearchResource/get/"
            f"?source_url={quote_plus(source_url)}&data={encoded}&_={epoch}"
        )

        headers = self._rotate_headers()
        headers["X-Pinterest-Source-Url"] = source_url

        try:
            response = self._request_with_backoff(url, headers)
        except Exception as e:
            logging.error(f"Request failed for '{query}': {e}")
            self.errors.append(str(e))
            return None

        if response is None:
            logging.error(f"All retries exhausted for '{query}'")
            return None

        self._jitter_sleep()

        if response.status_code != 200:
            logging.warning(f"Search failed for '{query}': {response.status_code}")
            return None

        try:
            data = SearchResponse(**response.json())
        except Exception as e:
            logging.error(f"Failed to parse response for '{query}': {e}")
            self.errors.append(str(e))
            return None

        if not data.resource_response.data:
            return []

        return data.resource_response.data.results

    def _search_raw(self, query: str, page_size: int):
        if self.warmup:
            self._do_warmup(query)

        results = self._call_search_results(query, page_size)
        if results:
            return results

        if not self.warmup:
            logging.info(f"Empty/failed response for '{query}', warmup fallback + retry")
            self._do_warmup(query)
            results = self._call_search_results(query, page_size)

        return results or []

    def search_combined(self, query: str, page_size: int = 26) -> dict:
        results = self._search_raw(query, page_size)
        images = []
        descriptions = []
        for r in results:
            try:
                desc = r.description or ""
                if not isinstance(desc, str):
                    desc = ""
                if desc.strip():
                    descriptions.append(desc.strip())
            except Exception:
                desc = ""

            try:
                if r.images and "orig" in r.images:
                    title = r.title or r.grid_title or r.seo_alt_text or query
                    if isinstance(title, dict):
                        title = title.get("text") or title.get("format") or query
                    if not isinstance(title, str):
                        title = query
                    images.append({
                        "title": title,
                        "image_url": str(r.images["orig"].url),
                        "description": desc,
                    })
            except Exception:
                continue

        return {"images": images, "descriptions": descriptions}

    def search(self, query: str, page_size: int = 26) -> list:
        return self.search_combined(query, page_size)["images"]

    def search_descriptions(self, query: str, page_size: int = 26) -> list:
        return self.search_combined(query, page_size)["descriptions"]


def _clean_suggestion(text) -> str:
    if not isinstance(text, str):
        return ""
    text = re.sub(r'<[^>]+>', '', text)
    text = html.unescape(text)
    return text.strip()


def _next_proxy_dict(proxy_list: list):
    if not proxy_list:
        return None
    proxy_url = _next_proxy_url(proxy_list)
    return {"http": proxy_url, "https": proxy_url}


def _parse_ddg_suggestions(raw: str, max_results: int) -> list:
    if raw.startswith(")]}'"):
        raw = raw[4:].lstrip("\n")
    try:
        top = json.loads(raw)
    except Exception:
        return []

    items = []
    if isinstance(top, list):
        if len(top) >= 2 and isinstance(top[0], str) and isinstance(top[1], list):
            items = top[1]
        elif top and isinstance(top[0], dict):
            items = top
        elif top and isinstance(top[0], list):
            items = top[0]
        else:
            items = top
    elif isinstance(top, dict):
        phrase = top.get("phrase")
        if isinstance(phrase, list):
            items = phrase

    results = []
    for item in items:
        if isinstance(item, dict):
            text = item.get("phrase") or item.get("text") or ""
        else:
            text = item
        text = _clean_suggestion(text)
        if text:
            results.append(text)
        if len(results) >= max_results:
            break
    return results


def fetch_related_keywords_ddg(keyword: str, proxy_list: list = None, max_results: int = 10, kl: str = "wt-wt") -> list:
    session = CurlSession(impersonate=random.choice(IMPERSONATE_PROFILES))
    proxy = _next_proxy_dict(proxy_list)

    endpoint = "https://duckduckgo.com/ac/"
    params = {"q": keyword, "kl": kl}

    try:
        resp = session.get(endpoint, params=params, proxies=proxy, timeout=15)
    except Exception as e:
        logging.warning(f"DuckDuckGo related kw failed for '{keyword}': {e}")
        return []
    finally:
        try:
            session.close()
        except Exception:
            pass

    if resp.status_code != 200:
        logging.warning(f"DuckDuckGo related kw HTTP {resp.status_code} for '{keyword}'")
        return []

    return _parse_ddg_suggestions(resp.text, max_results)


def fetch_related_keywords_google(keyword: str, proxy_list: list = None, max_results: int = 10, hl: str = "en", gl: str = "US") -> list:
    session = CurlSession(impersonate=random.choice(IMPERSONATE_PROFILES))
    proxy = _next_proxy_dict(proxy_list)

    endpoint = "https://www.google.com/complete/search"
    params = {
        "cp": "1",
        "client": "gws-wiz",
        "xssi": "t",
        "gs_pcrt": "undefined",
        "hl": hl,
        "gl": gl,
        "authuser": "0",
        "dpr": "1",
        "q": keyword,
    }

    try:
        resp = session.get(endpoint, params=params, proxies=proxy, timeout=15)
    except Exception as e:
        logging.warning(f"Google related kw failed for '{keyword}': {e}")
        return []
    finally:
        try:
            session.close()
        except Exception:
            pass

    if resp.status_code != 200:
        return []

    raw = resp.text
    if raw.startswith(")]}'"):
        raw = raw[4:].lstrip("\n")

    try:
        top = json.loads(raw)
    except Exception:
        return []

    if not isinstance(top, list) or len(top) == 0:
        return []

    raw_suggestions = top[0] if isinstance(top[0], list) else []
    results = []
    for item in raw_suggestions:
        if not isinstance(item, list) or len(item) == 0:
            continue
        text, _ = item[0], item[1:]
        text = _clean_suggestion(text)
        if text:
            results.append(text)
        if len(results) >= max_results:
            break

    return results


def fetch_related_keywords(keyword: str, proxy_list: list = None, max_results: int = 10, hl: str = "en", gl: str = "US") -> list:
    kl = f"{gl}-{hl}".strip().lower() or "wt-wt"
    results = fetch_related_keywords_ddg(keyword, proxy_list=proxy_list, max_results=max_results, kl=kl)
    if results:
        return results
    logging.info(f"DuckDuckGo related_kw empty/failed for '{keyword}', fallback to Google Suggest")
    return fetch_related_keywords_google(keyword, proxy_list=proxy_list, max_results=max_results, hl=hl, gl=gl)


if __name__ == "__main__":
    keyword = "loki"
    p = Pinterest(sleep_time=(1.5, 3.0))
    print(f"\nSearching for: {keyword}")
    results = p.search(keyword, 5)
    print(f"Found {len(results)} images")
    for r in results:
        print(f"  - {r['title']}: {r['image_url'][:60]}...")
