from __future__ import annotations import html import json import logging import math import re import threading import time from collections.abc import Iterator from dataclasses import dataclass from pathlib import Path from typing import Any from urllib.parse import quote, urljoin import requests from requests.adapters import HTTPAdapter from iaai_sync_service.runtime_config import RuntimeFilters from iaai_sync_service.settings import Settings logger = logging.getLogger(__name__) TRANSIENT_HTTP_CODES = {408, 425, 429, 500, 502, 503, 504} CHALLENGE_MARKERS = ( "_incapsula_resource", "incapsula", "incident id", "request unsuccessful", "access denied", ) COOKIE_ACCEPT_SELECTORS = ( "button:has-text('Accept All')", "button:has-text('Accept all')", "button:has-text('I Agree')", "button:has-text('Agree')", "button:has-text('Only necessary')", "button:has-text('Только необходимые')", "button:has-text('Принять все')", "[id*='accept']", "[class*='accept']", ) LISTING_MARKER = 'id="GBPSearchQuery"' DETAIL_MARKER = 'id="ProductDetailsVM"' RESIZER_URL = "https://vis.iaai.com/resizer" DEFAULT_USER_AGENT = ( "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " "AppleWebKit/537.36 (KHTML, like Gecko) " "Chrome/124.0.0.0 Safari/537.36" ) MAX_CHALLENGE_REFRESH_ATTEMPTS = 3 PLAYWRIGHT_REFRESH_POLLS = 8 @dataclass(frozen=True) class ListingVehicle: inventory_id: str tenant: str | None auction_id: str | None auction_date: str | None inventory_status: str | None currency: str | None timed_auction_closed: bool timed_auction_indicator: bool prebid_indicator: bool buynow_indicator: bool @dataclass(frozen=True) class ListingPage: vehicles: list[ListingVehicle] result_count: int page_size: int current_page: int gbp_search_query: dict[str, Any] class HybridSessionAuth: def __init__(self, settings: Settings) -> None: self._settings = settings self._thread_local = threading.local() self._lock = threading.Lock() self._refresh_lock = threading.Lock() self._bootstrap_cookies_loaded = False self._anonymous_bootstrap_attempted = False self._refresh_generation = 0 self._latest_refresh_cookies: list[dict[str, Any]] = [] def request( self, method: str, url: str, *, timeout: int, retries: int, retry_backoff_ms: int, headers: dict[str, str] | None = None, data: Any | None = None, json_body: Any | None = None, expected_marker: str | None = None, ) -> requests.Response: session = self._get_session() self._ensure_anonymous_session_bootstrap(session=session) self._sync_session_with_latest_refresh(session) last_error: Exception | None = None refresh_attempts = 0 attempt = 0 while attempt <= retries: try: response = session.request( method=method, url=url, headers=headers, data=data, json=json_body, timeout=timeout, ) except requests.RequestException as exc: last_error = exc if attempt >= retries: break self._sleep_backoff(retry_backoff_ms, attempt) attempt += 1 continue if response.status_code in TRANSIENT_HTTP_CODES and attempt < retries: response.close() self._sleep_backoff(retry_backoff_ms, attempt) attempt += 1 continue if is_challenge_response( status_code=response.status_code, body_text=response.text, expected_marker=expected_marker, ): response.close() if refresh_attempts >= MAX_CHALLENGE_REFRESH_ATTEMPTS: raise RuntimeError( "IAAI challenge persisted after " f"{refresh_attempts} Playwright refresh attempts for url={url}" ) refresh_attempts += 1 logger.warning( "Challenge detected for url=%s status=%s marker=%s refresh_attempt=%s/%s", url, response.status_code, expected_marker, refresh_attempts, MAX_CHALLENGE_REFRESH_ATTEMPTS, ) self._refresh_session_via_playwright(expected_marker=LISTING_MARKER, session=session) if refresh_attempts > 1: self._sleep_backoff(retry_backoff_ms, refresh_attempts - 1) continue return response if last_error is not None: raise RuntimeError(f"Request failed url={url}: {last_error}") from last_error raise RuntimeError(f"Request failed url={url} after retries") def persist_storage_state(self) -> None: session = self._get_session() with self._lock: self._save_storage_state(session) def _get_session(self) -> requests.Session: session = getattr(self._thread_local, "session", None) if session is None: session = requests.Session() adapter = HTTPAdapter(pool_connections=20, pool_maxsize=20) session.mount("http://", adapter) session.mount("https://", adapter) session.headers.update( { "user-agent": DEFAULT_USER_AGENT, "accept-language": "en-US,en;q=0.9", "cache-control": "no-cache", "pragma": "no-cache", } ) with self._lock: self._bootstrap_session_cookies(session) self._thread_local.session = session self._thread_local.session_generation = 0 self._sync_session_with_latest_refresh(session) return session def _sync_session_with_latest_refresh(self, session: requests.Session) -> None: with self._lock: latest_generation = self._refresh_generation session_generation = getattr(self._thread_local, "session_generation", 0) if latest_generation <= session_generation or not self._latest_refresh_cookies: return cookies = list(self._latest_refresh_cookies) self._apply_cookies_to_session(session, cookies) self._thread_local.session_generation = latest_generation def _ensure_anonymous_session_bootstrap(self, *, session: requests.Session) -> None: if self._anonymous_bootstrap_attempted: return if self._session_has_iaai_cookies(session): self._anonymous_bootstrap_attempted = True return with self._lock: if self._anonymous_bootstrap_attempted: return self._anonymous_bootstrap_attempted = True logger.info("No IAAI cookies preloaded. Attempting anonymous session bootstrap via Playwright.") try: self._refresh_session_via_playwright(expected_marker=LISTING_MARKER, session=session) except Exception as exc: # noqa: BLE001 logger.warning( "Anonymous session bootstrap via Playwright failed; continuing with direct HTTP flow: %s", exc, ) @staticmethod def _session_has_iaai_cookies(session: requests.Session) -> bool: cookies = getattr(session, "cookies", None) if cookies is None: return False for item in cookies: domain = "" if isinstance(item, dict): domain = parse_text(item.get("domain")) or "" else: domain = str(getattr(item, "domain", "") or "") if not domain: return True if "iaai.com" in domain.lower(): return True return False def _bootstrap_session_cookies(self, session: requests.Session) -> None: if self._bootstrap_cookies_loaded: return self._load_storage_state_cookies(session) self._load_env_cookies(session) self._bootstrap_cookies_loaded = True def _load_storage_state_cookies(self, session: requests.Session) -> None: path = self._settings.iaai_storage_state_path if not path.exists(): return try: payload = json.loads(path.read_text(encoding="utf-8")) except Exception as exc: # noqa: BLE001 logger.warning("Failed to read storage state file '%s': %s", path, exc) return cookies = payload.get("cookies") if isinstance(payload, dict) else None if not isinstance(cookies, list): return applied = 0 for item in cookies: if not isinstance(item, dict): continue name = parse_text(item.get("name")) value = parse_text(item.get("value")) if not name or value is None: continue domain = parse_text(item.get("domain")) or ".iaai.com" cookie_path = parse_text(item.get("path")) or "/" expires_raw = item.get("expires") expires = parse_int(expires_raw) session.cookies.set(name, value, domain=domain, path=cookie_path, expires=expires) applied += 1 if applied: logger.info("Loaded %s cookies from storage state", applied) def _load_env_cookies(self, session: requests.Session) -> None: raw = self._settings.iaai_session_cookies if not raw: return try: payload = json.loads(raw) except Exception: # noqa: BLE001 payload = None applied = 0 if isinstance(payload, list): for item in payload: if not isinstance(item, dict): continue name = parse_text(item.get("name")) value = parse_text(item.get("value")) if not name or value is None: continue domain = parse_text(item.get("domain")) or ".iaai.com" cookie_path = parse_text(item.get("path")) or "/" expires = parse_int(item.get("expires")) session.cookies.set(name, value, domain=domain, path=cookie_path, expires=expires) applied += 1 elif isinstance(payload, dict): for name, value in payload.items(): if not isinstance(name, str) or not isinstance(value, str): continue session.cookies.set(name.strip(), value.strip(), domain=".iaai.com", path="/") applied += 1 else: for part in raw.split(";"): if "=" not in part: continue name, value = part.split("=", 1) name = name.strip() value = value.strip() if not name: continue session.cookies.set(name, value, domain=".iaai.com", path="/") applied += 1 if applied: logger.info("Loaded %s cookies from IAAI_SESSION_COOKIES", applied) def _refresh_session_via_playwright( self, *, expected_marker: str | None = None, session: requests.Session | None = None, ) -> None: target_session = session or self._get_session() with self._lock: baseline_generation = self._refresh_generation with self._refresh_lock: with self._lock: if self._refresh_generation > baseline_generation and self._latest_refresh_cookies: self._apply_cookies_to_session(target_session, self._latest_refresh_cookies) self._thread_local.session_generation = self._refresh_generation return logger.warning("IAAI session challenge detected. Refreshing session via Playwright.") cookies = self._fetch_cookies_via_playwright(expected_marker=expected_marker) self._apply_cookies_to_session(target_session, cookies) with self._lock: self._refresh_generation += 1 self._latest_refresh_cookies = list(cookies) self._anonymous_bootstrap_attempted = True refreshed_generation = self._refresh_generation self._thread_local.session_generation = refreshed_generation self._save_storage_state(target_session) def _fetch_cookies_via_playwright(self, *, expected_marker: str | None = None) -> list[dict[str, Any]]: try: from playwright.sync_api import TimeoutError as PlaywrightTimeoutError from playwright.sync_api import sync_playwright except Exception as exc: # noqa: BLE001 raise RuntimeError( "Playwright is required for challenge fallback. Install dependency and browsers." ) from exc with sync_playwright() as playwright: browser = playwright.chromium.launch(headless=True) try: context = browser.new_context( locale="en-US", viewport={"width": 1366, "height": 768}, user_agent=DEFAULT_USER_AGENT, ) page = context.new_page() home_target = self._settings.iaai_base_url + "/" target = urljoin(self._settings.iaai_base_url + "/", "Vehiclelisting/Cars") timeout_ms = max(30_000, self._settings.http_timeout * 1000) page.goto(home_target, wait_until="domcontentloaded", timeout=timeout_ms) self._accept_cookie_banner(page) page.goto(target, wait_until="domcontentloaded", timeout=timeout_ms) self._accept_cookie_banner(page) self._try_login(page) self._wait_until_non_challenge( page=page, target=target, timeout_ms=timeout_ms, expected_marker=expected_marker or LISTING_MARKER, ) state = context.storage_state() except PlaywrightTimeoutError as exc: raise RuntimeError(f"Playwright refresh timed out: {exc}") from exc finally: browser.close() cookies = state.get("cookies") if isinstance(state, dict) else None if not isinstance(cookies, list) or not cookies: raise RuntimeError("Playwright refresh did not return cookies") return [cookie for cookie in cookies if isinstance(cookie, dict)] @staticmethod def _apply_cookies_to_session(session: requests.Session, cookies: list[dict[str, Any]]) -> None: session.cookies.clear() for cookie in cookies: name = parse_text(cookie.get("name")) value = parse_text(cookie.get("value")) if not name or value is None: continue domain = parse_text(cookie.get("domain")) or ".iaai.com" cookie_path = parse_text(cookie.get("path")) or "/" expires = parse_int(cookie.get("expires")) session.cookies.set(name, value, domain=domain, path=cookie_path, expires=expires) @staticmethod def _wait_until_non_challenge( *, page: Any, target: str, timeout_ms: int, expected_marker: str | None, ) -> None: # Give Incapsula redirect/challenge flow time to settle and verify page really opened. poll_ms = max(1000, min(5000, timeout_ms // PLAYWRIGHT_REFRESH_POLLS)) navigation_error_count = 0 for poll_index in range(PLAYWRIGHT_REFRESH_POLLS): try: page.wait_for_load_state("domcontentloaded", timeout=poll_ms) except Exception: # noqa: BLE001 # The challenge page frequently redirects; continue polling. pass page.wait_for_timeout(poll_ms) body: str | None = None for _ in range(3): try: body = page.content() break except Exception as exc: # noqa: BLE001 message = str(exc).lower() if "page.content" not in message or "navigating and changing the content" not in message: raise navigation_error_count += 1 page.wait_for_timeout(max(200, poll_ms // 4)) continue if body is not None and not is_challenge_response( status_code=200, body_text=body, expected_marker=expected_marker, ): return try: page.goto(target, wait_until="domcontentloaded", timeout=timeout_ms) except Exception as exc: # noqa: BLE001 logger.debug( "Playwright challenge retry navigation failed poll=%s/%s: %s", poll_index + 1, PLAYWRIGHT_REFRESH_POLLS, exc, ) raise RuntimeError( "Playwright refresh completed but challenge page is still active " f"(navigation_content_errors={navigation_error_count})" ) def _try_login(self, page: Any) -> None: if not self._settings.iaai_login or not self._settings.iaai_password: return try: login_selectors = ( "input[type='email']", "input[name*='email' i]", "input[id*='email' i]", "input[name*='username' i]", ) password_selector = "input[type='password']" submit_selector = "button[type='submit'],input[type='submit']" email_locator = None for selector in login_selectors: locator = page.locator(selector) if locator.count() > 0: email_locator = locator.first break if email_locator is None: return email_locator.fill(self._settings.iaai_login) password_locator = page.locator(password_selector) if password_locator.count() == 0: return password_locator.first.fill(self._settings.iaai_password) submit_locator = page.locator(submit_selector) if submit_locator.count() > 0: submit_locator.first.click() page.wait_for_load_state("domcontentloaded", timeout=30_000) except Exception as exc: # noqa: BLE001 logger.warning("Playwright login step failed: %s", exc) def _accept_cookie_banner(self, page: Any) -> None: for selector in COOKIE_ACCEPT_SELECTORS: try: locator = page.locator(selector).first if locator.count() == 0: continue if not locator.is_visible(timeout=500): continue locator.click(timeout=2_000) page.wait_for_timeout(250) return except Exception: # noqa: BLE001 continue def _save_storage_state(self, session: requests.Session) -> None: path = self._settings.iaai_storage_state_path cookies: list[dict[str, Any]] = [] for cookie in session.cookies: cookie_payload: dict[str, Any] = { "name": cookie.name, "value": cookie.value, "domain": cookie.domain or ".iaai.com", "path": cookie.path or "/", "httpOnly": False, "secure": bool(cookie.secure), "sameSite": "Lax", } if cookie.expires is not None: cookie_payload["expires"] = int(cookie.expires) cookies.append(cookie_payload) payload = {"cookies": cookies, "origins": []} path.parent.mkdir(parents=True, exist_ok=True) path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8") @staticmethod def _sleep_backoff(retry_backoff_ms: int, attempt: int) -> None: if retry_backoff_ms <= 0: return delay = retry_backoff_ms * (2**attempt) / 1000 time.sleep(delay) class IAAIClient: def __init__(self, settings: Settings) -> None: self._settings = settings self._auth = HybridSessionAuth(settings) def persist_session_state(self) -> None: self._auth.persist_storage_state() def keep_session_alive(self) -> dict[str, Any]: started = time.perf_counter() scope = resolve_listing_scope_paths( listing_start_url=self._settings.iaai_listing_start_url, brands=set(), )[0] page_html = self._fetch_listing_first_page(scope) marker_present = LISTING_MARKER in page_html result_count = parse_int(parse_hidden_input_value(page_html, "ResultCount")) self.persist_session_state() return { "scope": scope, "marker_present": marker_present, "result_count": result_count, "elapsed_ms": int((time.perf_counter() - started) * 1000), } def iter_listing_vehicles(self, filters: RuntimeFilters) -> Iterator[ListingVehicle]: seen_inventory_ids: set[str] = set() if self._settings.iaai_listing_start_url.strip() and filters.brands: logger.info( "Using explicit listing start URL; runtime brand scopes in listing step are ignored." ) scope_paths = resolve_listing_scope_paths( listing_start_url=self._settings.iaai_listing_start_url, brands=filters.brands, ) listing_started_at = time.perf_counter() for scope_index, scope_path in enumerate(scope_paths, start=1): logger.info( "IAAI listing scope start scope=%s/%s path=%s", scope_index, len(scope_paths), scope_path, ) first_page_html = self._fetch_listing_first_page(scope_path) first_page = parse_listing_page(first_page_html) for vehicle in first_page.vehicles: if vehicle.inventory_id in seen_inventory_ids: continue seen_inventory_ids.add(vehicle.inventory_id) yield vehicle page_size = max(1, first_page.page_size) total_pages = max(1, math.ceil(max(first_page.result_count, len(first_page.vehicles)) / page_size)) gbp_search_query = first_page.gbp_search_query logger.info( "IAAI listing scope loaded scope=%s/%s path=%s result_count=%s page_size=%s total_pages=%s unique_ids=%s elapsed=%.1fs", scope_index, len(scope_paths), scope_path, first_page.result_count, page_size, total_pages, len(seen_inventory_ids), time.perf_counter() - listing_started_at, ) for page_number in range(2, total_pages + 1): page_html = self._fetch_listing_page(scope_path, gbp_search_query, page_number, page_size) parsed_page = parse_listing_page(page_html) gbp_search_query = parsed_page.gbp_search_query for vehicle in parsed_page.vehicles: if vehicle.inventory_id in seen_inventory_ids: continue seen_inventory_ids.add(vehicle.inventory_id) yield vehicle if page_number % 10 == 0 or page_number == total_pages: logger.info( "IAAI listing page progress scope=%s/%s path=%s page=%s/%s unique_ids=%s elapsed=%.1fs", scope_index, len(scope_paths), scope_path, page_number, total_pages, len(seen_inventory_ids), time.perf_counter() - listing_started_at, ) def fetch_vehicle_detail_payload(self, inventory_id: str) -> dict[str, Any]: escaped_id = quote(inventory_id, safe="~") url = urljoin(self._settings.iaai_base_url + "/", f"VehicleDetail/{escaped_id}") response = self._auth.request( "GET", url, timeout=self._settings.http_timeout, retries=self._settings.http_retries, retry_backoff_ms=self._settings.http_retry_backoff_ms, headers={"accept": "text/html,application/xhtml+xml"}, expected_marker=DETAIL_MARKER, ) with response: if response.status_code >= 400: raise RuntimeError(f"Vehicle detail request failed id={inventory_id} status={response.status_code}") return parse_product_details_vm(response.text) def _fetch_listing_first_page(self, scope_path: str) -> str: if scope_path.lower().startswith(("http://", "https://")): url = scope_path else: url = urljoin(self._settings.iaai_base_url + "/", scope_path.lstrip("/")) response = self._auth.request( "GET", url, timeout=self._settings.http_timeout, retries=self._settings.http_retries, retry_backoff_ms=self._settings.http_retry_backoff_ms, headers={"accept": "text/html,application/xhtml+xml"}, expected_marker=LISTING_MARKER, ) with response: if response.status_code >= 400: raise RuntimeError(f"Listing request failed path={scope_path} status={response.status_code}") return response.text def _fetch_listing_page( self, scope_path: str, gbp_search_query: dict[str, Any], page_number: int, page_size: int, ) -> str: query_payload = dict(gbp_search_query) query_payload["CurrentPage"] = page_number query_payload["PageSize"] = page_size search_url = urljoin(self._settings.iaai_base_url + "/", "Search") common_headers = { "accept": "text/html,application/xhtml+xml,*/*", "x-requested-with": "XMLHttpRequest", } attempts: list[tuple[dict[str, str], Any, Any]] = [ ({**common_headers, "content-type": "application/json"}, None, query_payload), ({**common_headers, "content-type": "application/json"}, None, {"GBPSearchQuery": query_payload}), ({**common_headers}, {"GBPSearchQuery": json.dumps(query_payload, separators=(",", ":"))}, None), ( {**common_headers, "content-type": "application/json"}, json.dumps({"GBPSearchQuery": json.dumps(query_payload, separators=(",", ":"))}), None, ), ] last_error: Exception | None = None for headers, data, json_body in attempts: try: response = self._auth.request( "POST", search_url, timeout=self._settings.http_timeout, retries=self._settings.http_retries, retry_backoff_ms=self._settings.http_retry_backoff_ms, headers=headers, data=data, json_body=json_body, expected_marker=LISTING_MARKER, ) with response: if response.status_code >= 400: raise RuntimeError( f"Listing page request failed status={response.status_code} page={page_number}" ) body = response.text if LISTING_MARKER not in body: raise RuntimeError("Listing page response does not include GBPSearchQuery") return body except Exception as exc: # noqa: BLE001 last_error = exc continue if last_error is not None: raise RuntimeError(f"Failed to load listing page={page_number} for {scope_path}: {last_error}") from last_error raise RuntimeError(f"Failed to load listing page={page_number} for {scope_path}") def build_brand_scope_paths(brands: set[str]) -> list[str]: if not brands: return ["/Vehiclelisting/Cars"] paths: list[str] = [] for brand in sorted(brands): raw = brand.strip() if not raw: continue slug_hyphen = quote(raw.replace(" ", "-"), safe="-") slug_raw = quote(raw, safe="") for slug in (slug_hyphen, slug_raw): path = f"/Vehiclelisting/Cars/{slug}" if path not in paths: paths.append(path) return paths or ["/Vehiclelisting/Cars"] def resolve_listing_scope_paths(*, listing_start_url: str, brands: set[str]) -> list[str]: explicit_scope = listing_start_url.strip() if explicit_scope: return [explicit_scope] return build_brand_scope_paths(brands) def parse_listing_page(html_text: str) -> ListingPage: gbp_raw = parse_hidden_input_value(html_text, "GBPSearchQuery") vehicle_raw = parse_hidden_input_value(html_text, "VehicleDetails") result_count_raw = parse_hidden_input_value(html_text, "ResultCount") page_size_raw = parse_hidden_input_value(html_text, "PageSize") current_page_raw = parse_hidden_input_value(html_text, "CurrentPage") if not gbp_raw: raise RuntimeError("Listing page missing GBPSearchQuery") if vehicle_raw is None: raise RuntimeError("Listing page missing VehicleDetails") gbp_payload = json.loads(gbp_raw) if not isinstance(gbp_payload, dict): raise RuntimeError("GBPSearchQuery payload is not object") vehicle_payload = json.loads(vehicle_raw) if not isinstance(vehicle_payload, list): raise RuntimeError("VehicleDetails payload is not array") vehicles: list[ListingVehicle] = [] for item in vehicle_payload: if not isinstance(item, dict): continue inventory_id = parse_text(item.get("Id")) if not inventory_id: continue vehicles.append( ListingVehicle( inventory_id=inventory_id, tenant=parse_text(item.get("Tenant")), auction_id=parse_text(item.get("ActnLnId")), auction_date=parse_text(item.get("AuctionDate")) or parse_text(item.get("ActnDtTm")), inventory_status=parse_text(item.get("InventoryStatus")), currency=parse_text(item.get("Currency")), timed_auction_closed=parse_bool(item.get("TimedAuctionClosedIndicator")), timed_auction_indicator=parse_bool(item.get("TimedAuctionIndicator")), prebid_indicator=parse_bool(item.get("PreBidIndicator")), buynow_indicator=parse_bool(item.get("BuyNowIndicator")), ) ) return ListingPage( vehicles=vehicles, result_count=parse_int(result_count_raw) or len(vehicles), page_size=parse_int(page_size_raw) or max(1, len(vehicles)), current_page=parse_int(current_page_raw) or 1, gbp_search_query=gbp_payload, ) def parse_product_details_vm(html_text: str) -> dict[str, Any]: match = re.search( r"]*id=\"ProductDetailsVM\"[^>]*>\s*(\{.*?\})\s*", html_text, flags=re.DOTALL, ) if match is None: raise RuntimeError("ProductDetailsVM script not found") payload = json.loads(match.group(1)) if not isinstance(payload, dict): raise RuntimeError("ProductDetailsVM root is not object") return payload def parse_hidden_input_value(html_text: str, input_id: str) -> str | None: escaped_id = re.escape(input_id) patterns = ( rf"]*\bid=\"{escaped_id}\"[^>]*\bvalue=\"([^\"]*)\"", rf"]*\bid='{escaped_id}'[^>]*\bvalue='([^']*)'", ) for pattern in patterns: match = re.search(pattern, html_text, flags=re.IGNORECASE) if match is not None: return html.unescape(match.group(1)) return None def build_resizer_images_from_keys(image_keys: list[dict[str, Any]]) -> list[dict[str, str | int]]: seen_fullres: set[str] = set() images: list[dict[str, str | int]] = [] for index, item in enumerate(image_keys): if not isinstance(item, dict): continue key = parse_text(item.get("k")) if key is None: continue width = parse_int(item.get("w")) or 1600 height = parse_int(item.get("h")) or 1200 if width <= 0: width = 1600 if height <= 0: height = 1200 order_index = parse_int(item.get("i")) if order_index is None: order_index = parse_int(item.get("in")) if order_index is None: order_index = index preview_width = min(640, width) preview_height = max(1, int(round(height * (preview_width / width)))) escaped_key = quote(key, safe="~") fullres = f"{RESIZER_URL}?imageKeys={escaped_key}&width={width}&height={height}" preview = f"{RESIZER_URL}?imageKeys={escaped_key}&width={preview_width}&height={preview_height}" if fullres in seen_fullres: continue seen_fullres.add(fullres) images.append( { "order_index": order_index, "fullres_image": fullres, "preview_image": preview, } ) images.sort(key=lambda row: (parse_int(row.get("order_index")) or 0, str(row.get("fullres_image")))) return images def is_challenge_response(*, status_code: int, body_text: str, expected_marker: str | None = None) -> bool: if status_code in {401, 403}: return True if _expected_marker_present(body_text=body_text, expected_marker=expected_marker): # If expected listing/detail marker is present, this is a valid page even if # Incapsula script references are embedded in the HTML. return False lowered = (body_text or "").lower() if any(marker in lowered for marker in CHALLENGE_MARKERS): return True if expected_marker and not _expected_marker_present(body_text=body_text, expected_marker=expected_marker): # Expected hidden marker/script missing from HTML often means anti-bot interstitial. if " bool: if not expected_marker: return False if expected_marker in body_text: return True # Accept quote variants for marker fragments like id="GBPSearchQuery" / id='GBPSearchQuery'. if '"' in expected_marker: single_quoted = expected_marker.replace('"', "'") if single_quoted in body_text: return True if "'" in expected_marker: double_quoted = expected_marker.replace("'", '"') if double_quoted in body_text: return True marker_match = re.search(r"id=['\"]([^'\"]+)['\"]", expected_marker) if marker_match is None: return False marker_id = re.escape(marker_match.group(1)) return bool( re.search( rf"id\s*=\s*['\"]{marker_id}['\"]", body_text, flags=re.IGNORECASE, ) ) def parse_text(value: Any) -> str | None: if isinstance(value, str): text = value.strip() return text if text else None return None def parse_bool(value: Any) -> bool: if isinstance(value, bool): return value if isinstance(value, str): normalized = value.strip().lower() return normalized in {"true", "1", "yes", "on"} if isinstance(value, (int, float)) and not isinstance(value, bool): return value != 0 return False def parse_int(value: Any) -> int | None: if value is None or isinstance(value, bool): return None if isinstance(value, int): return value if isinstance(value, float): return int(round(value)) if isinstance(value, str): text = value.strip() if not text: return None normalized = text.replace(",", "").replace(" ", "") try: return int(round(float(normalized))) except ValueError: return None return None