from __future__ import annotations
import html
import json
import logging
import math
import re
import threading
import time
from collections.abc import Iterator
from dataclasses import dataclass
from pathlib import Path
from typing import Any
from urllib.parse import quote, urljoin
import requests
from requests.adapters import HTTPAdapter
from iaai_sync_service.runtime_config import RuntimeFilters
from iaai_sync_service.settings import Settings
logger = logging.getLogger(__name__)
TRANSIENT_HTTP_CODES = {408, 425, 429, 500, 502, 503, 504}
CHALLENGE_MARKERS = (
"_incapsula_resource",
"incapsula",
"incident id",
"request unsuccessful",
"access denied",
)
COOKIE_ACCEPT_SELECTORS = (
"button:has-text('Accept All')",
"button:has-text('Accept all')",
"button:has-text('I Agree')",
"button:has-text('Agree')",
"button:has-text('Only necessary')",
"button:has-text('Только необходимые')",
"button:has-text('Принять все')",
"[id*='accept']",
"[class*='accept']",
)
LISTING_MARKER = 'id="GBPSearchQuery"'
DETAIL_MARKER = 'id="ProductDetailsVM"'
RESIZER_URL = "https://vis.iaai.com/resizer"
DEFAULT_USER_AGENT = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/124.0.0.0 Safari/537.36"
)
MAX_CHALLENGE_REFRESH_ATTEMPTS = 3
PLAYWRIGHT_REFRESH_POLLS = 8
@dataclass(frozen=True)
class ListingVehicle:
inventory_id: str
tenant: str | None
auction_id: str | None
auction_date: str | None
inventory_status: str | None
currency: str | None
timed_auction_closed: bool
timed_auction_indicator: bool
prebid_indicator: bool
buynow_indicator: bool
@dataclass(frozen=True)
class ListingPage:
vehicles: list[ListingVehicle]
result_count: int
page_size: int
current_page: int
gbp_search_query: dict[str, Any]
class HybridSessionAuth:
def __init__(self, settings: Settings) -> None:
self._settings = settings
self._thread_local = threading.local()
self._lock = threading.Lock()
self._refresh_lock = threading.Lock()
self._bootstrap_cookies_loaded = False
self._anonymous_bootstrap_attempted = False
self._refresh_generation = 0
self._latest_refresh_cookies: list[dict[str, Any]] = []
def request(
self,
method: str,
url: str,
*,
timeout: int,
retries: int,
retry_backoff_ms: int,
headers: dict[str, str] | None = None,
data: Any | None = None,
json_body: Any | None = None,
expected_marker: str | None = None,
) -> requests.Response:
session = self._get_session()
self._ensure_anonymous_session_bootstrap(session=session)
self._sync_session_with_latest_refresh(session)
last_error: Exception | None = None
refresh_attempts = 0
attempt = 0
while attempt <= retries:
try:
response = session.request(
method=method,
url=url,
headers=headers,
data=data,
json=json_body,
timeout=timeout,
)
except requests.RequestException as exc:
last_error = exc
if attempt >= retries:
break
self._sleep_backoff(retry_backoff_ms, attempt)
attempt += 1
continue
if response.status_code in TRANSIENT_HTTP_CODES and attempt < retries:
response.close()
self._sleep_backoff(retry_backoff_ms, attempt)
attempt += 1
continue
if is_challenge_response(
status_code=response.status_code,
body_text=response.text,
expected_marker=expected_marker,
):
response.close()
if refresh_attempts >= MAX_CHALLENGE_REFRESH_ATTEMPTS:
raise RuntimeError(
"IAAI challenge persisted after "
f"{refresh_attempts} Playwright refresh attempts for url={url}"
)
refresh_attempts += 1
logger.warning(
"Challenge detected for url=%s status=%s marker=%s refresh_attempt=%s/%s",
url,
response.status_code,
expected_marker,
refresh_attempts,
MAX_CHALLENGE_REFRESH_ATTEMPTS,
)
self._refresh_session_via_playwright(expected_marker=LISTING_MARKER, session=session)
if refresh_attempts > 1:
self._sleep_backoff(retry_backoff_ms, refresh_attempts - 1)
continue
return response
if last_error is not None:
raise RuntimeError(f"Request failed url={url}: {last_error}") from last_error
raise RuntimeError(f"Request failed url={url} after retries")
def persist_storage_state(self) -> None:
session = self._get_session()
with self._lock:
self._save_storage_state(session)
def _get_session(self) -> requests.Session:
session = getattr(self._thread_local, "session", None)
if session is None:
session = requests.Session()
adapter = HTTPAdapter(pool_connections=20, pool_maxsize=20)
session.mount("http://", adapter)
session.mount("https://", adapter)
session.headers.update(
{
"user-agent": DEFAULT_USER_AGENT,
"accept-language": "en-US,en;q=0.9",
"cache-control": "no-cache",
"pragma": "no-cache",
}
)
with self._lock:
self._bootstrap_session_cookies(session)
self._thread_local.session = session
self._thread_local.session_generation = 0
self._sync_session_with_latest_refresh(session)
return session
def _sync_session_with_latest_refresh(self, session: requests.Session) -> None:
with self._lock:
latest_generation = self._refresh_generation
session_generation = getattr(self._thread_local, "session_generation", 0)
if latest_generation <= session_generation or not self._latest_refresh_cookies:
return
cookies = list(self._latest_refresh_cookies)
self._apply_cookies_to_session(session, cookies)
self._thread_local.session_generation = latest_generation
def _ensure_anonymous_session_bootstrap(self, *, session: requests.Session) -> None:
if self._anonymous_bootstrap_attempted:
return
if self._session_has_iaai_cookies(session):
self._anonymous_bootstrap_attempted = True
return
with self._lock:
if self._anonymous_bootstrap_attempted:
return
self._anonymous_bootstrap_attempted = True
logger.info("No IAAI cookies preloaded. Attempting anonymous session bootstrap via Playwright.")
try:
self._refresh_session_via_playwright(expected_marker=LISTING_MARKER, session=session)
except Exception as exc: # noqa: BLE001
logger.warning(
"Anonymous session bootstrap via Playwright failed; continuing with direct HTTP flow: %s",
exc,
)
@staticmethod
def _session_has_iaai_cookies(session: requests.Session) -> bool:
cookies = getattr(session, "cookies", None)
if cookies is None:
return False
for item in cookies:
domain = ""
if isinstance(item, dict):
domain = parse_text(item.get("domain")) or ""
else:
domain = str(getattr(item, "domain", "") or "")
if not domain:
return True
if "iaai.com" in domain.lower():
return True
return False
def _bootstrap_session_cookies(self, session: requests.Session) -> None:
if self._bootstrap_cookies_loaded:
return
self._load_storage_state_cookies(session)
self._load_env_cookies(session)
self._bootstrap_cookies_loaded = True
def _load_storage_state_cookies(self, session: requests.Session) -> None:
path = self._settings.iaai_storage_state_path
if not path.exists():
return
try:
payload = json.loads(path.read_text(encoding="utf-8"))
except Exception as exc: # noqa: BLE001
logger.warning("Failed to read storage state file '%s': %s", path, exc)
return
cookies = payload.get("cookies") if isinstance(payload, dict) else None
if not isinstance(cookies, list):
return
applied = 0
for item in cookies:
if not isinstance(item, dict):
continue
name = parse_text(item.get("name"))
value = parse_text(item.get("value"))
if not name or value is None:
continue
domain = parse_text(item.get("domain")) or ".iaai.com"
cookie_path = parse_text(item.get("path")) or "/"
expires_raw = item.get("expires")
expires = parse_int(expires_raw)
session.cookies.set(name, value, domain=domain, path=cookie_path, expires=expires)
applied += 1
if applied:
logger.info("Loaded %s cookies from storage state", applied)
def _load_env_cookies(self, session: requests.Session) -> None:
raw = self._settings.iaai_session_cookies
if not raw:
return
try:
payload = json.loads(raw)
except Exception: # noqa: BLE001
payload = None
applied = 0
if isinstance(payload, list):
for item in payload:
if not isinstance(item, dict):
continue
name = parse_text(item.get("name"))
value = parse_text(item.get("value"))
if not name or value is None:
continue
domain = parse_text(item.get("domain")) or ".iaai.com"
cookie_path = parse_text(item.get("path")) or "/"
expires = parse_int(item.get("expires"))
session.cookies.set(name, value, domain=domain, path=cookie_path, expires=expires)
applied += 1
elif isinstance(payload, dict):
for name, value in payload.items():
if not isinstance(name, str) or not isinstance(value, str):
continue
session.cookies.set(name.strip(), value.strip(), domain=".iaai.com", path="/")
applied += 1
else:
for part in raw.split(";"):
if "=" not in part:
continue
name, value = part.split("=", 1)
name = name.strip()
value = value.strip()
if not name:
continue
session.cookies.set(name, value, domain=".iaai.com", path="/")
applied += 1
if applied:
logger.info("Loaded %s cookies from IAAI_SESSION_COOKIES", applied)
def _refresh_session_via_playwright(
self,
*,
expected_marker: str | None = None,
session: requests.Session | None = None,
) -> None:
target_session = session or self._get_session()
with self._lock:
baseline_generation = self._refresh_generation
with self._refresh_lock:
with self._lock:
if self._refresh_generation > baseline_generation and self._latest_refresh_cookies:
self._apply_cookies_to_session(target_session, self._latest_refresh_cookies)
self._thread_local.session_generation = self._refresh_generation
return
logger.warning("IAAI session challenge detected. Refreshing session via Playwright.")
cookies = self._fetch_cookies_via_playwright(expected_marker=expected_marker)
self._apply_cookies_to_session(target_session, cookies)
with self._lock:
self._refresh_generation += 1
self._latest_refresh_cookies = list(cookies)
self._anonymous_bootstrap_attempted = True
refreshed_generation = self._refresh_generation
self._thread_local.session_generation = refreshed_generation
self._save_storage_state(target_session)
def _fetch_cookies_via_playwright(self, *, expected_marker: str | None = None) -> list[dict[str, Any]]:
try:
from playwright.sync_api import TimeoutError as PlaywrightTimeoutError
from playwright.sync_api import sync_playwright
except Exception as exc: # noqa: BLE001
raise RuntimeError(
"Playwright is required for challenge fallback. Install dependency and browsers."
) from exc
with sync_playwright() as playwright:
browser = playwright.chromium.launch(headless=True)
try:
context = browser.new_context(
locale="en-US",
viewport={"width": 1366, "height": 768},
user_agent=DEFAULT_USER_AGENT,
)
page = context.new_page()
home_target = self._settings.iaai_base_url + "/"
target = urljoin(self._settings.iaai_base_url + "/", "Vehiclelisting/Cars")
timeout_ms = max(30_000, self._settings.http_timeout * 1000)
page.goto(home_target, wait_until="domcontentloaded", timeout=timeout_ms)
self._accept_cookie_banner(page)
page.goto(target, wait_until="domcontentloaded", timeout=timeout_ms)
self._accept_cookie_banner(page)
self._try_login(page)
self._wait_until_non_challenge(
page=page,
target=target,
timeout_ms=timeout_ms,
expected_marker=expected_marker or LISTING_MARKER,
)
state = context.storage_state()
except PlaywrightTimeoutError as exc:
raise RuntimeError(f"Playwright refresh timed out: {exc}") from exc
finally:
browser.close()
cookies = state.get("cookies") if isinstance(state, dict) else None
if not isinstance(cookies, list) or not cookies:
raise RuntimeError("Playwright refresh did not return cookies")
return [cookie for cookie in cookies if isinstance(cookie, dict)]
@staticmethod
def _apply_cookies_to_session(session: requests.Session, cookies: list[dict[str, Any]]) -> None:
session.cookies.clear()
for cookie in cookies:
name = parse_text(cookie.get("name"))
value = parse_text(cookie.get("value"))
if not name or value is None:
continue
domain = parse_text(cookie.get("domain")) or ".iaai.com"
cookie_path = parse_text(cookie.get("path")) or "/"
expires = parse_int(cookie.get("expires"))
session.cookies.set(name, value, domain=domain, path=cookie_path, expires=expires)
@staticmethod
def _wait_until_non_challenge(
*,
page: Any,
target: str,
timeout_ms: int,
expected_marker: str | None,
) -> None:
# Give Incapsula redirect/challenge flow time to settle and verify page really opened.
poll_ms = max(1000, min(5000, timeout_ms // PLAYWRIGHT_REFRESH_POLLS))
navigation_error_count = 0
for poll_index in range(PLAYWRIGHT_REFRESH_POLLS):
try:
page.wait_for_load_state("domcontentloaded", timeout=poll_ms)
except Exception: # noqa: BLE001
# The challenge page frequently redirects; continue polling.
pass
page.wait_for_timeout(poll_ms)
body: str | None = None
for _ in range(3):
try:
body = page.content()
break
except Exception as exc: # noqa: BLE001
message = str(exc).lower()
if "page.content" not in message or "navigating and changing the content" not in message:
raise
navigation_error_count += 1
page.wait_for_timeout(max(200, poll_ms // 4))
continue
if body is not None and not is_challenge_response(
status_code=200,
body_text=body,
expected_marker=expected_marker,
):
return
try:
page.goto(target, wait_until="domcontentloaded", timeout=timeout_ms)
except Exception as exc: # noqa: BLE001
logger.debug(
"Playwright challenge retry navigation failed poll=%s/%s: %s",
poll_index + 1,
PLAYWRIGHT_REFRESH_POLLS,
exc,
)
raise RuntimeError(
"Playwright refresh completed but challenge page is still active "
f"(navigation_content_errors={navigation_error_count})"
)
def _try_login(self, page: Any) -> None:
if not self._settings.iaai_login or not self._settings.iaai_password:
return
try:
login_selectors = (
"input[type='email']",
"input[name*='email' i]",
"input[id*='email' i]",
"input[name*='username' i]",
)
password_selector = "input[type='password']"
submit_selector = "button[type='submit'],input[type='submit']"
email_locator = None
for selector in login_selectors:
locator = page.locator(selector)
if locator.count() > 0:
email_locator = locator.first
break
if email_locator is None:
return
email_locator.fill(self._settings.iaai_login)
password_locator = page.locator(password_selector)
if password_locator.count() == 0:
return
password_locator.first.fill(self._settings.iaai_password)
submit_locator = page.locator(submit_selector)
if submit_locator.count() > 0:
submit_locator.first.click()
page.wait_for_load_state("domcontentloaded", timeout=30_000)
except Exception as exc: # noqa: BLE001
logger.warning("Playwright login step failed: %s", exc)
def _accept_cookie_banner(self, page: Any) -> None:
for selector in COOKIE_ACCEPT_SELECTORS:
try:
locator = page.locator(selector).first
if locator.count() == 0:
continue
if not locator.is_visible(timeout=500):
continue
locator.click(timeout=2_000)
page.wait_for_timeout(250)
return
except Exception: # noqa: BLE001
continue
def _save_storage_state(self, session: requests.Session) -> None:
path = self._settings.iaai_storage_state_path
cookies: list[dict[str, Any]] = []
for cookie in session.cookies:
cookie_payload: dict[str, Any] = {
"name": cookie.name,
"value": cookie.value,
"domain": cookie.domain or ".iaai.com",
"path": cookie.path or "/",
"httpOnly": False,
"secure": bool(cookie.secure),
"sameSite": "Lax",
}
if cookie.expires is not None:
cookie_payload["expires"] = int(cookie.expires)
cookies.append(cookie_payload)
payload = {"cookies": cookies, "origins": []}
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
@staticmethod
def _sleep_backoff(retry_backoff_ms: int, attempt: int) -> None:
if retry_backoff_ms <= 0:
return
delay = retry_backoff_ms * (2**attempt) / 1000
time.sleep(delay)
class IAAIClient:
def __init__(self, settings: Settings) -> None:
self._settings = settings
self._auth = HybridSessionAuth(settings)
def persist_session_state(self) -> None:
self._auth.persist_storage_state()
def keep_session_alive(self) -> dict[str, Any]:
started = time.perf_counter()
scope = resolve_listing_scope_paths(
listing_start_url=self._settings.iaai_listing_start_url,
brands=set(),
)[0]
page_html = self._fetch_listing_first_page(scope)
marker_present = LISTING_MARKER in page_html
result_count = parse_int(parse_hidden_input_value(page_html, "ResultCount"))
self.persist_session_state()
return {
"scope": scope,
"marker_present": marker_present,
"result_count": result_count,
"elapsed_ms": int((time.perf_counter() - started) * 1000),
}
def iter_listing_vehicles(self, filters: RuntimeFilters) -> Iterator[ListingVehicle]:
seen_inventory_ids: set[str] = set()
if self._settings.iaai_listing_start_url.strip() and filters.brands:
logger.info(
"Using explicit listing start URL; runtime brand scopes in listing step are ignored."
)
scope_paths = resolve_listing_scope_paths(
listing_start_url=self._settings.iaai_listing_start_url,
brands=filters.brands,
)
listing_started_at = time.perf_counter()
for scope_index, scope_path in enumerate(scope_paths, start=1):
logger.info(
"IAAI listing scope start scope=%s/%s path=%s",
scope_index,
len(scope_paths),
scope_path,
)
first_page_html = self._fetch_listing_first_page(scope_path)
first_page = parse_listing_page(first_page_html)
for vehicle in first_page.vehicles:
if vehicle.inventory_id in seen_inventory_ids:
continue
seen_inventory_ids.add(vehicle.inventory_id)
yield vehicle
page_size = max(1, first_page.page_size)
total_pages = max(1, math.ceil(max(first_page.result_count, len(first_page.vehicles)) / page_size))
gbp_search_query = first_page.gbp_search_query
logger.info(
"IAAI listing scope loaded scope=%s/%s path=%s result_count=%s page_size=%s total_pages=%s unique_ids=%s elapsed=%.1fs",
scope_index,
len(scope_paths),
scope_path,
first_page.result_count,
page_size,
total_pages,
len(seen_inventory_ids),
time.perf_counter() - listing_started_at,
)
for page_number in range(2, total_pages + 1):
page_html = self._fetch_listing_page(scope_path, gbp_search_query, page_number, page_size)
parsed_page = parse_listing_page(page_html)
gbp_search_query = parsed_page.gbp_search_query
for vehicle in parsed_page.vehicles:
if vehicle.inventory_id in seen_inventory_ids:
continue
seen_inventory_ids.add(vehicle.inventory_id)
yield vehicle
if page_number % 10 == 0 or page_number == total_pages:
logger.info(
"IAAI listing page progress scope=%s/%s path=%s page=%s/%s unique_ids=%s elapsed=%.1fs",
scope_index,
len(scope_paths),
scope_path,
page_number,
total_pages,
len(seen_inventory_ids),
time.perf_counter() - listing_started_at,
)
def fetch_vehicle_detail_payload(self, inventory_id: str) -> dict[str, Any]:
escaped_id = quote(inventory_id, safe="~")
url = urljoin(self._settings.iaai_base_url + "/", f"VehicleDetail/{escaped_id}")
response = self._auth.request(
"GET",
url,
timeout=self._settings.http_timeout,
retries=self._settings.http_retries,
retry_backoff_ms=self._settings.http_retry_backoff_ms,
headers={"accept": "text/html,application/xhtml+xml"},
expected_marker=DETAIL_MARKER,
)
with response:
if response.status_code >= 400:
raise RuntimeError(f"Vehicle detail request failed id={inventory_id} status={response.status_code}")
return parse_product_details_vm(response.text)
def _fetch_listing_first_page(self, scope_path: str) -> str:
if scope_path.lower().startswith(("http://", "https://")):
url = scope_path
else:
url = urljoin(self._settings.iaai_base_url + "/", scope_path.lstrip("/"))
response = self._auth.request(
"GET",
url,
timeout=self._settings.http_timeout,
retries=self._settings.http_retries,
retry_backoff_ms=self._settings.http_retry_backoff_ms,
headers={"accept": "text/html,application/xhtml+xml"},
expected_marker=LISTING_MARKER,
)
with response:
if response.status_code >= 400:
raise RuntimeError(f"Listing request failed path={scope_path} status={response.status_code}")
return response.text
def _fetch_listing_page(
self,
scope_path: str,
gbp_search_query: dict[str, Any],
page_number: int,
page_size: int,
) -> str:
query_payload = dict(gbp_search_query)
query_payload["CurrentPage"] = page_number
query_payload["PageSize"] = page_size
search_url = urljoin(self._settings.iaai_base_url + "/", "Search")
common_headers = {
"accept": "text/html,application/xhtml+xml,*/*",
"x-requested-with": "XMLHttpRequest",
}
attempts: list[tuple[dict[str, str], Any, Any]] = [
({**common_headers, "content-type": "application/json"}, None, query_payload),
({**common_headers, "content-type": "application/json"}, None, {"GBPSearchQuery": query_payload}),
({**common_headers}, {"GBPSearchQuery": json.dumps(query_payload, separators=(",", ":"))}, None),
(
{**common_headers, "content-type": "application/json"},
json.dumps({"GBPSearchQuery": json.dumps(query_payload, separators=(",", ":"))}),
None,
),
]
last_error: Exception | None = None
for headers, data, json_body in attempts:
try:
response = self._auth.request(
"POST",
search_url,
timeout=self._settings.http_timeout,
retries=self._settings.http_retries,
retry_backoff_ms=self._settings.http_retry_backoff_ms,
headers=headers,
data=data,
json_body=json_body,
expected_marker=LISTING_MARKER,
)
with response:
if response.status_code >= 400:
raise RuntimeError(
f"Listing page request failed status={response.status_code} page={page_number}"
)
body = response.text
if LISTING_MARKER not in body:
raise RuntimeError("Listing page response does not include GBPSearchQuery")
return body
except Exception as exc: # noqa: BLE001
last_error = exc
continue
if last_error is not None:
raise RuntimeError(f"Failed to load listing page={page_number} for {scope_path}: {last_error}") from last_error
raise RuntimeError(f"Failed to load listing page={page_number} for {scope_path}")
def build_brand_scope_paths(brands: set[str]) -> list[str]:
if not brands:
return ["/Vehiclelisting/Cars"]
paths: list[str] = []
for brand in sorted(brands):
raw = brand.strip()
if not raw:
continue
slug_hyphen = quote(raw.replace(" ", "-"), safe="-")
slug_raw = quote(raw, safe="")
for slug in (slug_hyphen, slug_raw):
path = f"/Vehiclelisting/Cars/{slug}"
if path not in paths:
paths.append(path)
return paths or ["/Vehiclelisting/Cars"]
def resolve_listing_scope_paths(*, listing_start_url: str, brands: set[str]) -> list[str]:
explicit_scope = listing_start_url.strip()
if explicit_scope:
return [explicit_scope]
return build_brand_scope_paths(brands)
def parse_listing_page(html_text: str) -> ListingPage:
gbp_raw = parse_hidden_input_value(html_text, "GBPSearchQuery")
vehicle_raw = parse_hidden_input_value(html_text, "VehicleDetails")
result_count_raw = parse_hidden_input_value(html_text, "ResultCount")
page_size_raw = parse_hidden_input_value(html_text, "PageSize")
current_page_raw = parse_hidden_input_value(html_text, "CurrentPage")
if not gbp_raw:
raise RuntimeError("Listing page missing GBPSearchQuery")
if vehicle_raw is None:
raise RuntimeError("Listing page missing VehicleDetails")
gbp_payload = json.loads(gbp_raw)
if not isinstance(gbp_payload, dict):
raise RuntimeError("GBPSearchQuery payload is not object")
vehicle_payload = json.loads(vehicle_raw)
if not isinstance(vehicle_payload, list):
raise RuntimeError("VehicleDetails payload is not array")
vehicles: list[ListingVehicle] = []
for item in vehicle_payload:
if not isinstance(item, dict):
continue
inventory_id = parse_text(item.get("Id"))
if not inventory_id:
continue
vehicles.append(
ListingVehicle(
inventory_id=inventory_id,
tenant=parse_text(item.get("Tenant")),
auction_id=parse_text(item.get("ActnLnId")),
auction_date=parse_text(item.get("AuctionDate")) or parse_text(item.get("ActnDtTm")),
inventory_status=parse_text(item.get("InventoryStatus")),
currency=parse_text(item.get("Currency")),
timed_auction_closed=parse_bool(item.get("TimedAuctionClosedIndicator")),
timed_auction_indicator=parse_bool(item.get("TimedAuctionIndicator")),
prebid_indicator=parse_bool(item.get("PreBidIndicator")),
buynow_indicator=parse_bool(item.get("BuyNowIndicator")),
)
)
return ListingPage(
vehicles=vehicles,
result_count=parse_int(result_count_raw) or len(vehicles),
page_size=parse_int(page_size_raw) or max(1, len(vehicles)),
current_page=parse_int(current_page_raw) or 1,
gbp_search_query=gbp_payload,
)
def parse_product_details_vm(html_text: str) -> dict[str, Any]:
match = re.search(
r"",
html_text,
flags=re.DOTALL,
)
if match is None:
raise RuntimeError("ProductDetailsVM script not found")
payload = json.loads(match.group(1))
if not isinstance(payload, dict):
raise RuntimeError("ProductDetailsVM root is not object")
return payload
def parse_hidden_input_value(html_text: str, input_id: str) -> str | None:
escaped_id = re.escape(input_id)
patterns = (
rf"]*\bid=\"{escaped_id}\"[^>]*\bvalue=\"([^\"]*)\"",
rf"]*\bid='{escaped_id}'[^>]*\bvalue='([^']*)'",
)
for pattern in patterns:
match = re.search(pattern, html_text, flags=re.IGNORECASE)
if match is not None:
return html.unescape(match.group(1))
return None
def build_resizer_images_from_keys(image_keys: list[dict[str, Any]]) -> list[dict[str, str | int]]:
seen_fullres: set[str] = set()
images: list[dict[str, str | int]] = []
for index, item in enumerate(image_keys):
if not isinstance(item, dict):
continue
key = parse_text(item.get("k"))
if key is None:
continue
width = parse_int(item.get("w")) or 1600
height = parse_int(item.get("h")) or 1200
if width <= 0:
width = 1600
if height <= 0:
height = 1200
order_index = parse_int(item.get("i"))
if order_index is None:
order_index = parse_int(item.get("in"))
if order_index is None:
order_index = index
preview_width = min(640, width)
preview_height = max(1, int(round(height * (preview_width / width))))
escaped_key = quote(key, safe="~")
fullres = f"{RESIZER_URL}?imageKeys={escaped_key}&width={width}&height={height}"
preview = f"{RESIZER_URL}?imageKeys={escaped_key}&width={preview_width}&height={preview_height}"
if fullres in seen_fullres:
continue
seen_fullres.add(fullres)
images.append(
{
"order_index": order_index,
"fullres_image": fullres,
"preview_image": preview,
}
)
images.sort(key=lambda row: (parse_int(row.get("order_index")) or 0, str(row.get("fullres_image"))))
return images
def is_challenge_response(*, status_code: int, body_text: str, expected_marker: str | None = None) -> bool:
if status_code in {401, 403}:
return True
if _expected_marker_present(body_text=body_text, expected_marker=expected_marker):
# If expected listing/detail marker is present, this is a valid page even if
# Incapsula script references are embedded in the HTML.
return False
lowered = (body_text or "").lower()
if any(marker in lowered for marker in CHALLENGE_MARKERS):
return True
if expected_marker and not _expected_marker_present(body_text=body_text, expected_marker=expected_marker):
# Expected hidden marker/script missing from HTML often means anti-bot interstitial.
if " bool:
if not expected_marker:
return False
if expected_marker in body_text:
return True
# Accept quote variants for marker fragments like id="GBPSearchQuery" / id='GBPSearchQuery'.
if '"' in expected_marker:
single_quoted = expected_marker.replace('"', "'")
if single_quoted in body_text:
return True
if "'" in expected_marker:
double_quoted = expected_marker.replace("'", '"')
if double_quoted in body_text:
return True
marker_match = re.search(r"id=['\"]([^'\"]+)['\"]", expected_marker)
if marker_match is None:
return False
marker_id = re.escape(marker_match.group(1))
return bool(
re.search(
rf"id\s*=\s*['\"]{marker_id}['\"]",
body_text,
flags=re.IGNORECASE,
)
)
def parse_text(value: Any) -> str | None:
if isinstance(value, str):
text = value.strip()
return text if text else None
return None
def parse_bool(value: Any) -> bool:
if isinstance(value, bool):
return value
if isinstance(value, str):
normalized = value.strip().lower()
return normalized in {"true", "1", "yes", "on"}
if isinstance(value, (int, float)) and not isinstance(value, bool):
return value != 0
return False
def parse_int(value: Any) -> int | None:
if value is None or isinstance(value, bool):
return None
if isinstance(value, int):
return value
if isinstance(value, float):
return int(round(value))
if isinstance(value, str):
text = value.strip()
if not text:
return None
normalized = text.replace(",", "").replace(" ", "")
try:
return int(round(float(normalized)))
except ValueError:
return None
return None