from __future__ import annotations import json import re from html import unescape from html.parser import HTMLParser from typing import Any from urllib.parse import parse_qs, urlsplit NEXT_FLIGHT_RE = re.compile(r"self\.__next_f\.push\(\[1,\"(.*?)\"\]\)", re.DOTALL) SEARCH_CARD_TEST_ID_RE = re.compile(r"^base-result-listing-\d+$") VOID_HTML_TAGS = {"area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", "param", "source", "track", "wbr"} class _SearchCardEnrichmentParser(HTMLParser): """Collect title/subtitle from SSR cards while preserving the listing-ID relation.""" def __init__(self) -> None: super().__init__(convert_charrefs=True) self.enrichment_by_id: dict[str, dict[str, str]] = {} self._card: dict[str, Any] | None = None self._card_depth = 0 self._title_container_depth: int | None = None self._title_text_depth = 0 @staticmethod def _listing_id_from_href(href: str | None) -> str | None: if not href: return None values = parse_qs(urlsplit(unescape(href)).query).get("id", []) return str(values[0]).strip() if values and str(values[0]).strip() else None def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: attributes = dict(attrs) test_id = attributes.get("data-testid") or "" is_void = tag.lower() in VOID_HTML_TAGS if self._card is None: if SEARCH_CARD_TEST_ID_RE.fullmatch(test_id): self._card = {"id": None, "title": [], "subtitle": []} self._card_depth = 0 if is_void else 1 return if not is_void: self._card_depth += 1 if tag.lower() == "a" and self._card.get("id") is None: self._card["id"] = self._listing_id_from_href(attributes.get("href")) if test_id.endswith("-title"): self._title_container_depth = self._card_depth if test_id == "listing-title-card-view": self._title_text_depth += 1 title_attribute = (attributes.get("title") or "").strip() if title_attribute: if test_id == "listing-title-card-view": self._card["title"].append(title_attribute) elif self._title_container_depth is not None: self._card["subtitle"].append(title_attribute) def handle_data(self, data: str) -> None: if self._card is not None and self._title_text_depth: text = data.strip() if text: self._card["title"].append(text) def handle_endtag(self, tag: str) -> None: if self._card is None or tag.lower() in VOID_HTML_TAGS: return if self._title_text_depth and tag.lower() == "span": self._title_text_depth -= 1 if self._title_container_depth == self._card_depth: self._title_container_depth = None self._card_depth -= 1 if self._card_depth != 0: return listing_id = self._card.get("id") title = " ".join(dict.fromkeys(self._card["title"])).strip() subtitle = " ".join(dict.fromkeys(self._card["subtitle"])).strip() if listing_id and (title or subtitle): self.enrichment_by_id[str(listing_id)] = { "searchSsrTitle": title, "searchSsrSubtitle": subtitle, "searchSsrDriveText": " ".join(part for part in (title, subtitle) if part), } self._card = None self._title_container_depth = None self._title_text_depth = 0 def extract_search_card_enrichment(html: str) -> dict[str, dict[str, str]]: """Return SSR title/subtitle enrichment keyed by the Search listing ID. This reads only fields rendered inside the card's title block; it never uses generic dealer, location, or arbitrary card text as a vehicle attribute. """ parser = _SearchCardEnrichmentParser() parser.feed(html) parser.close() return parser.enrichment_by_id def extract_next_flight_strings(html: str) -> list[str]: """Extract decoded Next.js Flight chunks from mobile.de HTML.""" chunks: list[str] = [] for match in NEXT_FLIGHT_RE.finditer(html): raw = match.group(1) try: chunks.append(json.loads(f'"{raw}"')) except json.JSONDecodeError: # Резервный вариант: сохраняем работоспособность парсера, # даже если один из фрагментов имеет нестандартное экранирование. chunks.append(raw.encode("utf-8", errors="ignore").decode("unicode_escape", errors="ignore")) return chunks def extract_json_object_after(text: str, marker: str) -> dict[str, Any] | None: """Return JSON object that starts immediately after a marker in a decoded Flight chunk.""" marker_index = text.find(marker) if marker_index < 0: return None start = text.find("{", marker_index + len(marker)) if start < 0: return None depth = 0 in_string = False escaped = False for index in range(start, len(text)): char = text[index] if in_string: if escaped: escaped = False elif char == "\\": escaped = True elif char == '"': in_string = False continue if char == '"': in_string = True elif char == "{": depth += 1 elif char == "}": depth -= 1 if depth == 0: candidate = text[start : index + 1] try: return json.loads(candidate) except json.JSONDecodeError: return None return None def extract_search_results(html: str) -> dict[str, Any]: """Extract searchResults from mobile.de SRP HTML.""" for chunk in extract_next_flight_strings(html): if '"eventScope":"page-srp"' not in chunk or '"searchResults"' not in chunk: continue results = extract_json_object_after(chunk, '"searchResults":') if isinstance(results, dict): return results return {} def extract_detail_listing(html: str) -> dict[str, Any]: """Extract listing object from mobile.de VIP/detail HTML.""" for chunk in extract_next_flight_strings(html): if '"eventScope":"page-vip"' not in chunk or '"listing"' not in chunk: continue listing = extract_json_object_after(chunk, '"listing":') if isinstance(listing, dict): return listing return extract_legacy_detail_listing(html) def extract_legacy_detail_listing(html: str) -> dict[str, Any]: """Extract and normalize the compact legacy VIP INITIAL_STATE payload.""" state = extract_json_object_after(html, "window.__INITIAL_STATE__ =") if not isinstance(state, dict): return {} search = state.get("search") vip = search.get("vip") if isinstance(search, dict) else None ads = vip.get("ads") if isinstance(vip, dict) else None if not isinstance(ads, dict): return {} for payload in ads.values(): data = payload.get("data") if isinstance(payload, dict) else None ad = data.get("ad") if isinstance(data, dict) else None if not isinstance(ad, dict): continue listing = dict(ad) make = listing.get("make") if isinstance(make, str): listing["make"] = {"localized": make} model = listing.get("model") if isinstance(model, str): listing["model"] = {"localized": model} contact_info = listing.get("contactInfo") if "contact" not in listing and isinstance(contact_info, dict): listing["contact"] = contact_info gallery_images = listing.get("galleryImages") if "images" not in listing and isinstance(gallery_images, list): listing["images"] = gallery_images price = listing.get("price") if isinstance(price, dict) and not isinstance(price.get("grs"), dict): amount = price.get("grossAmount") currency = price.get("grossCurrency") if amount is not None or currency is not None: normalized_price = dict(price) normalized_price["grs"] = {"amount": amount, "currency": currency} listing["price"] = normalized_price return listing return {}