223 lines
8.3 KiB
Python
223 lines
8.3 KiB
Python
from __future__ import annotations
|
|
|
|
import json
|
|
import re
|
|
from html import unescape
|
|
from html.parser import HTMLParser
|
|
from typing import Any
|
|
from urllib.parse import parse_qs, urlsplit
|
|
|
|
NEXT_FLIGHT_RE = re.compile(r"self\.__next_f\.push\(\[1,\"(.*?)\"\]\)", re.DOTALL)
|
|
SEARCH_CARD_TEST_ID_RE = re.compile(r"^base-result-listing-\d+$")
|
|
VOID_HTML_TAGS = {"area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", "param", "source", "track", "wbr"}
|
|
|
|
|
|
class _SearchCardEnrichmentParser(HTMLParser):
|
|
"""Collect title/subtitle from SSR cards while preserving the listing-ID relation."""
|
|
|
|
def __init__(self) -> None:
|
|
super().__init__(convert_charrefs=True)
|
|
self.enrichment_by_id: dict[str, dict[str, str]] = {}
|
|
self._card: dict[str, Any] | None = None
|
|
self._card_depth = 0
|
|
self._title_container_depth: int | None = None
|
|
self._title_text_depth = 0
|
|
|
|
@staticmethod
|
|
def _listing_id_from_href(href: str | None) -> str | None:
|
|
if not href:
|
|
return None
|
|
values = parse_qs(urlsplit(unescape(href)).query).get("id", [])
|
|
return str(values[0]).strip() if values and str(values[0]).strip() else None
|
|
|
|
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
attributes = dict(attrs)
|
|
test_id = attributes.get("data-testid") or ""
|
|
is_void = tag.lower() in VOID_HTML_TAGS
|
|
|
|
if self._card is None:
|
|
if SEARCH_CARD_TEST_ID_RE.fullmatch(test_id):
|
|
self._card = {"id": None, "title": [], "subtitle": []}
|
|
self._card_depth = 0 if is_void else 1
|
|
return
|
|
|
|
if not is_void:
|
|
self._card_depth += 1
|
|
|
|
if tag.lower() == "a" and self._card.get("id") is None:
|
|
self._card["id"] = self._listing_id_from_href(attributes.get("href"))
|
|
|
|
if test_id.endswith("-title"):
|
|
self._title_container_depth = self._card_depth
|
|
if test_id == "listing-title-card-view":
|
|
self._title_text_depth += 1
|
|
|
|
title_attribute = (attributes.get("title") or "").strip()
|
|
if title_attribute:
|
|
if test_id == "listing-title-card-view":
|
|
self._card["title"].append(title_attribute)
|
|
elif self._title_container_depth is not None:
|
|
self._card["subtitle"].append(title_attribute)
|
|
|
|
def handle_data(self, data: str) -> None:
|
|
if self._card is not None and self._title_text_depth:
|
|
text = data.strip()
|
|
if text:
|
|
self._card["title"].append(text)
|
|
|
|
def handle_endtag(self, tag: str) -> None:
|
|
if self._card is None or tag.lower() in VOID_HTML_TAGS:
|
|
return
|
|
if self._title_text_depth and tag.lower() == "span":
|
|
self._title_text_depth -= 1
|
|
if self._title_container_depth == self._card_depth:
|
|
self._title_container_depth = None
|
|
self._card_depth -= 1
|
|
if self._card_depth != 0:
|
|
return
|
|
|
|
listing_id = self._card.get("id")
|
|
title = " ".join(dict.fromkeys(self._card["title"])).strip()
|
|
subtitle = " ".join(dict.fromkeys(self._card["subtitle"])).strip()
|
|
if listing_id and (title or subtitle):
|
|
self.enrichment_by_id[str(listing_id)] = {
|
|
"searchSsrTitle": title,
|
|
"searchSsrSubtitle": subtitle,
|
|
"searchSsrDriveText": " ".join(part for part in (title, subtitle) if part),
|
|
}
|
|
self._card = None
|
|
self._title_container_depth = None
|
|
self._title_text_depth = 0
|
|
|
|
|
|
def extract_search_card_enrichment(html: str) -> dict[str, dict[str, str]]:
|
|
"""Return SSR title/subtitle enrichment keyed by the Search listing ID.
|
|
|
|
This reads only fields rendered inside the card's title block; it never uses
|
|
generic dealer, location, or arbitrary card text as a vehicle attribute.
|
|
"""
|
|
parser = _SearchCardEnrichmentParser()
|
|
parser.feed(html)
|
|
parser.close()
|
|
return parser.enrichment_by_id
|
|
|
|
|
|
def extract_next_flight_strings(html: str) -> list[str]:
|
|
"""Extract decoded Next.js Flight chunks from mobile.de HTML."""
|
|
chunks: list[str] = []
|
|
for match in NEXT_FLIGHT_RE.finditer(html):
|
|
raw = match.group(1)
|
|
try:
|
|
chunks.append(json.loads(f'"{raw}"'))
|
|
except json.JSONDecodeError:
|
|
# Резервный вариант: сохраняем работоспособность парсера,
|
|
# даже если один из фрагментов имеет нестандартное экранирование.
|
|
chunks.append(raw.encode("utf-8", errors="ignore").decode("unicode_escape", errors="ignore"))
|
|
return chunks
|
|
|
|
|
|
def extract_json_object_after(text: str, marker: str) -> dict[str, Any] | None:
|
|
"""Return JSON object that starts immediately after a marker in a decoded Flight chunk."""
|
|
marker_index = text.find(marker)
|
|
if marker_index < 0:
|
|
return None
|
|
start = text.find("{", marker_index + len(marker))
|
|
if start < 0:
|
|
return None
|
|
|
|
depth = 0
|
|
in_string = False
|
|
escaped = False
|
|
for index in range(start, len(text)):
|
|
char = text[index]
|
|
if in_string:
|
|
if escaped:
|
|
escaped = False
|
|
elif char == "\\":
|
|
escaped = True
|
|
elif char == '"':
|
|
in_string = False
|
|
continue
|
|
if char == '"':
|
|
in_string = True
|
|
elif char == "{":
|
|
depth += 1
|
|
elif char == "}":
|
|
depth -= 1
|
|
if depth == 0:
|
|
candidate = text[start : index + 1]
|
|
try:
|
|
return json.loads(candidate)
|
|
except json.JSONDecodeError:
|
|
return None
|
|
return None
|
|
|
|
|
|
def extract_search_results(html: str) -> dict[str, Any]:
|
|
"""Extract searchResults from mobile.de SRP HTML."""
|
|
for chunk in extract_next_flight_strings(html):
|
|
if '"eventScope":"page-srp"' not in chunk or '"searchResults"' not in chunk:
|
|
continue
|
|
results = extract_json_object_after(chunk, '"searchResults":')
|
|
if isinstance(results, dict):
|
|
return results
|
|
return {}
|
|
|
|
|
|
def extract_detail_listing(html: str) -> dict[str, Any]:
|
|
"""Extract listing object from mobile.de VIP/detail HTML."""
|
|
for chunk in extract_next_flight_strings(html):
|
|
if '"eventScope":"page-vip"' not in chunk or '"listing"' not in chunk:
|
|
continue
|
|
listing = extract_json_object_after(chunk, '"listing":')
|
|
if isinstance(listing, dict):
|
|
return listing
|
|
return extract_legacy_detail_listing(html)
|
|
|
|
|
|
def extract_legacy_detail_listing(html: str) -> dict[str, Any]:
|
|
"""Extract and normalize the compact legacy VIP INITIAL_STATE payload."""
|
|
state = extract_json_object_after(html, "window.__INITIAL_STATE__ =")
|
|
if not isinstance(state, dict):
|
|
return {}
|
|
|
|
search = state.get("search")
|
|
vip = search.get("vip") if isinstance(search, dict) else None
|
|
ads = vip.get("ads") if isinstance(vip, dict) else None
|
|
if not isinstance(ads, dict):
|
|
return {}
|
|
|
|
for payload in ads.values():
|
|
data = payload.get("data") if isinstance(payload, dict) else None
|
|
ad = data.get("ad") if isinstance(data, dict) else None
|
|
if not isinstance(ad, dict):
|
|
continue
|
|
|
|
listing = dict(ad)
|
|
make = listing.get("make")
|
|
if isinstance(make, str):
|
|
listing["make"] = {"localized": make}
|
|
model = listing.get("model")
|
|
if isinstance(model, str):
|
|
listing["model"] = {"localized": model}
|
|
|
|
contact_info = listing.get("contactInfo")
|
|
if "contact" not in listing and isinstance(contact_info, dict):
|
|
listing["contact"] = contact_info
|
|
|
|
gallery_images = listing.get("galleryImages")
|
|
if "images" not in listing and isinstance(gallery_images, list):
|
|
listing["images"] = gallery_images
|
|
|
|
price = listing.get("price")
|
|
if isinstance(price, dict) and not isinstance(price.get("grs"), dict):
|
|
amount = price.get("grossAmount")
|
|
currency = price.get("grossCurrency")
|
|
if amount is not None or currency is not None:
|
|
normalized_price = dict(price)
|
|
normalized_price["grs"] = {"amount": amount, "currency": currency}
|
|
listing["price"] = normalized_price
|
|
|
|
return listing
|
|
return {}
|