Files
mobile.de/mobilede_scraper/mobile_de/flight.py
T
2026-08-19 13:28:00 +03:00

223 lines
8.3 KiB
Python

from __future__ import annotations
import json
import re
from html import unescape
from html.parser import HTMLParser
from typing import Any
from urllib.parse import parse_qs, urlsplit
NEXT_FLIGHT_RE = re.compile(r"self\.__next_f\.push\(\[1,\"(.*?)\"\]\)", re.DOTALL)
SEARCH_CARD_TEST_ID_RE = re.compile(r"^base-result-listing-\d+$")
VOID_HTML_TAGS = {"area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", "param", "source", "track", "wbr"}
class _SearchCardEnrichmentParser(HTMLParser):
"""Collect title/subtitle from SSR cards while preserving the listing-ID relation."""
def __init__(self) -> None:
super().__init__(convert_charrefs=True)
self.enrichment_by_id: dict[str, dict[str, str]] = {}
self._card: dict[str, Any] | None = None
self._card_depth = 0
self._title_container_depth: int | None = None
self._title_text_depth = 0
@staticmethod
def _listing_id_from_href(href: str | None) -> str | None:
if not href:
return None
values = parse_qs(urlsplit(unescape(href)).query).get("id", [])
return str(values[0]).strip() if values and str(values[0]).strip() else None
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
attributes = dict(attrs)
test_id = attributes.get("data-testid") or ""
is_void = tag.lower() in VOID_HTML_TAGS
if self._card is None:
if SEARCH_CARD_TEST_ID_RE.fullmatch(test_id):
self._card = {"id": None, "title": [], "subtitle": []}
self._card_depth = 0 if is_void else 1
return
if not is_void:
self._card_depth += 1
if tag.lower() == "a" and self._card.get("id") is None:
self._card["id"] = self._listing_id_from_href(attributes.get("href"))
if test_id.endswith("-title"):
self._title_container_depth = self._card_depth
if test_id == "listing-title-card-view":
self._title_text_depth += 1
title_attribute = (attributes.get("title") or "").strip()
if title_attribute:
if test_id == "listing-title-card-view":
self._card["title"].append(title_attribute)
elif self._title_container_depth is not None:
self._card["subtitle"].append(title_attribute)
def handle_data(self, data: str) -> None:
if self._card is not None and self._title_text_depth:
text = data.strip()
if text:
self._card["title"].append(text)
def handle_endtag(self, tag: str) -> None:
if self._card is None or tag.lower() in VOID_HTML_TAGS:
return
if self._title_text_depth and tag.lower() == "span":
self._title_text_depth -= 1
if self._title_container_depth == self._card_depth:
self._title_container_depth = None
self._card_depth -= 1
if self._card_depth != 0:
return
listing_id = self._card.get("id")
title = " ".join(dict.fromkeys(self._card["title"])).strip()
subtitle = " ".join(dict.fromkeys(self._card["subtitle"])).strip()
if listing_id and (title or subtitle):
self.enrichment_by_id[str(listing_id)] = {
"searchSsrTitle": title,
"searchSsrSubtitle": subtitle,
"searchSsrDriveText": " ".join(part for part in (title, subtitle) if part),
}
self._card = None
self._title_container_depth = None
self._title_text_depth = 0
def extract_search_card_enrichment(html: str) -> dict[str, dict[str, str]]:
"""Return SSR title/subtitle enrichment keyed by the Search listing ID.
This reads only fields rendered inside the card's title block; it never uses
generic dealer, location, or arbitrary card text as a vehicle attribute.
"""
parser = _SearchCardEnrichmentParser()
parser.feed(html)
parser.close()
return parser.enrichment_by_id
def extract_next_flight_strings(html: str) -> list[str]:
"""Extract decoded Next.js Flight chunks from mobile.de HTML."""
chunks: list[str] = []
for match in NEXT_FLIGHT_RE.finditer(html):
raw = match.group(1)
try:
chunks.append(json.loads(f'"{raw}"'))
except json.JSONDecodeError:
# Резервный вариант: сохраняем работоспособность парсера,
# даже если один из фрагментов имеет нестандартное экранирование.
chunks.append(raw.encode("utf-8", errors="ignore").decode("unicode_escape", errors="ignore"))
return chunks
def extract_json_object_after(text: str, marker: str) -> dict[str, Any] | None:
"""Return JSON object that starts immediately after a marker in a decoded Flight chunk."""
marker_index = text.find(marker)
if marker_index < 0:
return None
start = text.find("{", marker_index + len(marker))
if start < 0:
return None
depth = 0
in_string = False
escaped = False
for index in range(start, len(text)):
char = text[index]
if in_string:
if escaped:
escaped = False
elif char == "\\":
escaped = True
elif char == '"':
in_string = False
continue
if char == '"':
in_string = True
elif char == "{":
depth += 1
elif char == "}":
depth -= 1
if depth == 0:
candidate = text[start : index + 1]
try:
return json.loads(candidate)
except json.JSONDecodeError:
return None
return None
def extract_search_results(html: str) -> dict[str, Any]:
"""Extract searchResults from mobile.de SRP HTML."""
for chunk in extract_next_flight_strings(html):
if '"eventScope":"page-srp"' not in chunk or '"searchResults"' not in chunk:
continue
results = extract_json_object_after(chunk, '"searchResults":')
if isinstance(results, dict):
return results
return {}
def extract_detail_listing(html: str) -> dict[str, Any]:
"""Extract listing object from mobile.de VIP/detail HTML."""
for chunk in extract_next_flight_strings(html):
if '"eventScope":"page-vip"' not in chunk or '"listing"' not in chunk:
continue
listing = extract_json_object_after(chunk, '"listing":')
if isinstance(listing, dict):
return listing
return extract_legacy_detail_listing(html)
def extract_legacy_detail_listing(html: str) -> dict[str, Any]:
"""Extract and normalize the compact legacy VIP INITIAL_STATE payload."""
state = extract_json_object_after(html, "window.__INITIAL_STATE__ =")
if not isinstance(state, dict):
return {}
search = state.get("search")
vip = search.get("vip") if isinstance(search, dict) else None
ads = vip.get("ads") if isinstance(vip, dict) else None
if not isinstance(ads, dict):
return {}
for payload in ads.values():
data = payload.get("data") if isinstance(payload, dict) else None
ad = data.get("ad") if isinstance(data, dict) else None
if not isinstance(ad, dict):
continue
listing = dict(ad)
make = listing.get("make")
if isinstance(make, str):
listing["make"] = {"localized": make}
model = listing.get("model")
if isinstance(model, str):
listing["model"] = {"localized": model}
contact_info = listing.get("contactInfo")
if "contact" not in listing and isinstance(contact_info, dict):
listing["contact"] = contact_info
gallery_images = listing.get("galleryImages")
if "images" not in listing and isinstance(gallery_images, list):
listing["images"] = gallery_images
price = listing.get("price")
if isinstance(price, dict) and not isinstance(price.get("grs"), dict):
amount = price.get("grossAmount")
currency = price.get("grossCurrency")
if amount is not None or currency is not None:
normalized_price = dict(price)
normalized_price["grs"] = {"amount": amount, "currency": currency}
listing["price"] = normalized_price
return listing
return {}