368 lines
15 KiB
Python
368 lines
15 KiB
Python
import html as html_module
|
|
import json
|
|
import logging
|
|
import re
|
|
from typing import Any
|
|
|
|
from ..core.utils import LOT_RE, PRICE_RE, VIN_RE, deep_find_key, first_non_empty
|
|
|
|
logger = logging.getLogger("iaai_scraper.parsers")
|
|
|
|
|
|
class VehicleParser:
|
|
"""DOM + JSON парсер страницы авто."""
|
|
|
|
SUMMARY_KEY_MAP = {
|
|
"vin": {"vin", "vehicleidentificationnumber"},
|
|
"lot_number": {"lotnumber", "stockno", "itemid", "lotid", "itemnumber"},
|
|
"year": {"year"},
|
|
"make": {"make", "manufacturer", "brand"},
|
|
"model": {"model"},
|
|
"trim": {"trim", "series"},
|
|
"odometer": {"odometer", "odometermiles", "mileage", "actualcashvalueodometer"},
|
|
"primary_damage": {"primarydamage", "damage", "damagetype", "loss"},
|
|
"secondary_damage": {"secondarydamage"},
|
|
"run_and_drive": {"runanddrive", "canrunanddrive", "rundrive"},
|
|
"buy_now": {"buynowprice", "buyitnowprice", "instantpurchaseprice"},
|
|
"current_bid": {"currentbid", "highbid", "bidamount", "currenthighbid"},
|
|
"actual_cash_value": {"actualcashvalue", "acv"},
|
|
"estimated_repair_cost": {"estimatedrepaircost", "repaircost"},
|
|
"keys": {"keys", "keystatus"},
|
|
"title": {"titletype", "title", "documenttype"},
|
|
"seller": {"seller", "sellername"},
|
|
"location": {"location", "branchname", "auctionlocation", "branch"},
|
|
"auction_date": {"auctiondate", "saledate", "liveauctiondate"},
|
|
"body_type": {"bodytype", "bodystyle"},
|
|
"drive": {"driveline", "drive"},
|
|
"gearbox": {"transmission", "gearbox"},
|
|
"engine": {"engine", "enginevolume"},
|
|
"fuel_type": {"fueltype", "fuel"},
|
|
"cylinders": {"cylinders", "cylindercount"},
|
|
"color": {"color", "primarycolor", "exteriorcolor"},
|
|
}
|
|
|
|
DOM_LABEL_MAP: dict[str, str] = {
|
|
"stock #": "lot_number",
|
|
"vin (status)": "vin",
|
|
"vin": "vin",
|
|
"primary damage": "primary_damage",
|
|
"secondary damage": "secondary_damage",
|
|
"odometer": "odometer",
|
|
"body style": "body_type",
|
|
"engine": "engine",
|
|
"transmission": "gearbox",
|
|
"drive line type": "drive",
|
|
"fuel type": "fuel_type",
|
|
"cylinders": "cylinders",
|
|
"exterior/interior": "color",
|
|
"exterior color": "color",
|
|
"model": "model",
|
|
"series": "trim",
|
|
"selling branch": "location",
|
|
"vehicle location": "vehicle_location",
|
|
"auction date and time": "auction_date",
|
|
"lane/run #": "lane",
|
|
"actual cash value": "actual_cash_value",
|
|
"estimated repair cost": "estimated_repair_cost",
|
|
"seller": "seller",
|
|
"title/sale doc": "title",
|
|
"title/sale doc brand": "title_brand",
|
|
"start code": "run_and_drive",
|
|
"key": "keys",
|
|
"manufactured in": "manufactured_in",
|
|
"vehicle class": "vehicle_class",
|
|
}
|
|
|
|
def _parse_dom_key_value_pairs(self, dom_text: str) -> dict[str, str]:
|
|
result: dict[str, str] = {}
|
|
if not dom_text:
|
|
return result
|
|
lines = [line.strip() for line in dom_text.split("\n") if line.strip()]
|
|
known_labels = set(self.DOM_LABEL_MAP.keys())
|
|
for i, line in enumerate(lines):
|
|
clean = line.rstrip(":").lower().strip()
|
|
if clean in known_labels and i + 1 < len(lines):
|
|
value = lines[i + 1].strip()
|
|
if value.rstrip(":").lower().strip() in known_labels:
|
|
continue
|
|
if value.lower() in ("more actions", "view", "print", "share", "back to results"):
|
|
continue
|
|
field_name = self.DOM_LABEL_MAP[clean]
|
|
if field_name not in result or not result[field_name]:
|
|
result[field_name] = value
|
|
return result
|
|
|
|
def _parse_title_for_year_make_model(self, page_title: str, dom_text: str) -> dict[str, str | None]:
|
|
result: dict[str, str | None] = {"year": None, "make": None, "model": None}
|
|
title_match = re.match(r"(\d{4})\s+(\S+)\s+(.+?)(?:\s+for\s+)", page_title or "")
|
|
if title_match:
|
|
result["year"] = title_match.group(1)
|
|
result["make"] = title_match.group(2)
|
|
result["model"] = title_match.group(3)
|
|
return result
|
|
dom_match = re.search(r"(?:Search|Log In)\s*\n\s*(\d{4})\s+(\S+)\s+(.+?)(?:\n|$)", dom_text or "")
|
|
if dom_match:
|
|
result["year"] = dom_match.group(1)
|
|
result["make"] = dom_match.group(2)
|
|
result["model"] = dom_match.group(3).strip()
|
|
return result
|
|
|
|
def normalize(self, vehicle_url: str, page_html: str, dom_text: str, network_dump: dict[str, Any]) -> dict[str, Any]:
|
|
page_html = page_html or ""
|
|
dom_text = dom_text or ""
|
|
network_dump = network_dump or {}
|
|
responses = network_dump.get("json_responses", [])
|
|
payloads = [item.get("payload") for item in responses if isinstance(item.get("payload"), (dict, list))]
|
|
dom_kv = self._parse_dom_key_value_pairs(dom_text)
|
|
page_title = ""
|
|
title_match = re.search(r"<title[^>]*>(.*?)</title>", page_html or "", re.IGNORECASE | re.DOTALL)
|
|
if title_match:
|
|
page_title = title_match.group(1).strip()
|
|
title_parsed = self._parse_title_for_year_make_model(page_title, dom_text)
|
|
|
|
summary: dict[str, Any] = {"source_url": vehicle_url}
|
|
for field, candidate_keys in self.SUMMARY_KEY_MAP.items():
|
|
values: list[Any] = []
|
|
for payload in payloads:
|
|
values.extend(deep_find_key(payload, candidate_keys))
|
|
if field in dom_kv:
|
|
values.append(dom_kv[field])
|
|
summary[field] = first_non_empty(values)
|
|
|
|
summary["year"] = summary.get("year") or title_parsed.get("year")
|
|
summary["make"] = summary.get("make") or title_parsed.get("make")
|
|
summary["model"] = summary.get("model") or title_parsed.get("model")
|
|
summary["trim"] = summary.get("trim") or dom_kv.get("trim")
|
|
summary["vin"] = summary.get("vin") or self._extract_vin(dom_text) or self._extract_vin(page_html)
|
|
if summary.get("vin") and isinstance(summary["vin"], str):
|
|
vin_clean = re.sub(r"\s*\(.*?\)\s*$", "", summary["vin"]).strip()
|
|
vin_match = VIN_RE.search(vin_clean)
|
|
summary["vin"] = vin_match.group(1) if vin_match else vin_clean
|
|
summary["lot_number"] = summary.get("lot_number") or self._extract_lot_number(dom_text)
|
|
summary["image_urls"] = self._extract_image_urls(payloads, page_html, vehicle_url)
|
|
for dom_field, dom_value in dom_kv.items():
|
|
if dom_field not in summary or not summary[dom_field]:
|
|
summary[dom_field] = dom_value
|
|
prices = self._extract_prices_from_text(dom_text)
|
|
if not summary.get("actual_cash_value") and prices:
|
|
summary["actual_cash_value"] = prices[0]
|
|
if not summary.get("buy_now"):
|
|
buy_now_match = re.search(r"Buy\s+Now[:\s]*\$\s*([\d,]+(?:\.\d{1,2})?)", dom_text or "", re.IGNORECASE)
|
|
if buy_now_match:
|
|
summary["buy_now"] = buy_now_match.group(1)
|
|
elif prices:
|
|
summary["buy_now"] = prices[0]
|
|
if not summary.get("current_bid") and len(prices) > 1:
|
|
summary["current_bid"] = prices[1]
|
|
|
|
embedded = self._extract_embedded_json(page_html)
|
|
for item in embedded:
|
|
p = item.get("payload")
|
|
if isinstance(p, (dict, list)):
|
|
payloads.append(p)
|
|
for field, candidate_keys in self.SUMMARY_KEY_MAP.items():
|
|
if not summary.get(field):
|
|
val = first_non_empty(deep_find_key(p, candidate_keys))
|
|
if val:
|
|
summary[field] = val
|
|
|
|
return {
|
|
"vehicle_summary": summary,
|
|
"payload_insights": self._build_payload_insights(summary, responses, payloads, vehicle_url),
|
|
"embedded_json": embedded,
|
|
"dom_hints": self._dom_hints(dom_text),
|
|
"access_notes": self._build_access_notes(summary, responses),
|
|
}
|
|
|
|
def _build_payload_insights(self, summary: dict[str, Any], responses: list[dict[str, Any]], payloads: list[Any], vehicle_url: str = "") -> dict[str, Any]:
|
|
image_urls = self._extract_image_urls(payloads, "", vehicle_url)
|
|
return {
|
|
"vehicle_core": {
|
|
"vin": summary.get("vin"), "lot_number": summary.get("lot_number"), "year": summary.get("year"),
|
|
"make": summary.get("make"), "model": summary.get("model"), "trim": summary.get("trim"),
|
|
"odometer": summary.get("odometer"), "run_and_drive": summary.get("run_and_drive"),
|
|
"seller": summary.get("seller"), "location": summary.get("location"), "title": summary.get("title"),
|
|
"body_type": summary.get("body_type"), "drive": summary.get("drive"), "gearbox": summary.get("gearbox"),
|
|
"engine": summary.get("engine"), "fuel_type": summary.get("fuel_type"), "cylinders": summary.get("cylinders"),
|
|
"color": summary.get("color"), "keys": summary.get("keys"),
|
|
},
|
|
"pricing": {
|
|
"buy_now": summary.get("buy_now"), "current_bid": summary.get("current_bid"),
|
|
"actual_cash_value": summary.get("actual_cash_value"), "estimated_repair_cost": summary.get("estimated_repair_cost"),
|
|
"currency": self._guess_currency(summary),
|
|
},
|
|
"bids": self._build_bid_insights(summary, payloads),
|
|
"damage": {
|
|
"primary": summary.get("primary_damage"),
|
|
"secondary": summary.get("secondary_damage"),
|
|
"description": first_non_empty(self._find_in_payloads(payloads, {"damageDescription", "damageDetails"})),
|
|
},
|
|
"auction": {
|
|
"auction_date": summary.get("auction_date"),
|
|
"lane": first_non_empty(self._find_in_payloads(payloads, {"lane", "lanename"})),
|
|
"branch": first_non_empty([summary.get("location"), *self._find_in_payloads(payloads, {"branch", "branchname"})]),
|
|
"sale_status": first_non_empty(self._find_in_payloads(payloads, {"salestatus", "auctionstatus", "status"})),
|
|
"item_number": first_non_empty([summary.get("lot_number"), *self._find_in_payloads(payloads, {"itemnumber", "lotnumber", "lotid"})]),
|
|
},
|
|
"images": {"count": len(image_urls), "urls": image_urls},
|
|
"source_endpoints": self._build_source_endpoints(responses),
|
|
}
|
|
|
|
def _build_bid_insights(self, summary: dict[str, Any], payloads: list[Any]) -> dict[str, Any]:
|
|
return {
|
|
"amount": summary.get("current_bid"),
|
|
"currency": self._guess_currency(summary),
|
|
"bid_count": first_non_empty(self._find_in_payloads(payloads, {"bidcount", "numberofbids"})),
|
|
"status": first_non_empty(self._find_in_payloads(payloads, {"bidstatus", "biddingstatus"})),
|
|
}
|
|
|
|
def _build_source_endpoints(self, responses: list[dict[str, Any]]) -> dict[str, list[str]]:
|
|
mapping = {"vehicle": [], "pricing": [], "bids": [], "damage": [], "auction": [], "images": []}
|
|
for item in responses:
|
|
url = item.get("url", "")
|
|
category = item.get("category", "other")
|
|
if category == "vehicle":
|
|
mapping["vehicle"].append(url)
|
|
lowered = url.lower()
|
|
if any(token in lowered for token in ["bid", "offer"]):
|
|
mapping["bids"].append(url)
|
|
if any(token in lowered for token in ["damage", "report"]):
|
|
mapping["damage"].append(url)
|
|
if any(token in lowered for token in ["auction", "sale", "lane", "branch"]):
|
|
mapping["auction"].append(url)
|
|
elif category in mapping:
|
|
mapping[category].append(url)
|
|
return {key: list(dict.fromkeys(urls)) for key, urls in mapping.items()}
|
|
|
|
def _build_access_notes(self, summary: dict[str, Any], responses: list[dict[str, Any]]) -> dict[str, Any]:
|
|
endpoints = [item.get("url", "") for item in responses]
|
|
return {
|
|
"vin_visible": bool(summary.get("vin")),
|
|
"images_visible": bool(summary.get("image_urls")),
|
|
"network_json_count": len(responses),
|
|
"possible_captcha": False,
|
|
"possible_antibot": False,
|
|
"observed_endpoints": endpoints[:20],
|
|
}
|
|
|
|
@staticmethod
|
|
def _find_in_payloads(payloads: list[Any], keys: set[str]) -> list[Any]:
|
|
lowered = {key.lower() for key in keys}
|
|
values: list[Any] = []
|
|
for payload in payloads:
|
|
values.extend(deep_find_key(payload, lowered))
|
|
return values
|
|
|
|
@staticmethod
|
|
def _guess_currency(summary: dict[str, Any]) -> str:
|
|
for key in ["buy_now", "current_bid", "actual_cash_value", "estimated_repair_cost"]:
|
|
value = str(summary.get(key) or "")
|
|
if "$" in value:
|
|
return "USD"
|
|
if "€" in value:
|
|
return "EUR"
|
|
if "¥" in value:
|
|
return "JPY"
|
|
return "USD"
|
|
|
|
@staticmethod
|
|
def _extract_vin(text: str) -> str | None:
|
|
match = VIN_RE.search(text or "")
|
|
return match.group(1) if match else None
|
|
|
|
@staticmethod
|
|
def _extract_lot_number(text: str) -> str | None:
|
|
match = LOT_RE.search(text or "")
|
|
return match.group(1) if match else None
|
|
|
|
@staticmethod
|
|
def _extract_prices_from_text(text: str) -> list[str]:
|
|
return [match.group(1) for match in PRICE_RE.finditer(text or "")]
|
|
|
|
@staticmethod
|
|
def _extract_embedded_json(html: str) -> list[dict[str, Any]]:
|
|
scripts = re.findall(r"<script[^>]*>(.*?)</script>", html or "", flags=re.DOTALL | re.IGNORECASE)
|
|
extracted: list[dict[str, Any]] = []
|
|
for script_text in scripts:
|
|
if "{" not in script_text and "[" not in script_text:
|
|
continue
|
|
try:
|
|
parsed = json.loads(script_text.strip())
|
|
except Exception:
|
|
continue
|
|
extracted.append({"type": "inline_json", "payload": parsed})
|
|
return extracted
|
|
|
|
@staticmethod
|
|
def _extract_image_urls(payloads: list[Any], html: str, vehicle_url: str = "") -> list[str]:
|
|
vehicle_key = ""
|
|
key_match = re.search(r"VehicleDetail/(\d+)", vehicle_url or "")
|
|
if key_match:
|
|
vehicle_key = key_match.group(1)
|
|
found: list[str] = []
|
|
for payload in payloads:
|
|
found.extend(deep_find_key(payload, {"imageurl", "imageurls", "url", "fullsizeurl", "thumbnailurl", "originalurl"}))
|
|
flat: list[str] = []
|
|
seen_flat: set[str] = set()
|
|
for item in found:
|
|
if isinstance(item, str) and item.startswith("http"):
|
|
cleaned = html_module.unescape(item)
|
|
lowered = cleaned.lower()
|
|
if vehicle_key and "vis.iaai.com" in lowered and vehicle_key not in cleaned:
|
|
continue
|
|
if cleaned not in seen_flat:
|
|
seen_flat.add(cleaned)
|
|
flat.append(cleaned)
|
|
elif isinstance(item, list):
|
|
for child in item:
|
|
if isinstance(child, str) and child.startswith("http"):
|
|
cleaned = html_module.unescape(child)
|
|
lowered = cleaned.lower()
|
|
if vehicle_key and "vis.iaai.com" in lowered and vehicle_key not in cleaned:
|
|
continue
|
|
if cleaned not in seen_flat:
|
|
seen_flat.add(cleaned)
|
|
flat.append(cleaned)
|
|
for pattern in [r'<img[^>]+(?:src|data-src)\s*=\s*["\']([^"\']+)["\']', r'data-src\s*=\s*["\']([^"\']+)["\']']:
|
|
for match in re.finditer(pattern, html or "", re.IGNORECASE):
|
|
url = html_module.unescape(match.group(1).strip())
|
|
if not url.startswith("http") or url in seen_flat:
|
|
continue
|
|
lowered = url.lower()
|
|
if vehicle_key and vehicle_key in url:
|
|
seen_flat.add(url)
|
|
flat.append(url)
|
|
elif any(token in lowered for token in ["vis.iaai.com", "anvis", "vehicleimage"]):
|
|
if vehicle_key and vehicle_key not in url:
|
|
continue
|
|
seen_flat.add(url)
|
|
flat.append(url)
|
|
if vehicle_key:
|
|
for url in re.findall(r'https?://vis\.iaai\.com[^\s"\'<>]+', html or ""):
|
|
cleaned = html_module.unescape(url)
|
|
if cleaned not in seen_flat and vehicle_key in cleaned:
|
|
seen_flat.add(cleaned)
|
|
flat.append(cleaned)
|
|
filtered: list[str] = []
|
|
for url in flat:
|
|
lowered = url.lower()
|
|
if any(pat in lowered for pat in {"dimensions", "threesixty", "360view", ".js", ".css", ".svg", "/home/", "iframeview"}):
|
|
continue
|
|
if "vis.iaai.com" in lowered and "/resizer" not in lowered:
|
|
continue
|
|
filtered.append(url)
|
|
return filtered
|
|
|
|
@staticmethod
|
|
def _dom_hints(text: str) -> dict[str, Any]:
|
|
lowered = (text or "").lower()
|
|
return {
|
|
"has_buy_now_text": "buy now" in lowered,
|
|
"has_run_and_drive_text": "run & drive" in lowered or "run and drive" in lowered,
|
|
"has_damage_text": "damage" in lowered,
|
|
"has_vin_text": "vin" in lowered,
|
|
"has_title_text": "title" in lowered,
|
|
"has_captcha_text": any(token in lowered for token in ["captcha", "verify you are human", "i am human", "recaptcha", "cloudflare"]),
|
|
"has_antibot_text": any(token in lowered for token in ["incapsula", "access denied", "request unsuccessful", "bot detection"]),
|
|
}
|