Compare commits

..
2 Commits
Author SHA1 Message Date
qananasikq 1270d488c9 Update export script 2026-07-01 14:41:04 +03:00
qananasikq 3cd7e785da fix car mapping 2026-07-01 13:56:00 +03:00
3 changed files with 0 additions and 95 deletions
-19
View File
@@ -1,19 +0,0 @@
select 'seen_1h' as metric, count(*)::text as value from cars where last_seen_at >= now() - interval '1 hour'
union all
select 'seen_24h', count(*)::text from cars where last_seen_at >= now() - interval '24 hours'
union all
select 'old_rows_touched_24h', count(*)::text from cars where last_seen_at >= now() - interval '24 hours' and id <= (select greatest(max(id) - 5000, 0) from cars)
union all
select 'min_recent_id_1h', coalesce(min(id)::text, 'null') from cars where last_seen_at >= now() - interval '1 hour'
union all
select 'max_recent_id_1h', coalesce(max(id)::text, 'null') from cars where last_seen_at >= now() - interval '1 hour'
union all
select 'top_5_recent_old_ids_24h', coalesce(string_agg(id::text, ', ' order by last_seen_at desc), 'none')
from (
select id, last_seen_at
from cars
where last_seen_at >= now() - interval '24 hours'
and id <= (select greatest(max(id) - 5000, 0) from cars)
order by last_seen_at desc
limit 5
) t;
-30
View File
@@ -1,30 +0,0 @@
from playwright.sync_api import sync_playwright
import re
url = "https://www.dubizzle.com/Vehiclelisting/Cars?Make=FORD"
with sync_playwright() as p:
browser = p.chromium.launch(headless=True, args=["--headless=new"])
page = browser.new_page()
page.goto(url, wait_until="commit", timeout=60000)
try:
page.wait_for_load_state("domcontentloaded", timeout=5000)
except Exception:
pass
title = page.title()
text = (page.evaluate("() => document.body ? document.body.innerText : ''") or "")
html = page.content()
combined = (text + "\n" + html).lower()
print("TITLE=", title)
print("URL=", page.url)
print("HAS_INCAPSULA=", "incapsula" in combined)
print("HAS_CAPTCHA=", "captcha" in combined)
print("HAS_CHALLENGE=", "challenge" in combined)
print("HAS_VEHICLEDETAIL=", "/vehicledetail/" in combined)
print("HAS_MOTORS_USED_CARS=", "/motors/used-cars/" in combined)
print("TEXT_PREVIEW=", re.sub(r"\s+", " ", text)[:1000])
browser.close()
-46
View File
@@ -1,46 +0,0 @@
from playwright.sync_api import sync_playwright
from dubizzle_scraper.scraper import DUBIZZLEScraper
from dubizzle_scraper.parsing.parser import VehicleParser
import json
import re
url = 'https://www.dubizzle.com/VehicleDetail/45394480~US'
out = {}
with sync_playwright() as p:
browser = p.chromium.launch(headless=True)
page = browser.new_page()
page.goto(url, wait_until='commit', timeout=60000)
try:
page.wait_for_load_state('domcontentloaded', timeout=5000)
except Exception:
pass
try:
page.wait_for_selector("#VehicleDetailViewModel, .veh-details, .vehicle-details, [data-uname='vehicleDetailPage']", timeout=1000)
except Exception:
pass
html = page.content()
try:
dom_text = page.evaluate("() => document.body?.textContent || ''")
except Exception:
dom_text = ''
title = page.title()
js_data = {}
try:
js_data = page.evaluate(DUBIZZLEScraper._JS_EXTRACT) or {}
except Exception:
js_data = {'eval_error': True}
hints = VehicleParser._dom_hints(dom_text)
out = {
'final_url': page.url,
'title': title,
'html_len': len(html),
'dom_text_len': len(dom_text),
'selector_present': any(token in html for token in ['VehicleDetailViewModel', 'veh-details', 'vehicle-details', 'vehicleDetailPage']),
'js_ok': bool(js_data.get('ok')),
'js_keys': sorted(list(js_data.keys()))[:20],
'dom_hints': hints,
'dom_text_preview': re.sub(r'\s+', ' ', dom_text)[:1500],
'html_preview': re.sub(r'\s+', ' ', html)[:2000],
}
browser.close()
print(json.dumps(out, ensure_ascii=False))