#!/usr/bin/env python3 """Authorized scraper for dfimoveis.com.br. Features: - Multiple search URLs - pagina=N pagination - Async bounded HTTP requests with retries and jitter - Listing detail extraction from JSON-LD, OpenGraph and HTML text - Full image URL discovery and concurrent photo downloads - SQLite persistence, change history, raw HTML snapshots and resumability - Playwright (preferring the Patchright driver, if installed, to avoid Cloudflare's CDP-leak detection) driving real Google Chrome by default - Automatic Cloudflare-challenge handling: broader challenge detection, a persistent browser profile that keeps its clearance cookie across runs, and a fallback that reuses the browser session's cookies for plain HTTP and photo requests so they don't have to re-solve the challenge themselves - Optional direct HTTP-only mode Python 3.11+ """ from __future__ import annotations import argparse import asyncio import contextlib import dataclasses import datetime as dt import gzip import hashlib import html import json import logging import mimetypes import os import random import re import sqlite3 import sys import time from pathlib import Path from typing import Any, Iterable from urllib.parse import parse_qsl, urlencode, urljoin, urlparse, urlunparse import requests from bs4 import BeautifulSoup BASE_URL = "https://www.dfimoveis.com.br" DEFAULT_USER_AGENT = "AuthorizedDFImoveisResearchBot/1.0 (+contact@example.com)" LISTING_HREF_RE = re.compile(r"^/imovel/[^?#]+-(\d+)(?:[/?#]|$)", re.I) ABS_LISTING_RE = re.compile(r"https?://(?:www\.)?dfimoveis\.com\.br/imovel/[^\"'<>\\s]+?-(\d+)(?:[/?#]|$)", re.I) IMAGE_URL_RE = re.compile( r"https?://[^\"'<>\\s]+?\.(?:jpe?g|png|webp|avif)(?:\?[^\"'<>\\s]*)?", re.I, ) MONEY_RE = re.compile(r"R\$\s*([\d.]+(?:,\d{1,2})?)") AREA_RE = re.compile(r"([\d.,]+)\s*m[²2]", re.I) NUMERIC_ID_RE = re.compile(r"-(\d+)(?:[/?#]|$)") def now_utc() -> str: return dt.datetime.now(dt.timezone.utc).isoformat(timespec="seconds") def clean_text(value: Any) -> str | None: if value is None: return None text = re.sub(r"\s+", " ", html.unescape(str(value))).strip() return text or None def parse_brl(value: str | None) -> float | None: if not value: return None m = MONEY_RE.search(value) raw = m.group(1) if m else value raw = re.sub(r"[^\d,.]", "", raw) if not raw: return None if "," in raw: raw = raw.replace(".", "").replace(",", ".") else: parts = raw.split(".") if len(parts) > 1 and all(len(p) == 3 for p in parts[1:]): raw = "".join(parts) with contextlib.suppress(ValueError): return float(raw) return None def parse_number(value: str | None) -> float | None: if not value: return None raw = re.sub(r"[^\d,.]", "", value) if not raw: return None if "," in raw: raw = raw.replace(".", "").replace(",", ".") with contextlib.suppress(ValueError): return float(raw) return None def int_near_label(text: str, labels: Iterable[str]) -> int | None: for label in labels: patterns = [ rf"(\d+)\s*{label}", rf"{label}\s*[:\-]?\s*(\d+)", ] for pattern in patterns: m = re.search(pattern, text, re.I) if m: return int(m.group(1)) return None def canonical_url(url: str) -> str: p = urlparse(urljoin(BASE_URL, url)) return urlunparse((p.scheme or "https", p.netloc.lower(), p.path.rstrip("/"), "", p.query, "")) def page_url(search_url: str, page: int) -> str: p = urlparse(search_url) q = dict(parse_qsl(p.query, keep_blank_values=True)) q["pagina"] = str(page) return urlunparse((p.scheme, p.netloc, p.path, p.params, urlencode(q, doseq=True), "")) def listing_id_from_url(url: str) -> str | None: m = NUMERIC_ID_RE.search(urlparse(url).path) return m.group(1) if m else None def sha256_bytes(data: bytes) -> str: return hashlib.sha256(data).hexdigest() def safe_name(value: str) -> str: return re.sub(r"[^a-zA-Z0-9._-]+", "_", value).strip("_")[:120] or "unnamed" @dataclasses.dataclass(slots=True) class Config: searches: list[str] output_dir: Path database: Path max_pages: int = 0 concurrency: int = 2 photo_concurrency: int = 2 delay_min: float = 3.0 delay_max: float = 6.0 cooldown_base: float = 30.0 cooldown_max: float = 300.0 timeout: float = 30.0 retries: int = 4 download_photos: bool = True save_raw_html: bool = True use_playwright: bool = True playwright_headless: bool = False user_agent: str = DEFAULT_USER_AGENT browser_user_agent: str | None = None browser_channel: str = "chrome" attach_chrome: bool = False cdp_url: str = "http://localhost:9222" cloudflare_fallback: bool = True challenge_timeout: float = 180.0 class Database: def __init__(self, path: Path) -> None: path.parent.mkdir(parents=True, exist_ok=True) self.conn = sqlite3.connect(path) self.conn.row_factory = sqlite3.Row self.conn.execute("PRAGMA journal_mode=WAL") self.conn.execute("PRAGMA foreign_keys=ON") self._init_schema() def _init_schema(self) -> None: self.conn.executescript( """ CREATE TABLE IF NOT EXISTS listings ( listing_id TEXT PRIMARY KEY, url TEXT NOT NULL, title TEXT, description TEXT, transaction_type TEXT, property_type TEXT, price_brl REAL, condominium_brl REAL, iptu_brl REAL, area_m2 REAL, bedrooms INTEGER, suites INTEGER, parking_spaces INTEGER, address TEXT, neighborhood TEXT, city TEXT, state TEXT, advertiser_name TEXT, advertiser_code TEXT, creci TEXT, latitude REAL, longitude REAL, published_at TEXT, first_seen_at TEXT NOT NULL, last_seen_at TEXT NOT NULL, inactive_at TEXT, content_hash TEXT, raw_html_path TEXT, extra_json TEXT NOT NULL DEFAULT '{}' ); CREATE TABLE IF NOT EXISTS listing_searches ( listing_id TEXT NOT NULL, search_url TEXT NOT NULL, first_seen_at TEXT NOT NULL, last_seen_at TEXT NOT NULL, PRIMARY KEY (listing_id, search_url), FOREIGN KEY (listing_id) REFERENCES listings(listing_id) ); CREATE TABLE IF NOT EXISTS listing_history ( id INTEGER PRIMARY KEY AUTOINCREMENT, listing_id TEXT NOT NULL, captured_at TEXT NOT NULL, content_hash TEXT NOT NULL, data_json TEXT NOT NULL, UNIQUE(listing_id, content_hash), FOREIGN KEY (listing_id) REFERENCES listings(listing_id) ); CREATE TABLE IF NOT EXISTS photos ( listing_id TEXT NOT NULL, ordinal INTEGER NOT NULL, source_url TEXT NOT NULL, sha256 TEXT, mime_type TEXT, bytes INTEGER, local_path TEXT, first_seen_at TEXT NOT NULL, last_seen_at TEXT NOT NULL, PRIMARY KEY (listing_id, source_url), FOREIGN KEY (listing_id) REFERENCES listings(listing_id) ); CREATE TABLE IF NOT EXISTS crawl_runs ( id INTEGER PRIMARY KEY AUTOINCREMENT, started_at TEXT NOT NULL, finished_at TEXT, status TEXT NOT NULL, stats_json TEXT NOT NULL DEFAULT '{}' ); """ ) self.conn.commit() def start_run(self) -> int: cur = self.conn.execute( "INSERT INTO crawl_runs(started_at,status) VALUES(?,?)", (now_utc(), "running") ) self.conn.commit() return int(cur.lastrowid) def finish_run(self, run_id: int, status: str, stats: dict[str, Any]) -> None: self.conn.execute( "UPDATE crawl_runs SET finished_at=?, status=?, stats_json=? WHERE id=?", (now_utc(), status, json.dumps(stats, ensure_ascii=False), run_id), ) self.conn.commit() def upsert_listing(self, data: dict[str, Any], search_url: str) -> bool: ts = now_utc() existing = self.conn.execute( "SELECT content_hash FROM listings WHERE listing_id=?", (data["listing_id"],) ).fetchone() changed = not existing or existing["content_hash"] != data["content_hash"] cols = [ "listing_id", "url", "title", "description", "transaction_type", "property_type", "price_brl", "condominium_brl", "iptu_brl", "area_m2", "bedrooms", "suites", "parking_spaces", "address", "neighborhood", "city", "state", "advertiser_name", "advertiser_code", "creci", "latitude", "longitude", "published_at", "content_hash", "raw_html_path", "extra_json" ] values = [data.get(c) for c in cols] self.conn.execute( f""" INSERT INTO listings ({','.join(cols)}, first_seen_at, last_seen_at, inactive_at) VALUES ({','.join('?' for _ in cols)}, ?, ?, NULL) ON CONFLICT(listing_id) DO UPDATE SET {','.join(f'{c}=excluded.{c}' for c in cols if c != 'listing_id')}, last_seen_at=excluded.last_seen_at, inactive_at=NULL """, (*values, ts, ts), ) self.conn.execute( """ INSERT INTO listing_searches(listing_id,search_url,first_seen_at,last_seen_at) VALUES(?,?,?,?) ON CONFLICT(listing_id,search_url) DO UPDATE SET last_seen_at=excluded.last_seen_at """, (data["listing_id"], search_url, ts, ts), ) if changed: snapshot = {k: v for k, v in data.items() if k not in {"raw_html_path"}} self.conn.execute( "INSERT OR IGNORE INTO listing_history(listing_id,captured_at,content_hash,data_json) VALUES(?,?,?,?)", (data["listing_id"], ts, data["content_hash"], json.dumps(snapshot, ensure_ascii=False)), ) self.conn.commit() return changed def upsert_photo(self, listing_id: str, ordinal: int, url: str, **meta: Any) -> None: ts = now_utc() self.conn.execute( """ INSERT INTO photos(listing_id,ordinal,source_url,sha256,mime_type,bytes,local_path,first_seen_at,last_seen_at) VALUES(?,?,?,?,?,?,?,?,?) ON CONFLICT(listing_id,source_url) DO UPDATE SET ordinal=excluded.ordinal, sha256=COALESCE(excluded.sha256,photos.sha256), mime_type=COALESCE(excluded.mime_type,photos.mime_type), bytes=COALESCE(excluded.bytes,photos.bytes), local_path=COALESCE(excluded.local_path,photos.local_path), last_seen_at=excluded.last_seen_at """, (listing_id, ordinal, url, meta.get("sha256"), meta.get("mime_type"), meta.get("bytes"), meta.get("local_path"), ts, ts), ) self.conn.commit() def get_photo(self, listing_id: str, url: str) -> sqlite3.Row | None: return self.conn.execute( "SELECT sha256, local_path FROM photos WHERE listing_id=? AND source_url=?", (listing_id, url), ).fetchone() def mark_inactive(self, search_url: str, seen_ids: set[str], run_started_at: str) -> int: rows = self.conn.execute( "SELECT listing_id FROM listing_searches WHERE search_url=? AND last_seen_at < ?", (search_url, run_started_at), ).fetchall() stale = [r[0] for r in rows if r[0] not in seen_ids] if stale: self.conn.executemany( "UPDATE listings SET inactive_at=COALESCE(inactive_at, ?) WHERE listing_id=?", [(now_utc(), x) for x in stale], ) self.conn.commit() return len(stale) def close(self) -> None: self.conn.close() @dataclasses.dataclass(slots=True) class _BrowserResponse: """Duck-types the subset of requests.Response used elsewhere in this file, so a fetch made through the Playwright browser context (see Fetcher._resolve_via_browser) can be returned interchangeably with a plain requests.Session response.""" status_code: int headers: requests.structures.CaseInsensitiveDict text: str content: bytes url: str def raise_for_status(self) -> None: if self.status_code >= 400: raise requests.exceptions.HTTPError(f"HTTP {self.status_code}", response=self) class Fetcher: def __init__(self, cfg: Config) -> None: self.cfg = cfg self.sem = asyncio.Semaphore(cfg.concurrency) self.photo_sem = asyncio.Semaphore(cfg.photo_concurrency) self._playwright = None self._browser_context = None self._owns_browser_context = True self._page = None self._browser_lock = asyncio.Lock() self._rate_lock = asyncio.Lock() self._next_request_at = 0.0 self._cooldown_until = 0.0 self._consecutive_429 = 0 # requests.Session is synchronous, so blocking calls are pushed onto # worker threads with asyncio.to_thread below. self.session = requests.Session() self.session.headers.update( { "User-Agent": cfg.user_agent, "Accept-Language": "pt-BR,pt;q=0.9,en;q=0.7", "Accept": "text/html,application/xhtml+xml,application/json;q=0.9,*/*;q=0.8", # A closer match to a real browser's request fingerprint makes # Cloudflare's heuristic (non-challenge) checks less likely to fire. "Upgrade-Insecure-Requests": "1", "Sec-Fetch-Dest": "document", "Sec-Fetch-Mode": "navigate", "Sec-Fetch-Site": "none", "Sec-Fetch-User": "?1", } ) async def close(self) -> None: if self._browser_context is not None: if self._owns_browser_context: try: await self._browser_context.close() except Exception as exc: # The operator may close Chrome while handling a Cloudflare # challenge, or the browser process may already have exited. # Cleanup must not hide the exception that caused the scraper # to stop. logging.debug("Browser context was already closed: %s", exc) else: # Attached via --attach-chrome to a Chrome window the operator # started themselves; closing it here would be surprising and # unwanted, so just detach the Playwright client. logging.info("Leaving the attached Chrome window open.") self._browser_context = None self._page = None if self._playwright is not None: try: await self._playwright.stop() except Exception as exc: logging.debug("Playwright driver was already stopped: %s", exc) self._playwright = None await asyncio.to_thread(self.session.close) async def _wait_for_request_slot(self) -> None: """Space request starts globally, including concurrent photo downloads.""" loop = asyncio.get_running_loop() while True: async with self._rate_lock: now = loop.time() wait = max(self._next_request_at, self._cooldown_until) - now if wait <= 0: self._next_request_at = now + random.uniform( self.cfg.delay_min, self.cfg.delay_max ) return await asyncio.sleep(wait) async def _register_429(self, retry_after: float) -> float: """Apply one shared exponential cooldown so queued tasks also slow down.""" async with self._rate_lock: self._consecutive_429 += 1 exponential = self.cfg.cooldown_base * (2 ** min(self._consecutive_429 - 1, 3)) cooldown = max( retry_after, min( self.cfg.cooldown_max, exponential + random.uniform(0, self.cfg.cooldown_base), ), ) self._cooldown_until = max( self._cooldown_until, asyncio.get_running_loop().time() + cooldown ) return cooldown async def _register_success(self) -> None: async with self._rate_lock: if asyncio.get_running_loop().time() >= self._cooldown_until: self._consecutive_429 = max(0, self._consecutive_429 - 1) async def get(self, url: str, *, photo: bool = False) -> requests.Response: sem = self.photo_sem if photo else self.sem async with sem: for attempt in range(self.cfg.retries + 1): try: await self._wait_for_request_slot() response = await asyncio.to_thread( self.session.get, url, timeout=self.cfg.timeout ) if self._looks_like_cloudflare_challenge( response.status_code, response.text, response.headers ): response = await self._resolve_via_browser(url) if response.status_code in {429, 500, 502, 503, 504}: raise requests.exceptions.HTTPError("retryable status", response=response) response.raise_for_status() await self._register_success() return response except ( requests.exceptions.Timeout, requests.exceptions.ConnectionError, requests.exceptions.HTTPError, ) as exc: status_code = exc.response.status_code if exc.response is not None else None retryable = status_code is None or status_code in {429, 500, 502, 503, 504} if not retryable or attempt >= self.cfg.retries: raise retry_after = 0.0 if exc.response is not None: with contextlib.suppress(ValueError, TypeError): retry_after = float(exc.response.headers.get("Retry-After", 0)) if status_code == 429: cooldown = await self._register_429(retry_after) logging.warning( "HTTP 429; pausing all requests for %.1f seconds (attempt %d/%d)", cooldown, attempt + 1, self.cfg.retries + 1, ) continue await asyncio.sleep(max(retry_after, min(60.0, (2**attempt) + random.random()))) raise RuntimeError("unreachable") @staticmethod def _looks_like_cloudflare_challenge( status_code: int, text: str, headers: requests.structures.CaseInsensitiveDict ) -> bool: lowered = text.lower() markers = ( "challenges.cloudflare.com", "just a moment", "cf-chl-", "cf-browser-verification", "turnstile", "attention required", "verify you are human", "checking your browser before accessing", ) return ( status_code in {403, 503} or headers.get("cf-mitigated", "").lower() == "challenge" or any(marker in lowered for marker in markers) ) async def _resolve_via_browser(self, url: str) -> "_BrowserResponse": """Reuse the persistent Edge session to satisfy a Cloudflare challenge. The browser context keeps its own cf_clearance cookies across runs (it is a persistent profile), and once a human has solved a challenge in it, plain API-style requests issued through that same context (no page navigation, no JS execution needed) inherit the clearance. This lets lightweight HTTP-only fetches -- including photo downloads -- ride along on a browser session that has already passed Cloudflare's check, instead of failing outright the moment a challenge is served. """ if not self.cfg.cloudflare_fallback: raise RuntimeError( f"Cloudflare challenge detected fetching {url}. Run without " "--http-only, or drop --no-cloudflare-fallback, to let the browser " "session handle it." ) logging.warning("Cloudflare challenge on %s; retrying through the browser session", url) async with self._browser_lock: context = await self._ensure_browser_context() try: api_response = await context.request.get(url, timeout=self.cfg.timeout * 1000) body = await api_response.body() except Exception as exc: raise RuntimeError( f"Cloudflare challenge fallback failed for {url}: {exc}. If this " "is the first request of the run, open the visible Edge window " "and complete the challenge there, then retry." ) from exc headers = requests.structures.CaseInsensitiveDict(api_response.headers) await self._sync_session_from_browser(context) response = _BrowserResponse( status_code=api_response.status, headers=headers, text=body.decode("utf-8", "replace"), content=body, url=api_response.url, ) if self._looks_like_cloudflare_challenge(response.status_code, response.text, response.headers): raise RuntimeError( f"Cloudflare challenge persisted for {url} even through the browser " "session. Complete it manually in the visible Microsoft Edge window " "(disable --playwright-headless if it's on) and rerun." ) return response async def _sync_session_from_browser(self, context) -> None: """Mirror the browser's cookies (and real UA) onto the plain HTTP session so subsequent lightweight requests are less likely to need the fallback.""" try: cookies = await context.cookies() except Exception as exc: logging.debug("Could not read browser cookies: %s", exc) return for cookie in cookies: with contextlib.suppress(Exception): self.session.cookies.set( cookie["name"], cookie["value"], domain=cookie.get("domain", "") or urlparse(BASE_URL).netloc, path=cookie.get("path", "/"), ) page = context.pages[0] if context.pages else None if page is not None: with contextlib.suppress(Exception): real_ua = await page.evaluate("() => navigator.userAgent") if real_ua: self.session.headers["User-Agent"] = real_ua logging.debug("Synced %d browser cookies onto the HTTP session", len(cookies)) async def get_html(self, url: str) -> tuple[str, bytes, str]: if self.cfg.use_playwright: rendered, final_url = await self.rendered_html(url) return rendered, rendered.encode("utf-8"), final_url response = await self.get(url) return response.text, response.content, str(response.url) async def _ensure_browser_context(self): if self._browser_context is not None: return self._browser_context using_patchright = True try: # Patchright is a maintained, drop-in fork of Playwright's async API. # Plain Playwright issues the CDP command Runtime.enable to manage JS # execution contexts; Cloudflare's bot management specifically checks # for that call, which is why a challenge can clear once and then # start looping again on a later page. Patchright avoids that command # entirely (it runs JS in isolated contexts instead), which is what # closes that gap. Falls back to stock Playwright if it isn't installed. try: from patchright.async_api import async_playwright except ImportError: using_patchright = False logging.warning( "patchright not installed; falling back to plain Playwright, " "which is more likely to trip Cloudflare's CDP-leak detection. " "Install with: pip install patchright" ) from playwright.async_api import async_playwright except ImportError as exc: raise RuntimeError( "Install a Playwright-compatible driver: pip install patchright " "(preferred, avoids Cloudflare's CDP-leak detection) or pip install " "playwright." ) from exc self._playwright = await async_playwright().start() if self.cfg.attach_chrome: # Attaches to a Chrome window the operator started themselves (e.g. # `chrome.exe --remote-debugging-port=9222 --user-data-dir=...`) # instead of one Playwright/Patchright spawns. Cloudflare's current # Turnstile check can flag a browser process for having been *launched* # by an automation framework at all, independent of anything done # inside it afterwards -- so even a genuine manual click on the # checkbox won't validate in a Playwright-launched Chrome. Attaching # over CDP to a normally-started Chrome sidesteps that specific signal # (though CDP attachment itself can still be detected by other means). try: browser = await self._playwright.chromium.connect_over_cdp(self.cfg.cdp_url) except Exception as exc: port = urlparse(self.cfg.cdp_url).port or 9222 raise RuntimeError( f"Could not connect to Chrome at {self.cfg.cdp_url}. Start Chrome " "yourself first with a remote debugging port open, e.g.:\n" f' "C:\\Program Files\\Google\\Chrome\\Application\\chrome.exe" ' f'--remote-debugging-port={port} --user-data-dir="C:\\chrome-scrape-profile"\n' "(use a separate --user-data-dir, not your everyday profile, since " "Chrome won't let a second instance share one already in use) " "then re-run with --attach-chrome." ) from exc self._owns_browser_context = False context = browser.contexts[0] if browser.contexts else await browser.new_context() self._browser_context = context await self._sync_session_from_browser(context) return context profile_dir = self.cfg.output_dir / "browser-profile" profile_dir.mkdir(parents=True, exist_ok=True) browser_options: dict[str, Any] = { "user_data_dir": str(profile_dir), "channel": self.cfg.browser_channel, "headless": self.cfg.playwright_headless, "locale": "pt-BR", "timezone_id": "America/Sao_Paulo", # Patchright's own guidance: a fixed viewport size is itself a # fingerprintable tell (it never matches a real, resized browser # window). no_viewport lets the actual window size drive it instead. "no_viewport": True, } if using_patchright: # Patchright's docs are explicit: don't add custom UA/headers or extra # flags on top of it -- real Chrome's own defaults are more convincing # than anything we'd hardcode, and layering our own JS patches on top # of Patchright's isolated-context patches can reintroduce the very # inconsistencies it's designed to avoid. if self.cfg.browser_user_agent: logging.warning( "--browser-user-agent is set but patchright is active; patchright " "recommends leaving Chrome's own user agent untouched for the " "most convincing fingerprint." ) browser_options["user_agent"] = self.cfg.browser_user_agent else: # No Patchright available: fall back to the manual mitigations, since # stock Playwright needs them (imperfect, but better than nothing). browser_options["args"] = ["--disable-blink-features=AutomationControlled"] browser_options["ignore_default_args"] = ["--enable-automation"] if self.cfg.browser_user_agent: browser_options["user_agent"] = self.cfg.browser_user_agent self._browser_context = await self._playwright.chromium.launch_persistent_context( **browser_options, ) if not using_patchright: # Belt-and-braces JS patch for the handful of properties bot-detection # scripts check most often. Only needed without Patchright's deeper, # driver-level patches -- redundant (and possibly counter-productive) # alongside them. await self._browser_context.add_init_script( """ Object.defineProperty(navigator, 'webdriver', { get: () => undefined }); window.chrome = window.chrome || { runtime: {} }; Object.defineProperty(navigator, 'languages', { get: () => ['pt-BR', 'pt', 'en-US', 'en'] }); Object.defineProperty(navigator, 'plugins', { get: () => [1, 2, 3, 4, 5] }); const originalQuery = window.navigator.permissions.query; window.navigator.permissions.query = (parameters) => ( parameters.name === 'notifications' ? Promise.resolve({ state: Notification.permission }) : originalQuery(parameters) ); """ ) # The persistent profile may already carry a cf_clearance cookie from an # earlier run; make it available to the plain HTTP session right away. await self._sync_session_from_browser(self._browser_context) return self._browser_context async def rendered_html(self, url: str) -> tuple[str, str]: # Serialize browser navigation so one persistent Cloudflare session is reused safely. async with self._browser_lock: context = await self._ensure_browser_context() if self._page is None or self._page.is_closed(): # Reuse the persistent context's first (blank) tab if we launched # it ourselves; open a fresh tab if we attached to a Chrome window # the operator already had open, so we don't hijack one of their tabs. if self._owns_browser_context and context.pages: self._page = context.pages[0] else: self._page = await context.new_page() page = self._page # Page navigations previously bypassed the rate limiter entirely, so # they fired back-to-back with no pacing -- a strong behavioral tell # on top of anything fingerprint-related, and a likely reason a # challenge would clear once and then reappear on the very next page. await self._wait_for_request_slot() await page.goto( url, wait_until="domcontentloaded", timeout=int(self.cfg.timeout * 1000), ) try: await page.wait_for_load_state( "networkidle", timeout=int(self.cfg.timeout * 1000) ) except Exception: logging.debug("Browser did not reach networkidle for %s", url) with contextlib.suppress(Exception): # A small human-like gesture; real visitors rarely leave the # mouse at (0, 0) the instant a page finishes loading. await page.mouse.move( random.uniform(100, 800), random.uniform(100, 600), steps=random.randint(5, 15) ) # A visible browser lets the operator solve a Cloudflare challenge manually. challenge_markers = ( "challenges.cloudflare.com", "just a moment", "cf-chl-", "cf-browser-verification", "turnstile", "attention required", "verify you are human", "checking your browser before accessing", ) deadline = asyncio.get_running_loop().time() + self.cfg.challenge_timeout while True: try: content = await page.content() except Exception as exc: if exc.__class__.__name__ in {"TargetClosedError", "Error"}: raise RuntimeError( "Microsoft Edge closed while the scraper was waiting for the " "Cloudflare verification. Keep the browser window open, complete " "the verification, and let the scraper close Edge when it finishes." ) from exc raise lowered = content.lower() if not any(marker in lowered for marker in challenge_markers): # Solved (or no challenge was served) -- let plain HTTP requests # (search pages in --http-only mode, photo downloads) ride on # this session's clearance too. await self._sync_session_from_browser(context) return content, page.url if self.cfg.playwright_headless: raise RuntimeError( "Cloudflare challenge persisted in headless mode. " "Run without --playwright-headless and solve it in the browser window." ) if asyncio.get_running_loop().time() >= deadline: raise RuntimeError( f"Cloudflare challenge was not completed in the browser window within " f"{self.cfg.challenge_timeout:.0f}s. Re-run, or raise --challenge-timeout " "if you need more time to solve it manually." ) logging.warning("Waiting for Cloudflare challenge to be completed in the browser...") await page.wait_for_timeout(2000) class Parser: @staticmethod def search_links(page_html: str, page_base: str) -> list[str]: soup = BeautifulSoup(page_html, "lxml") out: dict[str, None] = {} for tag in soup.select("a[href]"): href = tag.get("href", "") absolute = canonical_url(urljoin(page_base, href)) if listing_id_from_url(absolute) and "/imovel/" in urlparse(absolute).path: out[absolute] = None for match in ABS_LISTING_RE.finditer(page_html): out[canonical_url(match.group(0))] = None return list(out) @staticmethod def json_ld(soup: BeautifulSoup) -> list[dict[str, Any]]: objects: list[dict[str, Any]] = [] for node in soup.select('script[type="application/ld+json"]'): raw = node.string or node.get_text() try: parsed = json.loads(raw) except (json.JSONDecodeError, TypeError): continue candidates = parsed if isinstance(parsed, list) else [parsed] for item in candidates: if isinstance(item, dict) and isinstance(item.get("@graph"), list): candidates.extend(x for x in item["@graph"] if isinstance(x, dict)) if isinstance(item, dict): objects.append(item) return objects @staticmethod def _meta(soup: BeautifulSoup, *names: str) -> str | None: for name in names: node = soup.find("meta", attrs={"property": name}) or soup.find("meta", attrs={"name": name}) if node and node.get("content"): return clean_text(node["content"]) return None @classmethod def detail(cls, url: str, page_html: str, raw_path: str | None = None) -> tuple[dict[str, Any], list[str]]: soup = BeautifulSoup(page_html, "lxml") full_text = clean_text(soup.get_text(" ", strip=True)) or "" listing_id = listing_id_from_url(url) if not listing_id: raise ValueError(f"Cannot determine listing id from {url}") ld = cls.json_ld(soup) primary = next((x for x in ld if str(x.get("@type", "")).lower() in { "product", "realestatelisting", "apartment", "house", "residence", "singlefamilyresidence" }), ld[0] if ld else {}) title = clean_text(primary.get("name")) or cls._meta(soup, "og:title", "twitter:title") if not title and soup.title: title = clean_text(soup.title.get_text()) description = clean_text(primary.get("description")) or cls._meta(soup, "og:description", "description") offers = primary.get("offers") if isinstance(primary.get("offers"), dict) else {} price = parse_brl(str(offers.get("price", ""))) if price is None: price = parse_brl(cls._meta(soup, "product:price:amount")) if price is None: # Prefer visible price blocks, then first BRL occurrence. price_nodes = soup.select('[class*="preco" i], [class*="price" i], [itemprop="price"]') for node in price_nodes: price = parse_brl(node.get("content") or node.get_text(" ", strip=True)) if price is not None: break address_obj = primary.get("address") if isinstance(primary.get("address"), dict) else {} geo = primary.get("geo") if isinstance(primary.get("geo"), dict) else {} address = clean_text(address_obj.get("streetAddress")) city = clean_text(address_obj.get("addressLocality")) state = clean_text(address_obj.get("addressRegion")) area = None for pattern in [r"(?:área(?:\s+privativa|\s+útil)?|area)\s*[:\-]?\s*([\d.,]+)\s*m[²2]", r"([\d.,]+)\s*m[²2]"]: m = re.search(pattern, full_text, re.I) if m: area = parse_number(m.group(1)) break def label_money(*labels: str) -> float | None: for label in labels: m = re.search(rf"{label}\s*[:\-]?\s*(R\$\s*[\d.]+(?:,\d{{1,2}})?)", full_text, re.I) if m: return parse_brl(m.group(1)) return None def label_text(*labels: str) -> str | None: for label in labels: m = re.search(rf"{label}\s*[:\-]?\s*([^|•]+?)(?=\s{{2,}}|\b(?:Código|Creci|Condomínio|IPTU)\b|$)", full_text, re.I) if m: return clean_text(m.group(1)) return None path_parts = [x for x in urlparse(url).path.split("/") if x] slug = path_parts[-1] if path_parts else "" transaction_type = "venda" if "venda" in slug else "aluguel" if "aluguel" in slug else None property_type = next((x for x in ["apartamento", "casa", "cobertura", "terreno", "loja", "sala", "kitnet", "galpao"] if x in slug), None) image_urls: dict[str, None] = {} candidates: list[Any] = [primary.get("image"), cls._meta(soup, "og:image", "twitter:image")] for obj in ld: candidates.append(obj.get("image")) for candidate in candidates: vals = candidate if isinstance(candidate, list) else [candidate] for val in vals: if isinstance(val, dict): val = val.get("url") or val.get("contentUrl") if isinstance(val, str) and val.startswith(("http://", "https://")): image_urls[html.unescape(val)] = None for tag in soup.select("img, source"): for attr in ("src", "data-src", "data-lazy", "data-original", "data-zoom-image", "srcset", "data-srcset"): raw = tag.get(attr) if not raw: continue for part in str(raw).split(","): candidate = part.strip().split(" ")[0] if candidate and not candidate.startswith("data:"): absolute = urljoin(url, candidate) if re.search(r"\.(?:jpe?g|png|webp|avif)(?:\?|$)", absolute, re.I): image_urls[absolute] = None for match in IMAGE_URL_RE.finditer(page_html.replace("\\/", "/")): image_urls[html.unescape(match.group(0))] = None # Exclude common site chrome and tiny assets based on URL hints. photos = [u for u in image_urls if not re.search(r"logo|favicon|sprite|icon|avatar|banner", u, re.I)] data = { "listing_id": listing_id, "url": canonical_url(url), "title": title, "description": description, "transaction_type": transaction_type, "property_type": property_type, "price_brl": price, "condominium_brl": label_money("Condomínio", "Condominio"), "iptu_brl": label_money("IPTU"), "area_m2": area, "bedrooms": int_near_label(full_text, ["quartos?", "dormitórios?"]), "suites": int_near_label(full_text, ["suítes?", "suites?"]), "parking_spaces": int_near_label(full_text, ["vagas?"]), "address": address or label_text("Endereço", "Endereco"), "neighborhood": clean_text(address_obj.get("addressSubregion")) or label_text("Bairro"), "city": city, "state": state, "advertiser_name": clean_text(primary.get("seller", {}).get("name")) if isinstance(primary.get("seller"), dict) else None, "advertiser_code": label_text("Código", "Codigo"), "creci": label_text("CRECI"), "latitude": parse_number(str(geo.get("latitude"))) if geo.get("latitude") is not None else None, "longitude": parse_number(str(geo.get("longitude"))) if geo.get("longitude") is not None else None, "published_at": clean_text(primary.get("datePosted") or primary.get("datePublished")), "raw_html_path": raw_path, "extra_json": json.dumps({"json_ld": ld}, ensure_ascii=False), } stable = {k: v for k, v in data.items() if k not in {"content_hash", "raw_html_path"}} data["content_hash"] = hashlib.sha256(json.dumps(stable, sort_keys=True, ensure_ascii=False).encode()).hexdigest() return data, photos class Scraper: def __init__(self, cfg: Config) -> None: self.cfg = cfg self.db = Database(cfg.database) self.fetcher = Fetcher(cfg) self.stats = {"search_pages": 0, "listing_urls": 0, "details_ok": 0, "details_failed": 0, "changed": 0, "photos_ok": 0, "photos_skipped": 0, "photos_failed": 0, "inactive": 0} self.run_started_at = now_utc() def save_raw(self, kind: str, name: str, content: bytes) -> str: stamp = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%SZ") path = self.cfg.output_dir / "raw" / kind / safe_name(name) / f"{stamp}.html.gz" path.parent.mkdir(parents=True, exist_ok=True) with gzip.open(path, "wb", compresslevel=6) as f: f.write(content) return str(path.relative_to(self.cfg.output_dir)) async def discover(self, search_url: str) -> list[str]: found: dict[str, None] = {} page = 1 consecutive_empty = 0 while True: if self.cfg.max_pages and page > self.cfg.max_pages: break url = page_url(search_url, page) logging.info("Search page %d: %s", page, url) page_html, body, final_url = await self.fetcher.get_html(url) links = Parser.search_links(page_html, final_url) self.stats["search_pages"] += 1 if self.cfg.save_raw_html: self.save_raw("search", f"{hashlib.sha1(search_url.encode()).hexdigest()[:12]}_p{page}", body) new = [x for x in links if x not in found] logging.info("Found %d links (%d new)", len(links), len(new)) for x in new: found[x] = None consecutive_empty = consecutive_empty + 1 if not new else 0 if not links or consecutive_empty >= 2: break page += 1 self.stats["listing_urls"] += len(found) return list(found) async def scrape_detail(self, url: str, search_url: str) -> None: try: page_html, body, final_url = await self.fetcher.get_html(url) raw_path = self.save_raw("listing", listing_id_from_url(url) or "unknown", body) if self.cfg.save_raw_html else None data, photos = Parser.detail(final_url, page_html, raw_path) changed = self.db.upsert_listing(data, search_url) self.stats["details_ok"] += 1 self.stats["changed"] += int(changed) logging.info("Listing %s: %s; %d photos", data["listing_id"], "changed" if changed else "unchanged", len(photos)) for ordinal, photo_url in enumerate(photos, start=1): self.db.upsert_photo(data["listing_id"], ordinal, photo_url) if self.cfg.download_photos: await asyncio.gather(*(self.download_photo(data["listing_id"], i, u) for i, u in enumerate(photos, 1))) except Exception: self.stats["details_failed"] += 1 logging.exception("Failed listing: %s", url) async def download_photo(self, listing_id: str, ordinal: int, url: str) -> None: try: existing = self.db.get_photo(listing_id, url) if existing and existing["sha256"] and existing["local_path"]: local_path = self.cfg.output_dir / existing["local_path"] if local_path.is_file(): self.stats["photos_skipped"] += 1 logging.debug("Reusing photo for %s: %s", listing_id, local_path) return response = await self.fetcher.get(url, photo=True) content_type = response.headers.get("content-type", "").split(";", 1)[0].lower() if not content_type.startswith("image/"): raise ValueError(f"Not an image: {content_type}") content = response.content digest = sha256_bytes(content) suffix = mimetypes.guess_extension(content_type) or Path(urlparse(url).path).suffix or ".img" if suffix == ".jpe": suffix = ".jpg" path = self.cfg.output_dir / "photos" / listing_id / f"{ordinal:03d}_{digest[:16]}{suffix}" path.parent.mkdir(parents=True, exist_ok=True) if not path.exists(): path.write_bytes(content) self.db.upsert_photo(listing_id, ordinal, url, sha256=digest, mime_type=content_type, bytes=len(content), local_path=str(path.relative_to(self.cfg.output_dir))) self.stats["photos_ok"] += 1 except Exception: self.stats["photos_failed"] += 1 logging.exception("Failed photo for %s: %s", listing_id, url) async def run(self) -> dict[str, Any]: run_id = self.db.start_run() status = "ok" try: for search_url in self.cfg.searches: urls = await self.discover(search_url) seen_ids = {x for u in urls if (x := listing_id_from_url(u))} # Detail concurrency remains bounded inside Fetcher. await asyncio.gather(*(self.scrape_detail(url, search_url) for url in urls)) self.stats["inactive"] += self.db.mark_inactive(search_url, seen_ids, self.run_started_at) except Exception: status = "failed" raise finally: self.db.finish_run(run_id, status, self.stats) await self.fetcher.close() self.db.close() return self.stats def load_searches(args: argparse.Namespace) -> list[str]: searches = list(args.search or []) if args.search_file: for line in Path(args.search_file).read_text(encoding="utf-8").splitlines(): line = line.strip() if line and not line.startswith("#"): searches.append(line) normalized = [] for url in searches: if not urlparse(url).netloc: url = urljoin(BASE_URL, url) normalized.append(canonical_url(url)) return list(dict.fromkeys(normalized)) def build_arg_parser() -> argparse.ArgumentParser: p = argparse.ArgumentParser(description="Authorized DFImoveis scraper") p.add_argument("--search", action="append", help="Search URL; repeat for multiple searches") p.add_argument("--search-file", help="Text file containing one search URL per line") p.add_argument("--output", default="dfimoveis_data", help="Output directory") p.add_argument("--database", help="SQLite path; defaults to OUTPUT/dfimoveis.sqlite3") p.add_argument("--max-pages", type=int, default=0, help="0 means continue until exhausted") p.add_argument("--concurrency", type=int, default=2) p.add_argument("--photo-concurrency", type=int, default=2) p.add_argument("--delay-min", type=float, default=3.0) p.add_argument("--delay-max", type=float, default=6.0) p.add_argument("--cooldown-base", type=float, default=30.0) p.add_argument("--cooldown-max", type=float, default=300.0) p.add_argument("--timeout", type=float, default=30.0) p.add_argument("--retries", type=int, default=4) p.add_argument("--no-photos", action="store_true") p.add_argument("--no-raw-html", action="store_true") p.add_argument( "--http-only", action="store_true", help="Use direct HTTP requests instead of the default local Microsoft Edge browser", ) p.add_argument( "--playwright-headless", action="store_true", help="Run Playwright without a visible browser (not recommended for Cloudflare challenges)", ) p.add_argument( "--no-cloudflare-fallback", action="store_true", help=( "Disable the automatic fallback that retries a challenged HTTP-only or " "photo request through the persistent browser session" ), ) p.add_argument( "--challenge-timeout", type=float, default=180.0, help="Seconds to wait for a Cloudflare challenge to be solved in the visible browser window", ) p.add_argument( "--attach-chrome", action="store_true", help=( "Attach to a Chrome window you started yourself (via --remote-debugging-port) " "instead of one this script launches. Needed when Cloudflare's Turnstile flags " "automation-launched Chrome even after a genuine manual click; see --cdp-url." ), ) p.add_argument( "--cdp-url", default=os.getenv("DFIMOVEIS_CDP_URL", "http://localhost:9222"), help="CDP endpoint of the already-running Chrome (used with --attach-chrome)", ) p.add_argument("--user-agent", default=os.getenv("DFIMOVEIS_USER_AGENT", DEFAULT_USER_AGENT)) p.add_argument( "--browser-channel", default=os.getenv("DFIMOVEIS_BROWSER_CHANNEL", "chrome"), choices=["chrome", "msedge", "chromium", "chrome-beta", "msedge-beta", "msedge-dev"], help=( "Which installed browser Playwright/Patchright should drive. 'chrome' " "(the default) is what Patchright's own guidance recommends -- the " "bundled 'chromium' binary is a stealth downgrade, not an upgrade." ), ) p.add_argument( "--browser-user-agent", default=os.getenv("DFIMOVEIS_BROWSER_USER_AGENT"), help=( "Override the browser's native user agent. Not recommended when " "patchright is installed -- its own default is more convincing than " "any override." ), ) p.add_argument("--log-level", default="INFO", choices=["DEBUG", "INFO", "WARNING", "ERROR"]) return p async def async_main() -> int: args = build_arg_parser().parse_args() logging.basicConfig(level=getattr(logging, args.log_level), format="%(asctime)s %(levelname)s %(message)s") searches = load_searches(args) if not searches: print("Provide at least one --search URL or --search-file", file=sys.stderr) return 2 output = Path(args.output).expanduser().resolve() output.mkdir(parents=True, exist_ok=True) cfg = Config( searches=searches, output_dir=output, database=Path(args.database).expanduser().resolve() if args.database else output / "dfimoveis.sqlite3", max_pages=max(0, args.max_pages), concurrency=max(1, args.concurrency), photo_concurrency=max(1, args.photo_concurrency), delay_min=max(0, args.delay_min), delay_max=max(0, args.delay_min, args.delay_max), cooldown_base=max(1, args.cooldown_base), cooldown_max=max(1, args.cooldown_base, args.cooldown_max), timeout=max(1, args.timeout), retries=max(0, args.retries), download_photos=not args.no_photos, save_raw_html=not args.no_raw_html, use_playwright=not args.http_only, playwright_headless=args.playwright_headless, user_agent=args.user_agent, browser_user_agent=args.browser_user_agent, browser_channel=args.browser_channel, attach_chrome=args.attach_chrome, cdp_url=args.cdp_url, cloudflare_fallback=not args.no_cloudflare_fallback, challenge_timeout=max(10.0, args.challenge_timeout), ) started = time.monotonic() scraper = Scraper(cfg) stats = await scraper.run() stats["elapsed_seconds"] = round(time.monotonic() - started, 2) print(json.dumps(stats, ensure_ascii=False, indent=2)) return 0 def main() -> None: try: raise SystemExit(asyncio.run(async_main())) except KeyboardInterrupt: raise SystemExit(130) if __name__ == "__main__": main()