#!/usr/bin/env python3 """ Scraper Yapo.cl – Bienes Raíces (Chile / Schibsted clasificados) Portal : https://www.yapo.cl DB : yapo_cl """ import os import re import time import logging from datetime import datetime import requests from bs4 import BeautifulSoup import psycopg2 from psycopg2.extras import execute_values # --------------------------------------------------------------------------- # CONFIG # --------------------------------------------------------------------------- PROXY = 'socks5h://127.0.0.1:1090' PROXIES = {'http': PROXY, 'https': PROXY} DB_HOST = '100.75.240.87' DB_PORT = 5432 DB_NAME = 'yapo_cl' DB_USER = 'pgadmin' DB_PASS = 'J5BVlq65JvedWxZVrcY96OQX' UA = ( 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) ' 'AppleWebKit/537.36 (KHTML, like Gecko) ' 'Chrome/124.0.0.0 Safari/537.36' ) DELAY = 0.8 # segundos entre páginas MAX_PAGES = 30 # límite de páginas NEW_STREAK = 2 # páginas consecutivas sin nuevos para detener BASE_URL = 'https://www.yapo.cl' # Patrones de URL en orden de preferencia URL_PATTERNS = [ BASE_URL + '/bienes_raices/departamentos/{n}.html', BASE_URL + '/bienes_raices/{n}.html', BASE_URL + '/region_metropolitana/bienes_raices.{n}.html', BASE_URL + '/bienes_raices.{n}.html', ] # --------------------------------------------------------------------------- # LOGGING # --------------------------------------------------------------------------- logging.basicConfig( level=logging.INFO, format='%(asctime)s [%(levelname)s] %(message)s', datefmt='%Y-%m-%d %H:%M:%S', ) log = logging.getLogger(__name__) # --------------------------------------------------------------------------- # SESSION # --------------------------------------------------------------------------- def get_session(): """Crea y devuelve una requests.Session con proxy y headers configurados.""" sess = requests.Session() sess.proxies.update(PROXIES) sess.headers.update({ 'User-Agent': UA, 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8', 'Accept-Language': 'es-CL,es;q=0.9,en;q=0.8', 'Accept-Encoding': 'gzip, deflate, br', 'Connection': 'keep-alive', 'Cache-Control': 'no-cache', 'DNT': '1', }) return sess # --------------------------------------------------------------------------- # BASE DE DATOS # --------------------------------------------------------------------------- def create_table(conn): """Crea la tabla listings si no existe y aplica migraciones de columnas.""" cur = conn.cursor() cur.execute(""" CREATE TABLE IF NOT EXISTS listings ( listing_id TEXT UNIQUE NOT NULL, titulo TEXT, precio BIGINT, moneda TEXT DEFAULT 'CLP', tipo TEXT, ubicacion TEXT, url TEXT UNIQUE, ambientes INT, superficie_m2 DECIMAL, scraped_at TIMESTAMP DEFAULT NOW() ) """) # Migraciones: añadir columnas que pudieran faltar en tablas ya existentes migration_cols = [ ('titulo', 'TEXT'), ('precio', 'BIGINT'), ('moneda', "TEXT DEFAULT 'CLP'"), ('tipo', 'TEXT'), ('ubicacion', 'TEXT'), ('url', 'TEXT'), ('ambientes', 'INT'), ('superficie_m2', 'DECIMAL'), ('scraped_at', 'TIMESTAMP DEFAULT NOW()'), ] for col, col_type in migration_cols: try: cur.execute( 'ALTER TABLE listings ADD COLUMN IF NOT EXISTS {} {}'.format(col, col_type) ) except Exception as exc: conn.rollback() log.debug('ALTER TABLE %s ignorado: %s', col, exc) conn.commit() cur.close() log.info('Tabla listings verificada/creada OK') def upsert(conn, rows): """ Inserta o actualiza filas en listings. Devuelve (nuevos, actualizados). """ if not rows: return 0, 0 cur = conn.cursor() # Determinar qué listing_ids ya existen para contar nuevos vs actualizados ids = [r['listing_id'] for r in rows] cur.execute('SELECT listing_id FROM listings WHERE listing_id = ANY(%s)', (ids,)) existing = {row[0] for row in cur.fetchall()} nuevos = sum(1 for r in rows if r['listing_id'] not in existing) actualizados = len(rows) - nuevos now = datetime.now() tuples = [ ( r['listing_id'], r.get('titulo'), r.get('precio'), r.get('moneda', 'CLP'), r.get('tipo'), r.get('ubicacion'), r.get('url'), r.get('ambientes'), r.get('superficie_m2'), now, ) for r in rows ] sql = """ INSERT INTO listings (listing_id, titulo, precio, moneda, tipo, ubicacion, url, ambientes, superficie_m2, scraped_at) VALUES %s ON CONFLICT (listing_id) DO UPDATE SET titulo = EXCLUDED.titulo, precio = EXCLUDED.precio, moneda = EXCLUDED.moneda, tipo = EXCLUDED.tipo, ubicacion = EXCLUDED.ubicacion, url = EXCLUDED.url, ambientes = EXCLUDED.ambientes, superficie_m2 = EXCLUDED.superficie_m2, scraped_at = EXCLUDED.scraped_at """ execute_values(cur, sql, tuples) conn.commit() cur.close() return nuevos, actualizados # --------------------------------------------------------------------------- # HELPERS DE PARSING # --------------------------------------------------------------------------- def _detect_tipo(text): """Infiere 'venta' o 'arriendo' desde texto.""" tl = text.lower() if any(k in tl for k in ('arriendo', 'arrendar', 'alquiler', 'alquilar', 'renta')): return 'arriendo' if any(k in tl for k in ('venta', 'vender', 'en venta')): return 'venta' return None def _clean_precio(text): """Extrae un entero de precio desde un string con formato CLP.""" nums = re.sub(r'[^\d]', '', text) if nums and len(nums) >= 3: try: return int(nums) except ValueError: pass return None def _parse_card(card): """ Parsea un elemento HTML de anuncio y devuelve un dict o None si no es válido. Robusto: usa .get() y try/except por cada campo. """ # --- URL y listing_id --- link = card.find('a', href=True) if not link: return None href = link.get('href', '') if not href: return None full_url = href if href.startswith('http') else BASE_URL + href # ID numérico desde la URL (mínimo 5 dígitos) m = re.search(r'/(\d{5,})', href) if not m: m = re.search(r'[_\-](\d{4,})', href) if not m: return None listing_id = m.group(1) card_text = card.get_text(separator=' ', strip=True) # --- Título --- titulo = None for sel in ('h2', 'h3', 'h4', '.item__title', '.item-title', '.ad-title', '.title', '[class*="title"]', '[class*="Title"]'): el = card.select_one(sel) if el: t = el.get_text(strip=True) if t: titulo = t[:255] break if not titulo: titulo = card_text[:120] or None # --- Precio --- precio = None for sel in ('.item__price', '.price', '.ad-price', '[class*="price"]', '[class*="Price"]', '[class*="precio"]', '.value', '.amount'): el = card.select_one(sel) if el: precio = _clean_precio(el.get_text()) if precio: break if not precio: # Buscar patrón "$NNN" en el texto completo de la card m_p = re.search(r'\$\s*([\d\.\,]+)', card_text) if m_p: precio = _clean_precio(m_p.group(1)) # --- Tipo --- tipo = _detect_tipo(href + ' ' + card_text) # --- Ubicacion --- ubicacion = None for sel in ('.item__location', '.item-location', '.location', '.commune', '.city', '.region', '[class*="location"]', '[class*="Location"]', '[class*="address"]', '[class*="Address"]'): el = card.select_one(sel) if el: t = el.get_text(strip=True) if t: ubicacion = t[:200] break # --- Ambientes / dormitorios --- ambientes = None m_amb = re.search( r'(\d+)\s*(?:amb(?:iente)?s?|dorm(?:itorio)?s?|hab(?:itaci[oó]n)?(?:es)?)', card_text, re.IGNORECASE ) if m_amb: try: ambientes = int(m_amb.group(1)) except ValueError: pass # --- Superficie m² --- superficie_m2 = None m_sup = re.search(r'(\d+(?:[,\.]\d+)?)\s*m[²2]', card_text, re.IGNORECASE) if m_sup: try: superficie_m2 = float(m_sup.group(1).replace(',', '.')) except ValueError: pass return { 'listing_id': listing_id, 'titulo': titulo, 'precio': precio, 'moneda': 'CLP', 'tipo': tipo, 'ubicacion': ubicacion, 'url': full_url, 'ambientes': ambientes, 'superficie_m2': superficie_m2, } # --------------------------------------------------------------------------- # SCRAPING DE PÁGINA # --------------------------------------------------------------------------- def scrape_page(sess, url): """ Obtiene una página de listados con reintentos y backoff. Devuelve lista de dicts (vacía si no hay resultados o hay error). """ backoff_delays = [2, 5, 15] resp = None for attempt in range(3): try: r = sess.get(url, timeout=30) if r.status_code == 404: log.info('404 – página inexistente: %s', url) return [] if r.status_code in (403, 429): log.warning('HTTP %s en %s – esperando 30 s antes de reintentar...', r.status_code, url) time.sleep(30) if attempt < 2: continue log.error('HTTP %s persiste tras espera, devolviendo []', r.status_code) return [] r.raise_for_status() resp = r break except requests.RequestException as exc: log.warning('Intento %d/3 fallido (%s): %s', attempt + 1, url, exc) if attempt < 2: time.sleep(backoff_delays[attempt]) else: log.error('Todos los reintentos fallaron para %s', url) return [] if resp is None: return [] # Detectar bloqueo por Cloudflare / CAPTCHA if any(k in resp.text for k in ('cf-browser-verification', 'challenge-form', 'Pardon Our Interruption')): log.warning('Posible bloqueo / CAPTCHA detectado en %s', url) return [] soup = BeautifulSoup(resp.text, 'lxml') # Selectores de cards en orden de especificidad (Yapo.cl / Schibsted) cards = ( soup.select('article.item') or soup.select('li.item') or soup.select('.ad-listing-item') or soup.select('article[data-ad-id]') or soup.select('[data-listing-id]') or soup.select('.listing-card') or soup.select('.aditem') or soup.select('ul.items > li') or soup.select('.item') or [] ) if not cards: # Fallback: recoger links directos a anuncios con ID numérico largo all_links = soup.select('a[href*="/bienes_raices/"]') seen_ids = set() items = [] for a in all_links: href = a.get('href', '') m = re.search(r'/(\d{6,})', href) if not m: continue lid = m.group(1) if lid in seen_ids: continue seen_ids.add(lid) full_url = href if href.startswith('http') else BASE_URL + href titulo = a.get_text(strip=True) or None parent_text = a.parent.get_text(separator=' ', strip=True) if a.parent else '' items.append({ 'listing_id': lid, 'titulo': titulo, 'precio': None, 'moneda': 'CLP', 'tipo': _detect_tipo(href + ' ' + (titulo or '') + ' ' + parent_text), 'ubicacion': None, 'url': full_url, 'ambientes': None, 'superficie_m2': None, }) if items: log.info('%s: %d anuncios (modo fallback links)', url, len(items)) else: log.info('Sin resultados en %s', url) return items items = [] for card in cards: try: item = _parse_card(card) if item: items.append(item) except Exception as exc: log.debug('Error parseando card: %s', exc) log.info('%s: %d anuncios parseados', url, len(items)) return items # --------------------------------------------------------------------------- # DETECCIÓN DE PATRÓN DE URL # --------------------------------------------------------------------------- def detect_url_pattern(sess): """ Prueba los patrones de URL con la página 1 y devuelve el primero funcional. """ for pattern in URL_PATTERNS: url = pattern.format(n=1) try: r = sess.get(url, timeout=30) if r.status_code == 200 and len(r.text) > 2000: soup = BeautifulSoup(r.text, 'lxml') has_content = bool( soup.select('article') or soup.select('.item') or soup.select('a[href*="/bienes_raices/"]') ) if has_content: log.info('Patrón URL activo: %s', pattern) return pattern log.debug('Patrón descartado (%s): status=%s', pattern, r.status_code) except Exception as exc: log.debug('Patrón %s falló: %s', pattern, exc) time.sleep(1.0) log.warning('No se detectó patrón funcional; usando fallback: %s', URL_PATTERNS[1]) return URL_PATTERNS[1] # --------------------------------------------------------------------------- # MAIN # --------------------------------------------------------------------------- def main(): log.info('=== Scraper Yapo.cl iniciado ===') sess = get_session() conn = None try: conn = psycopg2.connect( host=DB_HOST, port=DB_PORT, dbname=DB_NAME, user=DB_USER, password=DB_PASS, connect_timeout=15, ) log.info('Conexión DB OK (%s@%s/%s)', DB_USER, DB_HOST, DB_NAME) except Exception as exc: raise RuntimeError('Error conectando a PostgreSQL: {}'.format(exc)) from exc try: create_table(conn) url_pattern = detect_url_pattern(sess) log.info('Patrón de URL seleccionado: %s', url_pattern) total_nuevos = 0 total_actualizados = 0 streak_sin_nuevos = 0 for page in range(1, MAX_PAGES + 1): url = url_pattern.format(n=page) log.info('--- Página %d/%d: %s ---', page, MAX_PAGES, url) try: rows = scrape_page(sess, url) except Exception as exc: log.error('Error inesperado en scrape_page(%s): %s', url, exc) rows = [] if not rows: streak_sin_nuevos += 1 log.info('Sin resultados en página %d (racha: %d/%d)', page, streak_sin_nuevos, NEW_STREAK) if streak_sin_nuevos >= NEW_STREAK: log.info('Deteniendo scraping — %d páginas consecutivas sin resultados', NEW_STREAK) break time.sleep(DELAY) continue try: nuevos, actualizados = upsert(conn, rows) except Exception as exc: log.error('Error en upsert página %d: %s', page, exc) try: conn.rollback() except Exception: pass nuevos, actualizados = 0, 0 total_nuevos += nuevos total_actualizados += actualizados log.info( 'Página %d: %d nuevos, %d actualizados (acum: %d nuevos / %d actualizados)', page, nuevos, actualizados, total_nuevos, total_actualizados ) if nuevos == 0: streak_sin_nuevos += 1 if streak_sin_nuevos >= NEW_STREAK: log.info('Deteniendo — %d páginas sin nuevos registros', NEW_STREAK) break else: streak_sin_nuevos = 0 time.sleep(DELAY) finally: if conn: conn.close() log.info('Conexión DB cerrada') log.info('=== Scraper Yapo.cl finalizado ===') print('Finalizado: {} nuevos, {} actualizados'.format(total_nuevos, total_actualizados)) if __name__ == '__main__': main()