Files
opinie/scraper.py
T
2026-08-31 20:16:08 +02:00

138 lines
5.8 KiB
Python

import os
import re
from urllib.parse import urlsplit, urlunsplit
from playwright.sync_api import TimeoutError as PlaywrightTimeoutError
from playwright.sync_api import sync_playwright
ALLOWED_HOSTS = {
"maps.app.goo.gl", "goo.gl", "google.com", "www.google.com",
"maps.google.com", "google.pl", "www.google.pl", "maps.google.pl",
}
MAX_URL_LENGTH = 1200
class ScrapeError(RuntimeError):
pass
def validate_maps_url(raw_url: str) -> str:
"""Przepuszcza wyłącznie zwykłe linki HTTPS do Google Maps."""
url = raw_url.strip()
if not url:
raise ValueError("Wklej link do miejsca w Google Maps.")
if len(url) > MAX_URL_LENGTH or any(ord(char) < 32 for char in url):
raise ValueError("Link jest nieprawidłowy albo zbyt długi.")
try:
parsed = urlsplit(url)
host = (parsed.hostname or "").lower().rstrip(".")
port = parsed.port
except ValueError:
raise ValueError("Link ma nieprawidłowy format.") from None
if parsed.scheme.lower() != "https" or not host or parsed.username or parsed.password:
raise ValueError("Dozwolony jest wyłącznie bezpieczny link HTTPS do Google Maps.")
if port not in (None, 443) or host not in ALLOWED_HOSTS:
raise ValueError("To nie jest obsługiwany link Google Maps.")
path = parsed.path or "/"
if host == "maps.app.goo.gl":
valid_path = re.fullmatch(r"/[A-Za-z0-9_-]{5,100}/?", path)
elif host == "goo.gl":
valid_path = re.fullmatch(r"/maps/[A-Za-z0-9_-]{5,100}/?", path)
else:
valid_path = path == "/maps" or path.startswith("/maps/")
if not valid_path:
raise ValueError("Link musi prowadzić do miejsca w Google Maps.")
return urlunsplit(("https", host, path, parsed.query, ""))
def _number(text: str) -> int | None:
compact = re.sub(r"[\s\u00a0\u202f.,]", "", text)
return int(compact) if compact.isdigit() else None
def parse_distribution_labels(labels: list[str]) -> dict[int, int]:
"""Czyta np. '5 gwiazdek, 123 opinie' — nie mierzy szerokości paska."""
result = {}
for label in labels:
clean = " ".join(label.lower().split())
if not re.search(r"gwiazd|star", clean) or not re.search(r"opini|recenz|review", clean):
continue
rating_match = re.search(r"(?:^|\D)([1-5])(?:\D|$)", clean)
if not rating_match:
continue
candidates = re.findall(r"\d[\d\s\u00a0\u202f.,]*", clean[rating_match.end():])
values = [_number(item) for item in candidates]
values = [item for item in values if item is not None]
if values:
result[int(rating_match.group(1))] = values[-1]
return result
def _accept_cookies(page):
for text in ("Zaakceptuj wszystko", "Odrzuć wszystko", "Accept all", "Reject all"):
try:
button = page.get_by_role("button", name=re.compile(f"^{re.escape(text)}$", re.I)).first
if button.is_visible(timeout=700):
button.click(timeout=2_000)
page.wait_for_timeout(500)
return
except Exception:
pass
def _open_reviews(page):
candidates = page.locator("button, [role=button]")
for index in range(min(candidates.count(), 250)):
element = candidates.nth(index)
try:
label = " ".join(filter(None, [
element.get_attribute("aria-label"), element.inner_text(timeout=100)
]))
if re.search(r"\d", label) and re.search(r"opini|recenz|review", label, re.I):
element.click(timeout=2_000)
page.wait_for_timeout(800)
return
except Exception:
pass
def get_rating_distribution(url: str) -> dict:
"""Pobiera pięć zbiorczych liczników. Treści opinii nie są odczytywane ani zapisywane."""
try:
with sync_playwright() as playwright:
launch = {"headless": True, "args": ["--disable-dev-shm-usage", "--no-first-run", "--disable-gpu"]}
if os.environ.get("CHROMIUM_PATH"):
launch["executable_path"] = os.environ["CHROMIUM_PATH"]
browser = playwright.chromium.launch(**launch)
try:
page = browser.new_page(locale="pl-PL", viewport={"width": 1100, "height": 800})
page.goto(url, wait_until="domcontentloaded", timeout=30_000)
_accept_cookies(page)
_open_reviews(page)
page.wait_for_timeout(700)
labels = page.locator("[aria-label]").evaluate_all(
"els => els.map(el => el.getAttribute('aria-label')).filter(Boolean)"
)
counts = parse_distribution_labels(labels)
if len(counts) != 5:
body_text = page.locator("body").inner_text().lower()
if "ograniczonego widoku" in body_text or "limited view" in body_text:
raise ScrapeError("Google włączyło ograniczony widok i ukryło opinie. Spróbuj później lub z innego łącza.")
raise ScrapeError("Nie udało się odczytać dokładnych pięciu liczników. Sprawdź link do miejsca.")
ratings = {str(stars): counts[stars] for stars in range(5, 0, -1)}
total = sum(ratings.values())
average = round(sum(int(stars) * count for stars, count in ratings.items()) / total, 2) if total else 0
return {"total": total, "average": average, "ratings": ratings}
finally:
browser.close()
except ScrapeError:
raise
except PlaywrightTimeoutError:
raise ScrapeError("Google Maps nie odpowiedziało w wyznaczonym czasie.") from None
except Exception as exc:
if "Executable doesn't exist" in str(exc):
raise ScrapeError("Chromium nie jest zainstalowane na serwerze.") from None
raise