"""Private compliant website scraper using SeleniumBase/browser automation. The public skill is intentionally narrow: it accepts one starting URL, a CSS selector mapping, pagination limits, browser options, robots.txt behavior, and output format. It never attempts to bypass CAPTCHAs, anti-bot challenges, access controls, paywalls, login gates, robots.txt restrictions, or terms of service restrictions. """ from __future__ import annotations import json import re import textwrap import urllib.parse import urllib.request import urllib.robotparser from pathlib import Path from typing import Any, Literal from pydantic import BaseModel, Field from a2a_pack import ( A2AAgent, EgressPolicy, LLMProvisioning, NoAuth, Pricing, Resources, RunContext, WorkspaceAccess, WorkspaceMode, skill, ) OUTPUT_DIR = "outputs/seleniumbase-website-scraper" RUNTIME_SKILLS_DIR = "seleniumbase-website-scraper/.deepagents/skills/" class SeleniumbaseWebsiteScraperConfig(BaseModel): max_pages_limit: int = Field(default=50, ge=1, le=200) default_user_agent: str = "A2A seleniumbase-website-scraper/1.0 (+authorized public/user-owned scraping only)" class SeleniumbaseWebsiteScraper(A2AAgent[SeleniumbaseWebsiteScraperConfig, NoAuth]): name = "seleniumbase-website-scraper" description = ( "Private compliant website scraping agent using SeleniumBase/browser " "automation with robots.txt checks, same-domain pagination safeguards, " "screenshots, HTML capture, JSON/CSV output, retries, and polite rate limiting." ) version = "1.0.2" config_model = SeleniumbaseWebsiteScraperConfig auth_model = NoAuth llm_provisioning = LLMProvisioning.PLATFORM pricing = Pricing( price_per_call_usd=0.0, caller_pays_llm=True, notes=( "Browser scraping is deterministic; any future LLM use must read the " "caller-provided ctx.llm credential. The caller pays any forwarded LLM cost." ), ) resources = Resources(cpu="2", memory="2Gi", max_runtime_seconds=900) egress = EgressPolicy(deny_internet_by_default=False) tools_used = ("seleniumbase", "selenium", "chromium", "browser-automation") workspace_access = WorkspaceAccess.dynamic( max_files=256, allowed_modes=(WorkspaceMode.READ_ONLY, WorkspaceMode.READ_WRITE_OVERLAY), require_reason=False, max_total_size_bytes=250 * 1024 * 1024, ) @skill( description=( "Authorized website scraping with SeleniumBase/browser automation, " "robots.txt checks, selector extraction, same-domain pagination, " "screenshots, and JSON/CSV output. Stops on CAPTCHA, anti-bot, login, " "paywall, or access-control challenges." ), timeout_seconds=900, cost_class="browser-heavy", grant_outputs_prefix=OUTPUT_DIR, grant_write_prefixes=(OUTPUT_DIR,), grant_run_timeout_seconds=900, ) async def scrape_website( self, ctx: RunContext[NoAuth], url: str, selectors: dict[str, str], max_pages: int = 1, wait_seconds: float = 3.0, output_format: Literal["json", "csv"] = "json", headless: bool = True, user_agent: str = "", respect_robots: bool = True, rate_limit_seconds: float = 1.0, screenshot: bool = False, ) -> dict[str, Any]: """Scrape authorized public/user-owned pages and save structured output.""" creds = ctx.llm # Read platform-forwarded LLM metadata; this skill does not construct a model. await ctx.emit_progress(f"starting compliant scrape; llm credential source={creds.source}") validation = _validate_inputs( url=url, selectors=selectors, max_pages=max_pages, max_pages_limit=self.config.max_pages_limit, wait_seconds=wait_seconds, output_format=output_format, user_agent=user_agent, rate_limit_seconds=rate_limit_seconds, ) if validation["errors"]: return { "status": "invalid_input", "items": [], "saved_paths": [], "warnings": validation["errors"], "screenshots": [], } start_url = validation["url"] effective_user_agent = user_agent.strip() or self.config.default_user_agent warnings: list[str] = list(validation["warnings"]) if respect_robots: robots = _robots_allowed(start_url, effective_user_agent) warnings.extend(robots.get("warnings", [])) if not robots.get("allowed", False): return { "status": "blocked_by_robots_txt", "items": [], "saved_paths": [], "warnings": warnings + ["robots.txt disallows fetching the requested URL for this user agent."], "screenshots": [], "robots": robots, } script_payload = { "url": start_url, "selectors": selectors, "max_pages": max_pages, "wait_seconds": wait_seconds, "output_format": output_format, "headless": headless, "user_agent": effective_user_agent, "respect_robots": respect_robots, "rate_limit_seconds": rate_limit_seconds, "screenshot": screenshot, "output_dir": f"/workspace/{OUTPUT_DIR}", } script = _render_scraper_script(script_payload) try: result = await ctx.workspace_python( script, image="seleniumbase/seleniumbase:latest", timeout_seconds=900, memory_mib=2048, cpus=2, ) except Exception as exc: # noqa: BLE001 await ctx.emit_error(str(exc), code="browser_runtime_unavailable") return { "status": "browser_runtime_unavailable", "items": [], "saved_paths": [], "warnings": warnings + [ "Browser sandbox could not be started. The platform must allow the SeleniumBase image, " "or rerun where browser automation is available.", f"runtime error: {type(exc).__name__}: {exc}", ], "screenshots": [], } stdout = (result.stdout or "").strip() stderr = (result.stderr or "").strip() parsed = _parse_last_json(stdout) if not parsed: return { "status": "error", "items": [], "saved_paths": [], "warnings": warnings + [ "Browser script did not return a parseable JSON result.", f"returncode={getattr(result, 'exit_code', None)}", f"stdout_tail={stdout[-2000:]}", f"stderr_tail={stderr[-2000:]}", ], "screenshots": [], } parsed_warnings = list(parsed.get("warnings") or []) if getattr(result, "exit_code", 0) not in (0, None): parsed_warnings.append( f"browser command returned nonzero exit code {getattr(result, 'exit_code', None)}; usable emitted files were preserved when present" ) if stderr: parsed_warnings.append(f"stderr_tail={stderr[-2000:]}") saved_paths = [str(path).replace("/workspace/", "") for path in (parsed.get("saved_paths") or [])] screenshot_paths = [str(path).replace("/workspace/", "") for path in (parsed.get("screenshots") or [])] for path in saved_paths[:5] + screenshot_paths[:5]: await _emit_workspace_artifact_if_available(ctx, path) status = str(parsed.get("status") or "ok") if status in {"challenge_detected", "manual_intervention_required", "blocked"}: await ctx.emit_error("Scrape stopped because a CAPTCHA, anti-bot, login, paywall, or access-control challenge was detected.", code=status) else: await ctx.emit_progress(f"scrape finished with status={status}; items={len(parsed.get('items') or [])}") return { "status": status, "items": parsed.get("items") or [], "saved_paths": saved_paths, "warnings": warnings + parsed_warnings, "screenshots": screenshot_paths, "pages_visited": parsed.get("pages_visited") or [], "html_captures": [str(path).replace("/workspace/", "") for path in (parsed.get("html_captures") or [])], "robots": {"checked": bool(respect_robots)}, } def _validate_inputs( *, url: str, selectors: dict[str, str], max_pages: int, max_pages_limit: int, wait_seconds: float, output_format: str, user_agent: str, rate_limit_seconds: float, ) -> dict[str, Any]: errors: list[str] = [] warnings: list[str] = [] clean_url = str(url or "").strip() parsed = urllib.parse.urlparse(clean_url) if parsed.scheme not in {"http", "https"} or not parsed.netloc: errors.append("url must be an absolute http(s) URL.") if parsed.username or parsed.password: errors.append("url must not contain embedded credentials.") if not selectors: errors.append("selectors mapping must contain at least one CSS selector.") for key, value in selectors.items(): if not isinstance(key, str) or not key.strip(): errors.append("selector field names must be non-empty strings.") if not isinstance(value, str) or not value.strip(): errors.append(f"selector for {key!r} must be a non-empty string.") if re.search(r"password|token|secret|credential", key, re.I): warnings.append(f"selector field {key!r} looks sensitive; do not scrape secrets or credentials.") if max_pages < 1 or max_pages > max_pages_limit: errors.append(f"max_pages must be between 1 and {max_pages_limit}.") if wait_seconds < 0 or wait_seconds > 60: errors.append("wait_seconds must be between 0 and 60.") if output_format not in {"json", "csv"}: errors.append("output_format must be 'json' or 'csv'.") if "\n" in user_agent or "\r" in user_agent: errors.append("user_agent must be a single-line string.") if rate_limit_seconds < 0 or rate_limit_seconds > 120: errors.append("rate_limit_seconds must be between 0 and 120.") return {"errors": errors, "warnings": warnings, "url": clean_url} def _robots_allowed(url: str, user_agent: str) -> dict[str, Any]: parsed = urllib.parse.urlparse(url) robots_url = urllib.parse.urlunparse((parsed.scheme, parsed.netloc, "/robots.txt", "", "", "")) rp = urllib.robotparser.RobotFileParser() rp.set_url(robots_url) try: with urllib.request.urlopen( urllib.request.Request(robots_url, headers={"User-Agent": user_agent}), timeout=10, ) as response: body = response.read(2000000).decode("utf-8", errors="ignore") rp.parse(body.splitlines()) return {"allowed": bool(rp.can_fetch(user_agent, url)), "robots_url": robots_url, "warnings": []} except urllib.error.HTTPError as exc: if exc.code in {401, 403}: return { "allowed": False, "robots_url": robots_url, "warnings": [f"robots.txt returned HTTP {exc.code}; treating as disallow for safety."], } if exc.code == 404: return {"allowed": True, "robots_url": robots_url, "warnings": ["robots.txt not found; proceeding because no robots rules were published."]} return { "allowed": False, "robots_url": robots_url, "warnings": [f"robots.txt check failed with HTTP {exc.code}; treating as disallow for safety."], } except Exception as exc: # noqa: BLE001 return { "allowed": False, "robots_url": robots_url, "warnings": [f"robots.txt could not be checked ({type(exc).__name__}: {exc}); treating as disallow for safety."], } def _render_scraper_script(payload: dict[str, Any]) -> str: payload_json = json.dumps(payload, ensure_ascii=False) return textwrap.dedent( f""" import csv import json import os import re import sys import time import traceback import urllib.parse import urllib.request import urllib.robotparser from datetime import datetime, timezone PAYLOAD = json.loads({payload_json!r}) CHALLENGE_PATTERNS = [ r"captcha", r"recaptcha", r"hcaptcha", r"cf-challenge", r"cloudflare", r"turnstile", r"are you human", r"verify you are human", r"bot detection", r"automated traffic", r"access denied", r"temporarily blocked", r"unusual traffic", r"login required", r"sign in to continue", r"subscribe to continue", r"paywall", r"forbidden", ] PAGINATION_KEYS = {{"next", "_next", "__next__", "pagination_next", "next_page"}} def emit(payload): print(json.dumps(payload, ensure_ascii=False)) def safe_name(value): value = re.sub(r"[^a-zA-Z0-9._-]+", "-", value).strip("-._") return value[:80] or "page" def same_origin(a, b): pa, pb = urllib.parse.urlparse(a), urllib.parse.urlparse(b) return pa.scheme == pb.scheme and pa.netloc.lower() == pb.netloc.lower() def robots_allowed(url, user_agent): parsed = urllib.parse.urlparse(url) robots_url = urllib.parse.urlunparse((parsed.scheme, parsed.netloc, "/robots.txt", "", "", "")) rp = urllib.robotparser.RobotFileParser() rp.set_url(robots_url) try: req = urllib.request.Request(robots_url, headers={{"User-Agent": user_agent}}) with urllib.request.urlopen(req, timeout=10) as response: body = response.read(2000000).decode("utf-8", errors="ignore") rp.parse(body.splitlines()) return bool(rp.can_fetch(user_agent, url)), None except Exception as exc: return False, f"robots.txt could not be checked for pagination URL {{url}}: {{type(exc).__name__}}: {{exc}}" def split_selector(selector): selector = selector.strip() m = re.search(r"::attr\\(([^)]+)\\)\\s*$", selector) if m: return selector[:m.start()].strip(), "attr", m.group(1).strip() if selector.endswith("::text"): return selector[:-6].strip(), "text", None if selector.endswith("::html"): return selector[:-6].strip(), "html", None return selector, "text", None def detect_challenge(page_source, title, current_url): haystack = "\\n".join([title or "", current_url or "", page_source[:50000] or ""]).lower() matched = [pat for pat in CHALLENGE_PATTERNS if re.search(pat, haystack, re.I)] if matched: return "manual_intervention_required", "challenge/access-control indicator detected: " + ", ".join(sorted(set(matched))[:8]) return None, None def extract_with_bs4(page_source, selectors): from bs4 import BeautifulSoup soup = BeautifulSoup(page_source, "html.parser") item = {{}} next_url = None for key, raw_selector in selectors.items(): css, mode, attr = split_selector(str(raw_selector)) try: elements = soup.select(css) except Exception as exc: item[key] = None item.setdefault("_selector_warnings", []).append(f"selector {{key}}={{raw_selector!r}} failed: {{exc}}") continue values = [] for el in elements: if mode == "attr": values.append(el.get(attr)) elif mode == "html": values.append(el.decode_contents()) else: values.append(el.get_text(" ", strip=True)) values = [v for v in values if v is not None] if key in PAGINATION_KEYS: next_url = values[0] if values else None else: item[key] = values if len(values) != 1 else values[0] if values else None return item, next_url def write_outputs(items, pages_visited, html_captures, screenshots, warnings, status): out_dir = PAYLOAD["output_dir"] os.makedirs(out_dir, exist_ok=True) stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ") base = safe_name(urllib.parse.urlparse(PAYLOAD["url"]).netloc) + "-" + stamp metadata = {{ "status": status, "items": items, "pages_visited": pages_visited, "html_captures": html_captures, "screenshots": screenshots, "warnings": warnings, "safety": "Authorized public/user-owned scraping only. No CAPTCHA, anti-bot, login, paywall, robots.txt, ToS, or access-control circumvention attempted.", }} json_path = os.path.join(out_dir, base + ".json") with open(json_path, "w", encoding="utf-8") as f: json.dump(metadata, f, ensure_ascii=False, indent=2) saved_paths = [json_path] if PAYLOAD["output_format"] == "csv": csv_path = os.path.join(out_dir, base + ".csv") fieldnames = sorted({{k for item in items for k in item.keys() if k != "_selector_warnings"}}) with open(csv_path, "w", newline="", encoding="utf-8") as f: writer = csv.DictWriter(f, fieldnames=fieldnames or ["url"]) writer.writeheader() for item in items: row = {{k: (json.dumps(v, ensure_ascii=False) if isinstance(v, (list, dict)) else v) for k, v in item.items() if k in (fieldnames or ["url"])}} writer.writerow(row) saved_paths.append(csv_path) return saved_paths def run(): warnings = [] items = [] pages_visited = [] html_captures = [] screenshots = [] status = "ok" out_dir = PAYLOAD["output_dir"] os.makedirs(out_dir, exist_ok=True) current = PAYLOAD["url"] start = current try: from seleniumbase import SB browser_kwargs = {{ "headless": bool(PAYLOAD["headless"]), "uc": False, "test": False, "locale_code": "en", }} if PAYLOAD.get("user_agent"): browser_kwargs["agent"] = PAYLOAD["user_agent"] with SB(**browser_kwargs) as sb: for page_index in range(int(PAYLOAD["max_pages"])): if not same_origin(start, current): warnings.append(f"same-domain safeguard stopped pagination to {{current}}") break if PAYLOAD.get("respect_robots"): allowed, robot_warning = robots_allowed(current, PAYLOAD.get("user_agent") or "*") if robot_warning: warnings.append(robot_warning) if not allowed: status = "blocked_by_robots_txt" warnings.append(f"robots.txt disallows pagination URL {{current}}") break if page_index > 0 and float(PAYLOAD["rate_limit_seconds"]) > 0: time.sleep(float(PAYLOAD["rate_limit_seconds"])) sb.open(current) if float(PAYLOAD["wait_seconds"]) > 0: sb.sleep(float(PAYLOAD["wait_seconds"])) current_url = sb.get_current_url() title = sb.get_title() page_source = sb.get_page_source() pages_visited.append(current_url) challenge_status, challenge_warning = detect_challenge(page_source, title, current_url) if challenge_status: status = challenge_status warnings.append(challenge_warning) html_path = os.path.join(out_dir, f"page-{{page_index+1}}-challenge.html") with open(html_path, "w", encoding="utf-8") as f: f.write(page_source) html_captures.append(html_path) if PAYLOAD.get("screenshot"): shot_path = os.path.join(out_dir, f"page-{{page_index+1}}-challenge.png") try: sb.save_screenshot(shot_path) screenshots.append(shot_path) except Exception as exc: warnings.append(f"screenshot failed: {{exc}}") break html_path = os.path.join(out_dir, f"page-{{page_index+1}}.html") with open(html_path, "w", encoding="utf-8") as f: f.write(page_source) html_captures.append(html_path) if PAYLOAD.get("screenshot"): shot_path = os.path.join(out_dir, f"page-{{page_index+1}}.png") try: sb.save_screenshot(shot_path) screenshots.append(shot_path) except Exception as exc: warnings.append(f"screenshot failed: {{exc}}") item, next_raw = extract_with_bs4(page_source, PAYLOAD["selectors"]) item["url"] = current_url item["page_index"] = page_index + 1 if item.get("_selector_warnings"): warnings.extend(item.pop("_selector_warnings")) items.append(item) if page_index >= int(PAYLOAD["max_pages"]) - 1 or not next_raw: break next_url = urllib.parse.urljoin(current_url, str(next_raw)) if not same_origin(start, next_url): warnings.append(f"same-domain safeguard refused pagination URL {{next_url}}") break current = next_url except Exception as exc: status = "error" warnings.append(f"browser automation failed: {{type(exc).__name__}}: {{exc}}") warnings.append(traceback.format_exc()[-4000:]) saved_paths = write_outputs(items, pages_visited, html_captures, screenshots, warnings, status) emit({{ "status": status, "items": items, "saved_paths": saved_paths, "screenshots": screenshots, "html_captures": html_captures, "warnings": warnings, "pages_visited": pages_visited, }}) return 0 if status in {{"ok", "blocked_by_robots_txt", "manual_intervention_required"}} else 1 if __name__ == "__main__": sys.exit(run()) """ ) def _parse_last_json(stdout: str) -> dict[str, Any] | None: for line in reversed(stdout.splitlines()): line = line.strip() if not line.startswith("{"): continue try: value = json.loads(line) except json.JSONDecodeError: continue if isinstance(value, dict): return value return None async def _emit_workspace_artifact_if_available(ctx: RunContext[Any], path: str) -> None: try: reader = getattr(ctx.workspace, "read_bytes", None) if reader is None: return data = reader(path) if not isinstance(data, (bytes, bytearray)): return mime = "application/json" if path.endswith(".csv"): mime = "text/csv" elif path.endswith(".png"): mime = "image/png" elif path.endswith(".html"): mime = "text/html" ref = await ctx.write_artifact(Path(path).name, bytes(data), mime) await ctx.emit_artifact(ref) except Exception: # noqa: BLE001 return