diff --git a/agent.py b/agent.py new file mode 100644 index 0000000..18c8ef3 --- /dev/null +++ b/agent.py @@ -0,0 +1,577 @@ +"""Private compliant website scraper using SeleniumBase/browser automation. + +The public skill is intentionally narrow: it accepts one starting URL, a CSS +selector mapping, pagination limits, browser options, robots.txt behavior, and +output format. It never attempts to bypass CAPTCHAs, anti-bot challenges, +access controls, paywalls, login gates, robots.txt restrictions, or terms of +service restrictions. +""" +from __future__ import annotations + +import json +import re +import textwrap +import urllib.parse +import urllib.request +import urllib.robotparser +from pathlib import Path +from typing import Any, Literal + +from pydantic import BaseModel, Field + +from a2a_pack import ( + A2AAgent, + EgressPolicy, + LLMProvisioning, + NoAuth, + Pricing, + Resources, + RunContext, + WorkspaceAccess, + WorkspaceMode, + skill, +) + + +OUTPUT_DIR = "outputs/seleniumbase-website-scraper" +RUNTIME_SKILLS_DIR = "seleniumbase-website-scraper/.deepagents/skills/" + + +class SeleniumbaseWebsiteScraperConfig(BaseModel): + max_pages_limit: int = Field(default=50, ge=1, le=200) + default_user_agent: str = "A2A seleniumbase-website-scraper/1.0 (+authorized public/user-owned scraping only)" + + +class SeleniumbaseWebsiteScraper(A2AAgent[SeleniumbaseWebsiteScraperConfig, NoAuth]): + name = "seleniumbase-website-scraper" + description = ( + "Private compliant website scraping agent using SeleniumBase/browser " + "automation with robots.txt checks, same-domain pagination safeguards, " + "screenshots, HTML capture, JSON/CSV output, retries, and polite rate limiting." + ) + version = "1.0.2" + + config_model = SeleniumbaseWebsiteScraperConfig + auth_model = NoAuth + + llm_provisioning = LLMProvisioning.PLATFORM + pricing = Pricing( + price_per_call_usd=0.0, + caller_pays_llm=True, + notes=( + "Browser scraping is deterministic; any future LLM use must read the " + "caller-provided ctx.llm credential. The caller pays any forwarded LLM cost." + ), + ) + resources = Resources(cpu="2", memory="2Gi", max_runtime_seconds=900) + egress = EgressPolicy(deny_internet_by_default=False) + tools_used = ("seleniumbase", "selenium", "chromium", "browser-automation") + workspace_access = WorkspaceAccess.dynamic( + max_files=256, + allowed_modes=(WorkspaceMode.READ_ONLY, WorkspaceMode.READ_WRITE_OVERLAY), + require_reason=False, + max_total_size_bytes=250 * 1024 * 1024, + ) + + @skill( + description=( + "Authorized website scraping with SeleniumBase/browser automation, " + "robots.txt checks, selector extraction, same-domain pagination, " + "screenshots, and JSON/CSV output. Stops on CAPTCHA, anti-bot, login, " + "paywall, or access-control challenges." + ), + timeout_seconds=900, + cost_class="browser-heavy", + grant_outputs_prefix=OUTPUT_DIR, + grant_write_prefixes=(OUTPUT_DIR,), + grant_run_timeout_seconds=900, + ) + async def scrape_website( + self, + ctx: RunContext[NoAuth], + url: str, + selectors: dict[str, str], + max_pages: int = 1, + wait_seconds: float = 3.0, + output_format: Literal["json", "csv"] = "json", + headless: bool = True, + user_agent: str = "", + respect_robots: bool = True, + rate_limit_seconds: float = 1.0, + screenshot: bool = False, + ) -> dict[str, Any]: + """Scrape authorized public/user-owned pages and save structured output.""" + creds = ctx.llm # Read platform-forwarded LLM metadata; this skill does not construct a model. + await ctx.emit_progress(f"starting compliant scrape; llm credential source={creds.source}") + + validation = _validate_inputs( + url=url, + selectors=selectors, + max_pages=max_pages, + max_pages_limit=self.config.max_pages_limit, + wait_seconds=wait_seconds, + output_format=output_format, + user_agent=user_agent, + rate_limit_seconds=rate_limit_seconds, + ) + if validation["errors"]: + return { + "status": "invalid_input", + "items": [], + "saved_paths": [], + "warnings": validation["errors"], + "screenshots": [], + } + + start_url = validation["url"] + effective_user_agent = user_agent.strip() or self.config.default_user_agent + warnings: list[str] = list(validation["warnings"]) + + if respect_robots: + robots = _robots_allowed(start_url, effective_user_agent) + warnings.extend(robots.get("warnings", [])) + if not robots.get("allowed", False): + return { + "status": "blocked_by_robots_txt", + "items": [], + "saved_paths": [], + "warnings": warnings + ["robots.txt disallows fetching the requested URL for this user agent."], + "screenshots": [], + "robots": robots, + } + + script_payload = { + "url": start_url, + "selectors": selectors, + "max_pages": max_pages, + "wait_seconds": wait_seconds, + "output_format": output_format, + "headless": headless, + "user_agent": effective_user_agent, + "respect_robots": respect_robots, + "rate_limit_seconds": rate_limit_seconds, + "screenshot": screenshot, + "output_dir": f"/workspace/{OUTPUT_DIR}", + } + + script = _render_scraper_script(script_payload) + try: + result = await ctx.workspace_python( + script, + image="seleniumbase/seleniumbase:latest", + timeout_seconds=900, + memory_mib=2048, + cpus=2, + ) + except Exception as exc: # noqa: BLE001 + await ctx.emit_error(str(exc), code="browser_runtime_unavailable") + return { + "status": "browser_runtime_unavailable", + "items": [], + "saved_paths": [], + "warnings": warnings + + [ + "Browser sandbox could not be started. The platform must allow the SeleniumBase image, " + "or rerun where browser automation is available.", + f"runtime error: {type(exc).__name__}: {exc}", + ], + "screenshots": [], + } + + stdout = (result.stdout or "").strip() + stderr = (result.stderr or "").strip() + parsed = _parse_last_json(stdout) + if not parsed: + return { + "status": "error", + "items": [], + "saved_paths": [], + "warnings": warnings + + [ + "Browser script did not return a parseable JSON result.", + f"returncode={getattr(result, 'exit_code', None)}", + f"stdout_tail={stdout[-2000:]}", + f"stderr_tail={stderr[-2000:]}", + ], + "screenshots": [], + } + + parsed_warnings = list(parsed.get("warnings") or []) + if getattr(result, "exit_code", 0) not in (0, None): + parsed_warnings.append( + f"browser command returned nonzero exit code {getattr(result, 'exit_code', None)}; usable emitted files were preserved when present" + ) + if stderr: + parsed_warnings.append(f"stderr_tail={stderr[-2000:]}") + + saved_paths = [str(path).replace("/workspace/", "") for path in (parsed.get("saved_paths") or [])] + screenshot_paths = [str(path).replace("/workspace/", "") for path in (parsed.get("screenshots") or [])] + + for path in saved_paths[:5] + screenshot_paths[:5]: + await _emit_workspace_artifact_if_available(ctx, path) + + status = str(parsed.get("status") or "ok") + if status in {"challenge_detected", "manual_intervention_required", "blocked"}: + await ctx.emit_error("Scrape stopped because a CAPTCHA, anti-bot, login, paywall, or access-control challenge was detected.", code=status) + else: + await ctx.emit_progress(f"scrape finished with status={status}; items={len(parsed.get('items') or [])}") + + return { + "status": status, + "items": parsed.get("items") or [], + "saved_paths": saved_paths, + "warnings": warnings + parsed_warnings, + "screenshots": screenshot_paths, + "pages_visited": parsed.get("pages_visited") or [], + "html_captures": [str(path).replace("/workspace/", "") for path in (parsed.get("html_captures") or [])], + "robots": {"checked": bool(respect_robots)}, + } + + +def _validate_inputs( + *, + url: str, + selectors: dict[str, str], + max_pages: int, + max_pages_limit: int, + wait_seconds: float, + output_format: str, + user_agent: str, + rate_limit_seconds: float, +) -> dict[str, Any]: + errors: list[str] = [] + warnings: list[str] = [] + clean_url = str(url or "").strip() + parsed = urllib.parse.urlparse(clean_url) + if parsed.scheme not in {"http", "https"} or not parsed.netloc: + errors.append("url must be an absolute http(s) URL.") + if parsed.username or parsed.password: + errors.append("url must not contain embedded credentials.") + if not selectors: + errors.append("selectors mapping must contain at least one CSS selector.") + for key, value in selectors.items(): + if not isinstance(key, str) or not key.strip(): + errors.append("selector field names must be non-empty strings.") + if not isinstance(value, str) or not value.strip(): + errors.append(f"selector for {key!r} must be a non-empty string.") + if re.search(r"password|token|secret|credential", key, re.I): + warnings.append(f"selector field {key!r} looks sensitive; do not scrape secrets or credentials.") + if max_pages < 1 or max_pages > max_pages_limit: + errors.append(f"max_pages must be between 1 and {max_pages_limit}.") + if wait_seconds < 0 or wait_seconds > 60: + errors.append("wait_seconds must be between 0 and 60.") + if output_format not in {"json", "csv"}: + errors.append("output_format must be 'json' or 'csv'.") + if "\n" in user_agent or "\r" in user_agent: + errors.append("user_agent must be a single-line string.") + if rate_limit_seconds < 0 or rate_limit_seconds > 120: + errors.append("rate_limit_seconds must be between 0 and 120.") + return {"errors": errors, "warnings": warnings, "url": clean_url} + + +def _robots_allowed(url: str, user_agent: str) -> dict[str, Any]: + parsed = urllib.parse.urlparse(url) + robots_url = urllib.parse.urlunparse((parsed.scheme, parsed.netloc, "/robots.txt", "", "", "")) + rp = urllib.robotparser.RobotFileParser() + rp.set_url(robots_url) + try: + with urllib.request.urlopen( + urllib.request.Request(robots_url, headers={"User-Agent": user_agent}), + timeout=10, + ) as response: + body = response.read(2000000).decode("utf-8", errors="ignore") + rp.parse(body.splitlines()) + return {"allowed": bool(rp.can_fetch(user_agent, url)), "robots_url": robots_url, "warnings": []} + except urllib.error.HTTPError as exc: + if exc.code in {401, 403}: + return { + "allowed": False, + "robots_url": robots_url, + "warnings": [f"robots.txt returned HTTP {exc.code}; treating as disallow for safety."], + } + if exc.code == 404: + return {"allowed": True, "robots_url": robots_url, "warnings": ["robots.txt not found; proceeding because no robots rules were published."]} + return { + "allowed": False, + "robots_url": robots_url, + "warnings": [f"robots.txt check failed with HTTP {exc.code}; treating as disallow for safety."], + } + except Exception as exc: # noqa: BLE001 + return { + "allowed": False, + "robots_url": robots_url, + "warnings": [f"robots.txt could not be checked ({type(exc).__name__}: {exc}); treating as disallow for safety."], + } + + +def _render_scraper_script(payload: dict[str, Any]) -> str: + payload_json = json.dumps(payload, ensure_ascii=False) + return textwrap.dedent( + f""" + import csv + import json + import os + import re + import sys + import time + import traceback + import urllib.parse + import urllib.request + import urllib.robotparser + from datetime import datetime, timezone + + PAYLOAD = json.loads({payload_json!r}) + CHALLENGE_PATTERNS = [ + r"captcha", r"recaptcha", r"hcaptcha", r"cf-challenge", r"cloudflare", r"turnstile", + r"are you human", r"verify you are human", r"bot detection", r"automated traffic", + r"access denied", r"temporarily blocked", r"unusual traffic", r"login required", + r"sign in to continue", r"subscribe to continue", r"paywall", r"forbidden", + ] + PAGINATION_KEYS = {{"next", "_next", "__next__", "pagination_next", "next_page"}} + + def emit(payload): + print(json.dumps(payload, ensure_ascii=False)) + + def safe_name(value): + value = re.sub(r"[^a-zA-Z0-9._-]+", "-", value).strip("-._") + return value[:80] or "page" + + def same_origin(a, b): + pa, pb = urllib.parse.urlparse(a), urllib.parse.urlparse(b) + return pa.scheme == pb.scheme and pa.netloc.lower() == pb.netloc.lower() + + def robots_allowed(url, user_agent): + parsed = urllib.parse.urlparse(url) + robots_url = urllib.parse.urlunparse((parsed.scheme, parsed.netloc, "/robots.txt", "", "", "")) + rp = urllib.robotparser.RobotFileParser() + rp.set_url(robots_url) + try: + req = urllib.request.Request(robots_url, headers={{"User-Agent": user_agent}}) + with urllib.request.urlopen(req, timeout=10) as response: + body = response.read(2000000).decode("utf-8", errors="ignore") + rp.parse(body.splitlines()) + return bool(rp.can_fetch(user_agent, url)), None + except Exception as exc: + return False, f"robots.txt could not be checked for pagination URL {{url}}: {{type(exc).__name__}}: {{exc}}" + + def split_selector(selector): + selector = selector.strip() + m = re.search(r"::attr\\(([^)]+)\\)\\s*$", selector) + if m: + return selector[:m.start()].strip(), "attr", m.group(1).strip() + if selector.endswith("::text"): + return selector[:-6].strip(), "text", None + if selector.endswith("::html"): + return selector[:-6].strip(), "html", None + return selector, "text", None + + def detect_challenge(page_source, title, current_url): + haystack = "\\n".join([title or "", current_url or "", page_source[:50000] or ""]).lower() + matched = [pat for pat in CHALLENGE_PATTERNS if re.search(pat, haystack, re.I)] + if matched: + return "manual_intervention_required", "challenge/access-control indicator detected: " + ", ".join(sorted(set(matched))[:8]) + return None, None + + def extract_with_bs4(page_source, selectors): + from bs4 import BeautifulSoup + soup = BeautifulSoup(page_source, "html.parser") + item = {{}} + next_url = None + for key, raw_selector in selectors.items(): + css, mode, attr = split_selector(str(raw_selector)) + try: + elements = soup.select(css) + except Exception as exc: + item[key] = None + item.setdefault("_selector_warnings", []).append(f"selector {{key}}={{raw_selector!r}} failed: {{exc}}") + continue + values = [] + for el in elements: + if mode == "attr": + values.append(el.get(attr)) + elif mode == "html": + values.append(el.decode_contents()) + else: + values.append(el.get_text(" ", strip=True)) + values = [v for v in values if v is not None] + if key in PAGINATION_KEYS: + next_url = values[0] if values else None + else: + item[key] = values if len(values) != 1 else values[0] if values else None + return item, next_url + + def write_outputs(items, pages_visited, html_captures, screenshots, warnings, status): + out_dir = PAYLOAD["output_dir"] + os.makedirs(out_dir, exist_ok=True) + stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ") + base = safe_name(urllib.parse.urlparse(PAYLOAD["url"]).netloc) + "-" + stamp + metadata = {{ + "status": status, + "items": items, + "pages_visited": pages_visited, + "html_captures": html_captures, + "screenshots": screenshots, + "warnings": warnings, + "safety": "Authorized public/user-owned scraping only. No CAPTCHA, anti-bot, login, paywall, robots.txt, ToS, or access-control circumvention attempted.", + }} + json_path = os.path.join(out_dir, base + ".json") + with open(json_path, "w", encoding="utf-8") as f: + json.dump(metadata, f, ensure_ascii=False, indent=2) + saved_paths = [json_path] + if PAYLOAD["output_format"] == "csv": + csv_path = os.path.join(out_dir, base + ".csv") + fieldnames = sorted({{k for item in items for k in item.keys() if k != "_selector_warnings"}}) + with open(csv_path, "w", newline="", encoding="utf-8") as f: + writer = csv.DictWriter(f, fieldnames=fieldnames or ["url"]) + writer.writeheader() + for item in items: + row = {{k: (json.dumps(v, ensure_ascii=False) if isinstance(v, (list, dict)) else v) for k, v in item.items() if k in (fieldnames or ["url"])}} + writer.writerow(row) + saved_paths.append(csv_path) + return saved_paths + + def run(): + warnings = [] + items = [] + pages_visited = [] + html_captures = [] + screenshots = [] + status = "ok" + out_dir = PAYLOAD["output_dir"] + os.makedirs(out_dir, exist_ok=True) + current = PAYLOAD["url"] + start = current + + try: + from seleniumbase import SB + browser_kwargs = {{ + "headless": bool(PAYLOAD["headless"]), + "uc": False, + "test": False, + "locale_code": "en", + }} + if PAYLOAD.get("user_agent"): + browser_kwargs["agent"] = PAYLOAD["user_agent"] + with SB(**browser_kwargs) as sb: + for page_index in range(int(PAYLOAD["max_pages"])): + if not same_origin(start, current): + warnings.append(f"same-domain safeguard stopped pagination to {{current}}") + break + if PAYLOAD.get("respect_robots"): + allowed, robot_warning = robots_allowed(current, PAYLOAD.get("user_agent") or "*") + if robot_warning: + warnings.append(robot_warning) + if not allowed: + status = "blocked_by_robots_txt" + warnings.append(f"robots.txt disallows pagination URL {{current}}") + break + if page_index > 0 and float(PAYLOAD["rate_limit_seconds"]) > 0: + time.sleep(float(PAYLOAD["rate_limit_seconds"])) + + sb.open(current) + if float(PAYLOAD["wait_seconds"]) > 0: + sb.sleep(float(PAYLOAD["wait_seconds"])) + current_url = sb.get_current_url() + title = sb.get_title() + page_source = sb.get_page_source() + pages_visited.append(current_url) + + challenge_status, challenge_warning = detect_challenge(page_source, title, current_url) + if challenge_status: + status = challenge_status + warnings.append(challenge_warning) + html_path = os.path.join(out_dir, f"page-{{page_index+1}}-challenge.html") + with open(html_path, "w", encoding="utf-8") as f: + f.write(page_source) + html_captures.append(html_path) + if PAYLOAD.get("screenshot"): + shot_path = os.path.join(out_dir, f"page-{{page_index+1}}-challenge.png") + try: + sb.save_screenshot(shot_path) + screenshots.append(shot_path) + except Exception as exc: + warnings.append(f"screenshot failed: {{exc}}") + break + + html_path = os.path.join(out_dir, f"page-{{page_index+1}}.html") + with open(html_path, "w", encoding="utf-8") as f: + f.write(page_source) + html_captures.append(html_path) + if PAYLOAD.get("screenshot"): + shot_path = os.path.join(out_dir, f"page-{{page_index+1}}.png") + try: + sb.save_screenshot(shot_path) + screenshots.append(shot_path) + except Exception as exc: + warnings.append(f"screenshot failed: {{exc}}") + + item, next_raw = extract_with_bs4(page_source, PAYLOAD["selectors"]) + item["url"] = current_url + item["page_index"] = page_index + 1 + if item.get("_selector_warnings"): + warnings.extend(item.pop("_selector_warnings")) + items.append(item) + + if page_index >= int(PAYLOAD["max_pages"]) - 1 or not next_raw: + break + next_url = urllib.parse.urljoin(current_url, str(next_raw)) + if not same_origin(start, next_url): + warnings.append(f"same-domain safeguard refused pagination URL {{next_url}}") + break + current = next_url + except Exception as exc: + status = "error" + warnings.append(f"browser automation failed: {{type(exc).__name__}}: {{exc}}") + warnings.append(traceback.format_exc()[-4000:]) + + saved_paths = write_outputs(items, pages_visited, html_captures, screenshots, warnings, status) + emit({{ + "status": status, + "items": items, + "saved_paths": saved_paths, + "screenshots": screenshots, + "html_captures": html_captures, + "warnings": warnings, + "pages_visited": pages_visited, + }}) + return 0 if status in {{"ok", "blocked_by_robots_txt", "manual_intervention_required"}} else 1 + + if __name__ == "__main__": + sys.exit(run()) + """ + ) + + +def _parse_last_json(stdout: str) -> dict[str, Any] | None: + for line in reversed(stdout.splitlines()): + line = line.strip() + if not line.startswith("{"): + continue + try: + value = json.loads(line) + except json.JSONDecodeError: + continue + if isinstance(value, dict): + return value + return None + + +async def _emit_workspace_artifact_if_available(ctx: RunContext[Any], path: str) -> None: + try: + reader = getattr(ctx.workspace, "read_bytes", None) + if reader is None: + return + data = reader(path) + if not isinstance(data, (bytes, bytearray)): + return + mime = "application/json" + if path.endswith(".csv"): + mime = "text/csv" + elif path.endswith(".png"): + mime = "image/png" + elif path.endswith(".html"): + mime = "text/html" + ref = await ctx.write_artifact(Path(path).name, bytes(data), mime) + await ctx.emit_artifact(ref) + except Exception: # noqa: BLE001 + return