Files
agent-compliance-auditor/agent.py
2026-07-13 03:15:39 +00:00

619 lines
37 KiB
Python

"""Deterministic read-only compliance auditor for A2A agents."""
from __future__ import annotations
import hashlib
import json
import re
import time
from datetime import UTC, datetime
from enum import Enum
from fnmatch import fnmatch
from typing import Any, Literal
from urllib.parse import urlparse
import httpx
from pydantic import AnyHttpUrl, BaseModel, ConfigDict, Field, field_validator
import a2a_pack as a2a
from a2a_pack import (
A2AAgent,
ConsumerSetup,
ConsumerSetupField,
EgressPolicy,
NoAuth,
Pricing,
Resources,
RunContext,
WorkspaceAccess,
WorkspaceMode,
)
from a2a_pack.context import AgentEvent
from a2a_pack.workspace import FileType
MAX_TEXT_BYTES = 256_000
MAX_EVIDENCE_BYTES = 8_000
MAX_FILES_PER_AUDIT = 80
MAX_FINDINGS = 200
HTTP_TIMEOUT_SECONDS = 8.0
DENIED_REPO_GLOBS = (".git/**", "**/.env", "**/.env.*", "**/*secret*", "**/*credential*", "**/id_rsa", "**/*.pem", "**/*.key")
ALLOWED_REPO_GLOBS = ("agent.py", "a2a.yaml", "requirements.txt", "README.md", "pyproject.toml", "Dockerfile", "*.py", "*.yaml", "*.yml", "*.json", "*.md", "tests/*.py", "tests/**/*.py", "skills/**/SKILL.md")
SECRET_PATTERNS = (
re.compile(r"(?i)(api[_-]?key|token|secret|password)\s*[:=]\s*['\"]?([A-Za-z0-9_./+=-]{12,})"),
re.compile(r"sk-[A-Za-z0-9]{20,}"),
re.compile(r"(?i)bearer\s+[A-Za-z0-9_./+=-]{16,}"),
)
WRITE_INDICATORS = ("kubectl apply", "kubectl delete", "kubectl patch", "kubectl scale", "helm upgrade", "git push", "ctx.secret(", "OPENAI_API_KEY", "A2A_LITELLM_KEY")
class StrictModel(BaseModel):
model_config = ConfigDict(extra="forbid")
class Severity(str, Enum):
CRITICAL = "critical"
HIGH = "high"
MEDIUM = "medium"
LOW = "low"
INFO = "info"
class Confidence(str, Enum):
HIGH = "high"
MEDIUM = "medium"
LOW = "low"
class ProbeMode(str, Enum):
NONE = "none"
HEALTH_ONLY = "health_only"
SYNTHETIC = "synthetic"
class AuditTarget(StrictModel):
agent_name: str = Field(min_length=1, max_length=128, pattern=r"^[a-z0-9][a-z0-9-]{0,127}$")
agent_url: AnyHttpUrl | None = None
repository: str | None = Field(default=None, max_length=256)
namespace: str | None = Field(default=None, max_length=128)
deployment: str | None = Field(default=None, max_length=128)
@field_validator("repository", "namespace", "deployment")
@classmethod
def clean_optional(cls, value: str | None) -> str | None:
if value is None:
return None
return value.strip() or None
class AuditOptions(StrictModel):
audit_id: str | None = Field(default=None, max_length=80, pattern=r"^[A-Za-z0-9_.-]+$")
max_files: int = Field(default=40, ge=0, le=MAX_FILES_PER_AUDIT)
probe_mode: ProbeMode = ProbeMode.HEALTH_ONLY
include_repository: bool = True
include_runtime: bool = True
include_remediation: bool = True
allowed_hosts: list[str] = Field(default_factory=list, max_length=20)
allowed_namespaces: list[str] = Field(default_factory=list, max_length=20)
class Finding(StrictModel):
id: str
control_id: str
title: str
severity: Severity
confidence: Confidence
status: Literal["open", "passed", "unknown", "not_tested"] = "open"
evidence_refs: list[str] = Field(default_factory=list, max_length=20)
impact: str
remediation: str
verification_procedure: str
class Evidence(StrictModel):
id: str
source: str
path: str | None = None
status: Literal["observed", "missing", "unknown", "not_tested", "redacted"]
excerpt: str = Field(default="", max_length=MAX_EVIDENCE_BYTES)
sha256: str | None = None
class OutputFile(StrictModel):
path: str
artifact_uri: str | None = None
mime_type: str
size_bytes: int = Field(ge=0)
class InventoryRequest(StrictModel):
target: AuditTarget
options: AuditOptions = Field(default_factory=AuditOptions)
class AuditConfigurationRequest(StrictModel):
target: AuditTarget
options: AuditOptions = Field(default_factory=AuditOptions)
inventory: dict[str, Any] | None = None
class AuditRuntimeRequest(StrictModel):
target: AuditTarget
options: AuditOptions = Field(default_factory=AuditOptions)
inventory: dict[str, Any] | None = None
class AssessRiskRequest(StrictModel):
target: AuditTarget
options: AuditOptions = Field(default_factory=AuditOptions)
inventory: dict[str, Any] | None = None
configuration_findings: list[Finding] = Field(default_factory=list, max_length=MAX_FINDINGS)
runtime_findings: list[Finding] = Field(default_factory=list, max_length=MAX_FINDINGS)
class GenerateRemediationRequest(StrictModel):
target: AuditTarget
options: AuditOptions = Field(default_factory=AuditOptions)
findings: list[Finding] = Field(default_factory=list, max_length=MAX_FINDINGS)
risk_summary: dict[str, Any] | None = None
class InventoryResult(StrictModel):
audit_id: str
target: AuditTarget
status: Literal["ok", "partial", "unreachable", "setup_required"]
output_dir: str
generated_files: list[OutputFile] = Field(default_factory=list)
inventory: dict[str, Any]
evidence: list[Evidence] = Field(default_factory=list)
warnings: list[str] = Field(default_factory=list)
mutations_performed: bool = False
class FindingResult(StrictModel):
audit_id: str
target: AuditTarget
status: Literal["ok", "partial", "setup_required"]
output_dir: str
generated_files: list[OutputFile] = Field(default_factory=list)
findings: list[Finding]
evidence: list[Evidence] = Field(default_factory=list)
warnings: list[str] = Field(default_factory=list)
mutations_performed: bool = False
class RiskSummaryResult(StrictModel):
audit_id: str
target: AuditTarget
status: Literal["ok", "partial"]
output_dir: str
generated_files: list[OutputFile] = Field(default_factory=list)
risk_summary: dict[str, Any]
findings: list[Finding]
warnings: list[str] = Field(default_factory=list)
mutations_performed: bool = False
class RemediationResult(StrictModel):
audit_id: str
target: AuditTarget
status: Literal["ok", "partial"]
output_dir: str
generated_files: list[OutputFile] = Field(default_factory=list)
remediation_markdown: str
sarif: dict[str, Any]
warnings: list[str] = Field(default_factory=list)
mutations_performed: bool = False
class AgentComplianceAuditorConfig(StrictModel):
default_allowed_hosts: list[str] = Field(default_factory=list, max_length=50)
default_allowed_namespaces: list[str] = Field(default_factory=list, max_length=50)
class AgentComplianceAuditor(A2AAgent[AgentComplianceAuditorConfig, NoAuth]):
name = "agent-compliance-auditor"
description = "Read-only compliance auditor for deployed A2A agents and repositories."
version = "0.1.0"
config_model = AgentComplianceAuditorConfig
auth_model = NoAuth
pricing = Pricing(price_per_call_usd=0.0, caller_pays_llm=False, notes="Deterministic read-only auditing; no LLM calls and no production mutations.")
resources = Resources(cpu="500m", memory="512Mi", max_runtime_seconds=900)
workspace_access = WorkspaceAccess.dynamic(max_files=MAX_FILES_PER_AUDIT + 10, allowed_modes=(WorkspaceMode.READ_ONLY, WorkspaceMode.READ_WRITE_OVERLAY), require_reason=False, deny_patterns=DENIED_REPO_GLOBS, max_total_size_bytes=12 * 1024 * 1024)
egress = EgressPolicy(allow_hosts=("gitea.a2a.svc.cluster.local", "argocd-server.argocd.svc.cluster.local", "kubernetes.default.svc", "registry-1.docker.io", "ghcr.io"), deny_internet_by_default=True)
tools_used = ("httpx", "pydantic", "workspace")
consumer_setup = ConsumerSetup.from_fields(
ConsumerSetupField.config("GITEA_BASE_URL", label="Gitea base URL", required=False, input_type="url"),
ConsumerSetupField.secret("GITEA_TOKEN", label="Gitea read-only token", required=False),
ConsumerSetupField.config("KUBERNETES_BASE_URL", label="Kubernetes API base URL", required=False, input_type="url"),
ConsumerSetupField.secret("KUBERNETES_TOKEN", label="Kubernetes read-only bearer token", required=False),
ConsumerSetupField.config("ARGOCD_BASE_URL", label="Argo CD base URL", required=False, input_type="url"),
ConsumerSetupField.secret("ARGOCD_TOKEN", label="Argo CD read-only token", required=False),
ConsumerSetupField.config("REGISTRY_BASE_URL", label="Registry metadata base URL", required=False, input_type="url"),
ConsumerSetupField.secret("REGISTRY_TOKEN", label="Registry read-only token", required=False),
)
@a2a.tool(description="Inventory a deployed A2A agent card, repository evidence, manifests, exposure, provenance, resources, and health", timeout_seconds=180, idempotent=True, cost_class="read-only")
async def inventory_agent(self, ctx: RunContext[NoAuth], request: InventoryRequest) -> InventoryResult:
audit_id = _audit_id(request.target, request.options.audit_id)
await _event(ctx, "audit_started", audit_id=audit_id, skill="inventory_agent", target=request.target.agent_name)
inventory, evidence, warnings, status = await _collect_inventory(self, ctx, request.target, request.options, audit_id)
output_dir = _output_dir(audit_id)
files = await _write_outputs(ctx, output_dir, {"inventory.json": inventory}, warnings)
await _event(ctx, "audit_completed", audit_id=audit_id, skill="inventory_agent", findings=0)
return InventoryResult(audit_id=audit_id, target=request.target, status=status, output_dir=output_dir, generated_files=files, inventory=inventory, evidence=evidence, warnings=warnings)
@a2a.tool(description="Audit static configuration and repository files against A2A security, privacy, and platform controls", timeout_seconds=180, idempotent=True, cost_class="read-only")
async def audit_configuration(self, ctx: RunContext[NoAuth], request: AuditConfigurationRequest) -> FindingResult:
audit_id = _audit_id(request.target, request.options.audit_id)
await _event(ctx, "audit_started", audit_id=audit_id, skill="audit_configuration", target=request.target.agent_name)
inventory = request.inventory or (await _collect_inventory(self, ctx, request.target, request.options, audit_id))[0]
findings, evidence, warnings = _configuration_findings(request.target, inventory)
findings = _dedupe(findings)
output_dir = _output_dir(audit_id)
files = await _write_outputs(ctx, output_dir, {"findings.json": {"findings": [_dump(f) for f in findings]}}, warnings)
await _event(ctx, "audit_completed", audit_id=audit_id, skill="audit_configuration", findings=len(findings))
return FindingResult(audit_id=audit_id, target=request.target, status="ok", output_dir=output_dir, generated_files=files, findings=findings, evidence=evidence, warnings=warnings)
@a2a.tool(description="Run bounded non-destructive runtime probes and audit deployment health without mutations", timeout_seconds=180, idempotent=True, cost_class="read-only")
async def audit_runtime(self, ctx: RunContext[NoAuth], request: AuditRuntimeRequest) -> FindingResult:
audit_id = _audit_id(request.target, request.options.audit_id)
await _event(ctx, "audit_started", audit_id=audit_id, skill="audit_runtime", target=request.target.agent_name)
inventory = request.inventory or (await _collect_inventory(self, ctx, request.target, request.options, audit_id))[0]
findings, evidence, warnings = _runtime_findings(request.target, inventory)
findings = _dedupe(findings)
output_dir = _output_dir(audit_id)
files = await _write_outputs(ctx, output_dir, {"findings.json": {"findings": [_dump(f) for f in findings]}}, warnings)
await _event(ctx, "audit_completed", audit_id=audit_id, skill="audit_runtime", findings=len(findings))
return FindingResult(audit_id=audit_id, target=request.target, status="ok", output_dir=output_dir, generated_files=files, findings=findings, evidence=evidence, warnings=warnings)
@a2a.tool(description="Deduplicate findings, map evidence to controls, and score severity, confidence, and aggregate risk", timeout_seconds=120, idempotent=True, cost_class="read-only")
async def assess_risk(self, ctx: RunContext[NoAuth], request: AssessRiskRequest) -> RiskSummaryResult:
audit_id = _audit_id(request.target, request.options.audit_id)
await _event(ctx, "audit_started", audit_id=audit_id, skill="assess_risk", target=request.target.agent_name)
all_findings = _dedupe([*request.configuration_findings, *request.runtime_findings])
summary = _risk_summary(all_findings)
output_dir = _output_dir(audit_id)
files = await _write_outputs(ctx, output_dir, {"risk-summary.json": summary, "findings.json": {"findings": [_dump(f) for f in all_findings]}}, [])
await _event(ctx, "audit_completed", audit_id=audit_id, skill="assess_risk", findings=len(all_findings), score=summary["risk_score"])
return RiskSummaryResult(audit_id=audit_id, target=request.target, status="ok", output_dir=output_dir, generated_files=files, risk_summary=summary, findings=all_findings)
@a2a.tool(description="Generate actionable remediation and a machine-readable SARIF report for compliance findings", timeout_seconds=120, idempotent=True, cost_class="read-only")
async def generate_remediation(self, ctx: RunContext[NoAuth], request: GenerateRemediationRequest) -> RemediationResult:
audit_id = _audit_id(request.target, request.options.audit_id)
await _event(ctx, "audit_started", audit_id=audit_id, skill="generate_remediation", target=request.target.agent_name)
findings = _dedupe(request.findings)
risk_summary = request.risk_summary or _risk_summary(findings)
markdown = _remediation_md(request.target, audit_id, findings, risk_summary)
sarif = _sarif(request.target, findings)
output_dir = _output_dir(audit_id)
files = await _write_outputs(ctx, output_dir, {"remediation.md": markdown, "sarif.json": sarif, "risk-summary.json": risk_summary}, [])
await _event(ctx, "audit_completed", audit_id=audit_id, skill="generate_remediation", findings=len(findings))
return RemediationResult(audit_id=audit_id, target=request.target, status="ok", output_dir=output_dir, generated_files=files, remediation_markdown=markdown, sarif=sarif)
async def _collect_inventory(agent: AgentComplianceAuditor, ctx: RunContext[NoAuth], target: AuditTarget, options: AuditOptions, audit_id: str) -> tuple[dict[str, Any], list[Evidence], list[str], Literal["ok", "partial", "unreachable", "setup_required"]]:
warnings: list[str] = []
evidence: list[Evidence] = []
allowed_hosts = {h.lower().strip() for h in [*agent.config.default_allowed_hosts, *options.allowed_hosts] if h.strip()}
allowed_namespaces = {n.strip() for n in [*agent.config.default_allowed_namespaces, *options.allowed_namespaces] if n.strip()}
authorized = not allowed_namespaces or bool(target.namespace and target.namespace in allowed_namespaces)
status: Literal["ok", "partial", "unreachable", "setup_required"] = "ok" if authorized else "partial"
if not authorized:
warnings.append("Target namespace is outside the configured allowlist; runtime probes skipped.")
await _event(ctx, "audit_progress", audit_id=audit_id, phase="agent_card")
card = await _fetch_card(str(target.agent_url) if target.agent_url else None, allowed_hosts, warnings)
if card.get("status") == "unreachable" and status == "ok":
status = "unreachable"
evidence.append(_evidence("card", "agent_card", "observed" if card.get("card") else "unknown", card))
await _event(ctx, "audit_progress", audit_id=audit_id, phase="repository_static_inspection")
repo_files = await _read_repo(ctx, options.max_files if options.include_repository else 0, warnings)
evidence.extend(_repo_evidence(repo_files))
manifests = _manifests(repo_files)
health = {"status": "not_tested", "reason": "probe_mode is none"}
if options.probe_mode is not ProbeMode.NONE and authorized:
await _event(ctx, "audit_progress", audit_id=audit_id, phase="bounded_runtime_probe")
health = await _health(str(target.agent_url) if target.agent_url else None, allowed_hosts, warnings)
inventory = {
"schema_version": "2026-07-13",
"audit_id": audit_id,
"generated_at": datetime.now(UTC).isoformat(),
"identity": {"agent_name": target.agent_name, "agent_url": str(target.agent_url) if target.agent_url else None, "repository": target.repository, "namespace": target.namespace, "deployment": target.deployment, "authorized": authorized},
"agent_card": card,
"skills": _skills(card.get("card"), repo_files),
"schemas": _schemas(card.get("card")),
"manifests": manifests,
"rbac": _rbac(manifests),
"secrets_references": _secret_refs(repo_files),
"network_exposure": _exposure(manifests),
"image_provenance": _provenance(manifests, repo_files),
"resource_limits": _resources(card.get("card"), manifests),
"deployment_health": health,
"repository": _repo_summary(repo_files),
"controls": _controls(),
"unknowns": _unknowns(card, manifests, health, repo_files),
"mutations_performed": False,
}
return inventory, evidence, warnings, status
async def _fetch_card(agent_url: str | None, allowed_hosts: set[str], warnings: list[str]) -> dict[str, Any]:
if not agent_url:
return {"status": "unknown", "reason": "agent_url not provided"}
parsed = urlparse(agent_url)
if not _safe_url(parsed, allowed_hosts):
warnings.append("Agent URL host is not allowlisted or is unsafe; card fetch skipped.")
return {"status": "not_tested", "reason": "host_not_allowlisted"}
try:
async with httpx.AsyncClient(timeout=httpx.Timeout(HTTP_TIMEOUT_SECONDS), follow_redirects=False) as client:
resp = await client.get(agent_url.rstrip("/") + "/.well-known/agent-card", headers={"accept": "application/json"})
if resp.status_code >= 400:
return {"status": "unreachable", "http_status": resp.status_code, "body_excerpt": _redact(resp.text[:500])}
return {"status": "ok", "card": _bounded_json(resp.text, warnings)}
except Exception as exc: # noqa: BLE001
warnings.append(f"Agent card fetch failed: {type(exc).__name__}")
return {"status": "unreachable", "error_type": type(exc).__name__}
async def _health(agent_url: str | None, allowed_hosts: set[str], warnings: list[str]) -> dict[str, Any]:
if not agent_url:
return {"status": "unknown", "reason": "agent_url not provided"}
if not _safe_url(urlparse(agent_url), allowed_hosts):
return {"status": "not_tested", "reason": "host_not_allowlisted"}
started = time.perf_counter()
try:
async with httpx.AsyncClient(timeout=httpx.Timeout(HTTP_TIMEOUT_SECONDS), follow_redirects=False) as client:
resp = await client.get(agent_url.rstrip("/") + "/healthz")
return {"status": "ok" if resp.status_code < 500 else "degraded", "http_status": resp.status_code, "latency_ms": round((time.perf_counter() - started) * 1000, 2)}
except Exception as exc: # noqa: BLE001
warnings.append(f"Health probe failed: {type(exc).__name__}")
return {"status": "unreachable", "error_type": type(exc).__name__}
async def _read_repo(ctx: RunContext[NoAuth], max_files: int, warnings: list[str]) -> dict[str, str]:
if max_files <= 0:
return {}
try:
view = await ctx.workspace.open_view(purpose="Read-only static inspection of A2A agent source files", hints=("agent.py a2a.yaml requirements README tests skills",), file_types=(FileType.PYTHON, FileType.YAML, FileType.JSON, FileType.MARKDOWN, FileType.TOML), max_files=max_files, mode=WorkspaceMode.READ_ONLY, reason="Compliance audit reads bounded source files and never writes repository paths.")
except Exception as exc: # noqa: BLE001
warnings.append(f"Repository workspace unavailable: {type(exc).__name__}")
return {}
files: dict[str, str] = {}
for match in view.files[:max_files]:
path = match.path.replace("\\", "/").lstrip("/")
if not _path_allowed(path):
continue
try:
files[path] = (await view.read(path))[:MAX_TEXT_BYTES].decode("utf-8", errors="replace")
except Exception as exc: # noqa: BLE001
warnings.append(f"Could not read {path}: {type(exc).__name__}")
return files
async def _write_outputs(ctx: RunContext[NoAuth], output_dir: str, files: dict[str, Any], warnings: list[str]) -> list[OutputFile]:
out: list[OutputFile] = []
for name, payload in files.items():
rel = f"{output_dir}/{name}"
data = payload.encode("utf-8") if isinstance(payload, str) else json.dumps(payload, indent=2, sort_keys=True, default=str).encode("utf-8")
mime = "text/markdown" if name.endswith(".md") else "application/json"
uri = None
try:
ref = await ctx.write_artifact(rel, data, mime)
await ctx.emit_artifact(ref)
uri = ref.uri
except Exception as exc: # noqa: BLE001
warnings.append(f"Artifact write failed for {rel}: {type(exc).__name__}")
out.append(OutputFile(path=rel, artifact_uri=uri, mime_type=mime, size_bytes=len(data)))
return out
def _configuration_findings(target: AuditTarget, inventory: dict[str, Any]) -> tuple[list[Finding], list[Evidence], list[str]]:
findings: list[Finding] = []
card = (inventory.get("agent_card") or {}).get("card") or {}
skills = inventory.get("skills") or []
if not card:
findings.append(_finding(target, "A2A-CARD-001", "Agent Card could not be verified", Severity.HIGH, Confidence.MEDIUM, "card", "Consumers cannot validate skills, schemas, runtime, or billing metadata.", "Ensure /.well-known/agent-card is reachable over an approved host and returns a valid A2A Card.", "Fetch /.well-known/agent-card and validate metadata."))
if not skills:
findings.append(_finding(target, "A2A-SCHEMA-001", "No public skills discovered", Severity.HIGH, Confidence.MEDIUM, "skills", "Callers cannot rely on a stable typed API contract.", "Expose typed @a2a.tool methods with input and output schemas.", "Reload the live Agent Card and confirm skills[].input_schema and output_schema."))
for skill in skills:
schema = skill.get("input_schema") or skill.get("inputSchema") or {}
out_schema = skill.get("output_schema") or skill.get("outputSchema") or {}
if not (isinstance(schema, dict) and schema.get("type") == "object" and schema.get("additionalProperties") is False) or not out_schema:
findings.append(_finding(target, "A2A-SCHEMA-002", f"Skill {skill.get('name', '<unknown>')} has weak schema", Severity.MEDIUM, Confidence.HIGH, "schemas", "Weak schemas increase integration errors and injection surface.", "Use strict Pydantic models or explicit JSON Schema with additionalProperties=false.", "Inspect the Agent Card schemas and run negative validation tests."))
if (inventory.get("network_exposure") or {}).get("public") is True:
findings.append(_finding(target, "A2A-NET-001", "Agent is publicly exposed", Severity.MEDIUM, Confidence.MEDIUM, "network_exposure", "Unexpected public exposure can broaden abuse and exfiltration paths.", "Set expose.public=false unless public access is required and documented.", "Review a2a.yaml and ingress policy."))
if (inventory.get("resource_limits") or {}).get("max_runtime_seconds") in (None, "unknown"):
findings.append(_finding(target, "A2A-OPS-001", "Runtime limit is unknown", Severity.LOW, Confidence.MEDIUM, "resource_limits", "Unknown runtime budgets can cause reliability and cost issues.", "Declare Resources and mirror runtime.resources in a2a.yaml.", "Load Agent Card runtime.resources and deployment manifest."))
if inventory.get("secrets_references"):
findings.append(_finding(target, "A2A-SECRET-001", "Potential secret or provider-key reference found in source", Severity.HIGH, Confidence.HIGH, "secrets_references", "Secrets in source or logs can be exfiltrated.", "Remove literal secrets and broad provider-key reads; use declared consumer setup or runtime secrets.", "Run secret scanning and verify no artifacts contain secret values."))
if (inventory.get("repository") or {}).get("suspicious_instruction_count", 0) > 0:
findings.append(_finding(target, "A2A-UNTRUSTED-001", "Repository contains malicious instruction-like text", Severity.MEDIUM, Confidence.HIGH, "repository", "Auditors or LLM-backed tools could be manipulated if repo text is treated as instructions.", "Keep repository content in evidence fields only; never treat it as instructions.", "Confirm malicious text appears only as redacted evidence."))
return findings[:MAX_FINDINGS], [_evidence("configuration", "configuration_rules", "observed", {"finding_count": len(findings)})], []
def _runtime_findings(target: AuditTarget, inventory: dict[str, Any]) -> tuple[list[Finding], list[Evidence], list[str]]:
findings: list[Finding] = []
health = inventory.get("deployment_health") or {"status": "unknown"}
if (inventory.get("identity") or {}).get("authorized") is False:
findings.append(_finding(target, "A2A-TENANT-001", "Runtime audit skipped by namespace allowlist", Severity.HIGH, Confidence.HIGH, "identity", "Auditing outside the authorized tenant boundary could violate isolation policy.", "Configure allowed_namespaces for the target tenant or run from the owning tenant.", "Confirm namespace allowlist contains only tenant-owned namespaces and retry."))
if health.get("status") in {"unreachable", "degraded"}:
findings.append(_finding(target, "A2A-HEALTH-001", "Agent health probe failed or degraded", Severity.MEDIUM, Confidence.MEDIUM, "deployment_health", "Consumers may experience failed calls or stale Agent Cards.", "Inspect deployment status, readiness probes, service routing, and rollout events.", "Repeat health-only probe and verify readiness."))
if health.get("status") in {"unknown", "not_tested"}:
findings.append(_finding(target, "A2A-HEALTH-002", "Runtime health state is unknown", Severity.LOW, Confidence.MEDIUM, "deployment_health", "Unknown runtime state leaves operational risk unmeasured.", "Provide an allowlisted agent_url and enable health_only probes.", "Run audit_runtime with probe_mode=health_only."))
return findings[:MAX_FINDINGS], [_evidence("runtime", "runtime_rules", "observed", {"finding_count": len(findings), "health": health})], []
def _finding(target: AuditTarget, control_id: str, title: str, severity: Severity, confidence: Confidence, evidence_ref: str, impact: str, remediation: str, verification: str) -> Finding:
suffix = hashlib.sha256(f"{target.agent_name}|{control_id}|{title}".encode()).hexdigest()[:10]
return Finding(id=f"{control_id.lower()}-{suffix}", control_id=control_id, title=title, severity=severity, confidence=confidence, evidence_refs=[evidence_ref], impact=impact, remediation=remediation, verification_procedure=verification)
def _dedupe(findings: list[Finding]) -> list[Finding]:
rank = {Severity.CRITICAL: 5, Severity.HIGH: 4, Severity.MEDIUM: 3, Severity.LOW: 2, Severity.INFO: 1}
seen: dict[str, Finding] = {}
for f in findings:
if f.id not in seen or rank[f.severity] > rank[seen[f.id].severity]:
seen[f.id] = f
return sorted(seen.values(), key=lambda f: (-rank[f.severity], f.control_id, f.id))[:MAX_FINDINGS]
def _risk_summary(findings: list[Finding]) -> dict[str, Any]:
weights = {Severity.CRITICAL: 100, Severity.HIGH: 40, Severity.MEDIUM: 15, Severity.LOW: 5, Severity.INFO: 1}
counts = {s.value: 0 for s in Severity}
score = 0
for f in findings:
counts[f.severity.value] += 1
score += weights[f.severity]
score = min(score, 100)
rating = "critical" if counts["critical"] else "high" if score >= 70 else "medium" if score >= 30 else "low" if score else "clean"
return {"schema_version": "2026-07-13", "risk_score": score, "risk_rating": rating, "finding_counts": counts, "top_controls": sorted({f.control_id for f in findings})[:20], "unknown_or_not_tested": [f.control_id for f in findings if f.status in {"unknown", "not_tested"}], "mutations_performed": False}
def _remediation_md(target: AuditTarget, audit_id: str, findings: list[Finding], summary: dict[str, Any]) -> str:
lines = [f"# Remediation plan for {target.agent_name}", "", f"Audit ID: `{audit_id}`", f"Risk rating: **{summary.get('risk_rating', 'unknown')}** ({summary.get('risk_score', 0)}/100)", "", "This plan is read-only output. No production systems or repositories were changed.", ""]
if not findings:
lines += ["## No open findings", "", "The synthetic audit found no policy violations in the supplied evidence. Unknown/not-tested areas should still be reviewed."]
for f in findings:
lines += [f"## {f.severity.value.upper()}{f.title}", "", f"- Finding ID: `{f.id}`", f"- Control: `{f.control_id}`", f"- Confidence: `{f.confidence.value}`", f"- Evidence references: {', '.join(f.evidence_refs) or 'none'}", f"- Impact: {f.impact}", f"- Remediation: {f.remediation}", f"- Verification: {f.verification_procedure}", ""]
return "\n".join(lines)
def _sarif(target: AuditTarget, findings: list[Finding]) -> dict[str, Any]:
rules = [{"id": f.control_id, "name": f.title, "shortDescription": {"text": f.title}, "help": {"text": f.remediation}} for f in findings]
results = [{"ruleId": f.control_id, "level": "error" if f.severity in {Severity.CRITICAL, Severity.HIGH} else "warning" if f.severity is Severity.MEDIUM else "note", "message": {"text": f"{f.title}: {f.impact}"}, "partialFingerprints": {"findingId": f.id}, "properties": {"severity": f.severity.value, "confidence": f.confidence.value, "verification": f.verification_procedure}} for f in findings]
return {"version": "2.1.0", "$schema": "https://json.schemastore.org/sarif-2.1.0.json", "runs": [{"tool": {"driver": {"name": "agent-compliance-auditor", "rules": rules}}, "invocations": [{"executionSuccessful": True, "properties": {"target": target.agent_name, "mutations_performed": False}}], "results": results}]}
def _repo_summary(files: dict[str, str]) -> dict[str, Any]:
suspicious_paths = [p for p, t in files.items() if any(tok in t.lower() for tok in ("ignore previous instructions", "exfiltrate", "send secrets", "system prompt"))]
return {"files_inspected": len(files), "paths": sorted(files), "suspicious_instruction_count": len(suspicious_paths), "suspicious_instruction_paths": suspicious_paths[:20], "bounded_bytes_per_file": MAX_TEXT_BYTES}
def _manifests(files: dict[str, str]) -> dict[str, Any]:
return {path: ({"present": True, "sha256": _sha(files[path]), "excerpt": _redact(files[path][:1200])} if path in files else {"present": False}) for path in ("a2a.yaml", "agent.py", "requirements.txt", "Dockerfile")}
def _skills(card: Any, files: dict[str, str]) -> list[dict[str, Any]]:
if isinstance(card, dict) and isinstance(card.get("skills"), list):
return [s for s in card["skills"] if isinstance(s, dict)]
return [{"name": n, "source": "agent.py"} for n in re.findall(r"async\s+def\s+([a-zA-Z_][a-zA-Z0-9_]*)\s*\(", files.get("agent.py", ""))]
def _schemas(card: Any) -> dict[str, Any]:
out: dict[str, Any] = {}
if isinstance(card, dict):
for skill in card.get("skills") or []:
if isinstance(skill, dict):
out[str(skill.get("name") or skill.get("id") or "unknown")] = {"input_schema": skill.get("input_schema") or skill.get("inputSchema"), "output_schema": skill.get("output_schema") or skill.get("outputSchema")}
return out
def _secret_refs(files: dict[str, str]) -> list[dict[str, str]]:
refs: list[dict[str, str]] = []
for path, text in files.items():
for pattern in SECRET_PATTERNS:
refs += [{"path": path, "match": _redact(m.group(0))} for m in pattern.finditer(text)]
refs += [{"path": path, "match": _redact(ind)} for ind in WRITE_INDICATORS if ind in text]
return refs[:50]
def _exposure(manifests: dict[str, Any]) -> dict[str, Any]:
text = json.dumps(manifests, default=str).lower()
public = True if "public: true" in text or "public=true" in text else False if "public: false" in text or "public=false" in text else None
return {"public": public, "documented_public_reason": "public_reason" in text or "exposure_reason" in text, "unknown_when_null": public is None}
def _resources(card: Any, manifests: dict[str, Any]) -> dict[str, Any]:
runtime = card.get("runtime") if isinstance(card, dict) else {}
resources = runtime.get("resources") if isinstance(runtime, dict) else {}
out = dict(resources or {}) if isinstance(resources, dict) else {}
if "max_runtime_seconds" not in out:
match = re.search(r"max_runtime_seconds[\s:=]+([0-9]+)", json.dumps(manifests, default=str))
out["max_runtime_seconds"] = int(match.group(1)) if match else "unknown"
return out
def _provenance(manifests: dict[str, Any], files: dict[str, str]) -> dict[str, Any]:
dockerfile = files.get("Dockerfile", "")
return {"dockerfile_present": bool(dockerfile), "digest_pinned_base_image": ("@sha256:" in dockerfile) if dockerfile else "unknown", "source_hashes": {k: v.get("sha256") for k, v in manifests.items() if isinstance(v, dict) and v.get("sha256")}}
def _rbac(manifests: dict[str, Any]) -> dict[str, Any]:
text = json.dumps(manifests, default=str).lower()
return {"service_account_mentions": text.count("serviceaccount"), "cluster_role_mentions": text.count("clusterrole"), "status": "observed" if "role" in text else "unknown"}
def _unknowns(card: dict[str, Any], manifests: dict[str, Any], health: dict[str, Any], files: dict[str, str]) -> list[str]:
out = []
if not card.get("card"):
out.append("agent_card")
if not files:
out.append("repository_files")
if health.get("status") in {"unknown", "not_tested"}:
out.append("deployment_health")
if not any(v.get("present") for v in manifests.values() if isinstance(v, dict)):
out.append("source_manifests")
return out
def _controls() -> list[dict[str, str]]:
return [{"control_id": cid, "title": title} for cid, title in [("A2A-CARD-001", "Agent Card availability and integrity"), ("A2A-SCHEMA-001", "Public skill inventory"), ("A2A-SCHEMA-002", "Strict non-null schemas"), ("A2A-NET-001", "Network exposure minimization"), ("A2A-OPS-001", "Runtime resource limits"), ("A2A-SECRET-001", "Secret handling and redaction"), ("A2A-UNTRUSTED-001", "Untrusted content isolation"), ("A2A-TENANT-001", "Tenant and namespace isolation"), ("A2A-HEALTH-001", "Deployment health")]]
def _repo_evidence(files: dict[str, str]) -> list[Evidence]:
return [_evidence(f"repo:{path}", "repository", "observed", text[:1000], path=path) for path, text in list(files.items())[:MAX_FILES_PER_AUDIT]]
def _evidence(eid: str, source: str, status: Literal["observed", "missing", "unknown", "not_tested", "redacted"], payload: Any, path: str | None = None) -> Evidence:
text = payload if isinstance(payload, str) else json.dumps(payload, sort_keys=True, default=str)
return Evidence(id=eid, source=source, path=path, status=status, excerpt=_redact(text)[:MAX_EVIDENCE_BYTES], sha256=_sha(text))
def _audit_id(target: AuditTarget, supplied: str | None) -> str:
return supplied or f"audit-{target.agent_name}-{hashlib.sha256(f'{target.agent_name}|{target.repository or ""}|{int(time.time())}'.encode()).hexdigest()[:12]}"
def _output_dir(audit_id: str) -> str:
return f"outputs/compliance/{re.sub(r'[^A-Za-z0-9_.-]', '-', audit_id)[:80]}"
def _path_allowed(path: str) -> bool:
clean = path.replace("\\", "/").lstrip("/")
return ".." not in clean.split("/") and not any(fnmatch(clean, p) for p in DENIED_REPO_GLOBS) and any(fnmatch(clean, p) for p in ALLOWED_REPO_GLOBS)
def _safe_url(parsed: Any, allowed_hosts: set[str]) -> bool:
host = (parsed.hostname or "").lower()
return parsed.scheme in {"https", "http"} and bool(host) and host not in {"localhost", "127.0.0.1", "0.0.0.0"} and not host.startswith("169.254.") and not host.endswith(".local") and host in allowed_hosts
def _bounded_json(text: str, warnings: list[str]) -> Any:
if len(text.encode("utf-8")) > MAX_TEXT_BYTES:
warnings.append("HTTP JSON response exceeded evidence bound and was truncated before parsing.")
text = text[:MAX_TEXT_BYTES]
try:
return json.loads(text)
except json.JSONDecodeError:
warnings.append("HTTP response was not valid JSON.")
return {"invalid_json_excerpt": _redact(text[:1000])}
def _redact(text: str) -> str:
out = text
for pattern in SECRET_PATTERNS:
out = pattern.sub(lambda m: m.group(0)[:8] + "…REDACTED", out)
return out
def _sha(text: str) -> str:
return hashlib.sha256(text.encode("utf-8", errors="replace")).hexdigest()
def _dump(value: Any) -> Any:
return value.model_dump(mode="json") if isinstance(value, BaseModel) else value
async def _event(ctx: RunContext[NoAuth], kind: str, **payload: Any) -> None:
safe = json.loads(json.dumps(payload, default=str))
await ctx.emit_event(AgentEvent(kind=kind, payload=safe))
if kind in {"audit_started", "audit_progress", "audit_completed"}:
await ctx.emit_progress(f"{kind}: {safe.get('phase') or safe.get('skill') or kind}")