diff --git a/abevalflow/mcp/__init__.py b/abevalflow/mcp/__init__.py new file mode 100644 index 00000000..766bef67 --- /dev/null +++ b/abevalflow/mcp/__init__.py @@ -0,0 +1,12 @@ +"""MCP server evaluation pipeline. + +Isolated from the skill evaluation pipeline (see ADR: Evaluation Strategy for +MCP Servers as Standalone Software Components, Approach 4). Evaluates an MCP +server as a black box across three sequenced phases: + +- Phase 1 - static / build-time analysis (no running server). +- Phase 2 - deterministic contract/conformance testing (live, no LLM). +- Phase 3 - behavioral testing (mcpchecker, conditional on a task suite). + +Phases 1 and 2 are implemented here; Phase 3 is out of scope for now. +""" diff --git a/abevalflow/mcp/phase1/__init__.py b/abevalflow/mcp/phase1/__init__.py new file mode 100644 index 00000000..9823ca2c --- /dev/null +++ b/abevalflow/mcp/phase1/__init__.py @@ -0,0 +1,67 @@ +"""MCP Phase 1 - static / build-time analysis. + +Runs three static checks against the MCP server's built artifact/source, before +any deployment and without running the server (ADR Approach 4, Phase 1): + +- Secrets management (SecretsGate) +- No user-provided code execution (NoUserCodeGate) +- License compliance (LicenseGate) + +Each gate subclasses SecurityGate and reads a normalized ``{"findings": [...]}`` +scan JSON emitted by the corresponding scanner in ``scripts/mcp/``. Phase 1 is +kept isolated from the skill pipeline's global security-gate registry +(``abevalflow.gates.security``): these gates are instantiated only via this +module's :func:`run_phase1`, so onboarding an MCP server does not pull skill +scanners and vice versa. + +Phase 1 always executes, independent of runtime outcomes, so its checks resolve +to pass/fail only (never the not-evaluated state used for Phase 2/3). +""" + +from __future__ import annotations + +from pathlib import Path + +from abevalflow.gates.base import GateResult +from abevalflow.gates.security.base import SecurityGate +from abevalflow.mcp.phase1.license import DEFAULT_ALLOWED_LICENSES, LicenseGate +from abevalflow.mcp.phase1.no_user_code import NoUserCodeGate +from abevalflow.mcp.phase1.secrets import SecretsGate +from abevalflow.schemas import GatePolicy + +# Ordered list of the Phase 1 gate classes, in execution order. +PHASE1_GATES: tuple[type[SecurityGate], ...] = ( + SecretsGate, + NoUserCodeGate, + LicenseGate, +) + + +def get_phase1_gates() -> list[SecurityGate]: + """Instantiate all Phase 1 gates, in order.""" + return [gate_cls() for gate_cls in PHASE1_GATES] + + +def run_phase1(reports_dir: Path, policy: GatePolicy) -> list[GateResult]: + """Evaluate all Phase 1 gates against the scan reports. + + Args: + reports_dir: Path to ``reports/{submission-name}/`` holding the + normalized scan JSON files written by ``scripts/mcp/*``. + policy: Gate policy controlling mode (disabled/warn/block) per gate. + + Returns: + One GateResult per Phase 1 gate, in :data:`PHASE1_GATES` order. + """ + return [gate.evaluate(reports_dir, policy) for gate in get_phase1_gates()] + + +__all__ = [ + "DEFAULT_ALLOWED_LICENSES", + "PHASE1_GATES", + "LicenseGate", + "NoUserCodeGate", + "SecretsGate", + "get_phase1_gates", + "run_phase1", +] diff --git a/abevalflow/mcp/phase1/license.py b/abevalflow/mcp/phase1/license.py new file mode 100644 index 00000000..fe984eba --- /dev/null +++ b/abevalflow/mcp/phase1/license.py @@ -0,0 +1,46 @@ +"""Phase 1 license-compliance gate. + +Reads license-scan.json produced by scripts/mcp/run_license_scan.py (a licensee run +compared against the approved-license allow-list) and converts it into a +standardized GateResult. Fails in block mode when the top-level declared license +is missing or not in the allow-list (emitted as a HIGH finding by the scanner). + +ADR Phase 1: "License compliance - verify the MCP server's top-level declared +license against a list of approved/supported licenses ... Scoped to the +top-level declared license only; transitive dependency license scanning is out +of scope." +""" + +from __future__ import annotations + +from pathlib import Path + +from abevalflow.gates.base import GateResult +from abevalflow.gates.security.base import SecurityGate +from abevalflow.observability.decorators import timed_gate +from abevalflow.schemas import GatePolicy + +# Approved SPDX license identifiers for the top-level declared license. +# Compared case-insensitively (see run_license_scan.py::evaluate_licenses). +DEFAULT_ALLOWED_LICENSES: tuple[str, ...] = ( + "apache-2.0", + "mit", + "bsd-2-clause", + "bsd-3-clause", +) + + +class LicenseGate(SecurityGate): + """Top-level declared-license allow-list gate (licensee-backed).""" + + name = "mcp-license" + scan_filename = "license-scan.json" + + @timed_gate + def evaluate( + self, + reports_dir: Path, + policy: GatePolicy, + ) -> GateResult: + """Evaluate the license allow-list scan.""" + return self.evaluate_scan_json(reports_dir, policy) diff --git a/abevalflow/mcp/phase1/no_user_code.py b/abevalflow/mcp/phase1/no_user_code.py new file mode 100644 index 00000000..20dcef30 --- /dev/null +++ b/abevalflow/mcp/phase1/no_user_code.py @@ -0,0 +1,36 @@ +"""Phase 1 no-user-code-execution gate. + +Reads no-user-code-scan.json produced by scripts/mcp/run_no_user_code_scan.py (a +normalized semgrep run) and converts it into a standardized GateResult. Fails +in block mode when any HIGH/CRITICAL dynamic/arbitrary-execution pattern is +present (e.g. eval/exec, os.system, subprocess shell=True, Function(), +Runtime.exec). + +ADR Phase 1: "Does not execute user-provided code - static analysis of the +artifact for dynamic/arbitrary code execution." +""" + +from __future__ import annotations + +from pathlib import Path + +from abevalflow.gates.base import GateResult +from abevalflow.gates.security.base import SecurityGate +from abevalflow.observability.decorators import timed_gate +from abevalflow.schemas import GatePolicy + + +class NoUserCodeGate(SecurityGate): + """Static dynamic-execution pattern scan gate (semgrep-backed).""" + + name = "mcp-no-user-code" + scan_filename = "no-user-code-scan.json" + + @timed_gate + def evaluate( + self, + reports_dir: Path, + policy: GatePolicy, + ) -> GateResult: + """Evaluate the normalized semgrep dynamic-execution scan.""" + return self.evaluate_scan_json(reports_dir, policy) diff --git a/abevalflow/mcp/phase1/secrets.py b/abevalflow/mcp/phase1/secrets.py new file mode 100644 index 00000000..529944d0 --- /dev/null +++ b/abevalflow/mcp/phase1/secrets.py @@ -0,0 +1,34 @@ +"""Phase 1 secrets-management gate. + +Reads secrets-scan.json produced by scripts/mcp/run_secrets_scan.py (a normalized +gitleaks run) and converts it into a standardized GateResult. Fails in block +mode when any HIGH/CRITICAL hardcoded-credential finding is present. + +ADR Phase 1: "Secrets management - static scan of the artifact for hardcoded +credentials." +""" + +from __future__ import annotations + +from pathlib import Path + +from abevalflow.gates.base import GateResult +from abevalflow.gates.security.base import SecurityGate +from abevalflow.observability.decorators import timed_gate +from abevalflow.schemas import GatePolicy + + +class SecretsGate(SecurityGate): + """Static hardcoded-credential scan gate (gitleaks-backed).""" + + name = "mcp-secrets" + scan_filename = "secrets-scan.json" + + @timed_gate + def evaluate( + self, + reports_dir: Path, + policy: GatePolicy, + ) -> GateResult: + """Evaluate the normalized gitleaks secrets scan.""" + return self.evaluate_scan_json(reports_dir, policy) diff --git a/abevalflow/mcp/phase2/__init__.py b/abevalflow/mcp/phase2/__init__.py new file mode 100644 index 00000000..ef235e1a --- /dev/null +++ b/abevalflow/mcp/phase2/__init__.py @@ -0,0 +1,60 @@ +"""MCP evaluation Phase 2 - deterministic contract/conformance (live, no LLM). + +Phase 2 is black-box tested against the deployed, running MCP server (ADR +Approach 4). It is deterministic and uses NO LLM. Results follow the three-state +model (pass / fail / not_evaluated). + +Two sources feed the same set of gates: +* **Live probes** (scripts/mcp/run_phase2_probe.py) - checks that need a running + server: protocol compliance, schema conformance, tool annotations, HTTP + support, rate limiting, response-size limit, mandatory timeouts, OpenAPI + conformance. +* **Compass-consumed** (scripts/mcp/run_compass_fetch.py) - the Compass tier-1 + automated facts consumed rather than re-probed: tool-name rules and + OAuth-matches-catalog. + +Like Phase 1, this package is kept isolated from the skill pipeline: the gates +are instantiated only here via ``run_phase2`` and are never registered in the +global security-gate registry. +""" + +from __future__ import annotations + +from pathlib import Path + +from abevalflow.gates.base import GateResult +from abevalflow.mcp.phase2.base import Phase2Gate +from abevalflow.schemas import GatePolicy + +# Live probe checks (need a running server). +LIVE_CHECKS: tuple[str, ...] = ( + "http-support", + "protocol-compliance", + "schema-conformance", + "tool-annotations", + "rate-limiting", + "response-size-limit", + "mandatory-timeouts", + "openapi-conformance", +) + +# Checks consumed from existing Compass facts (not re-probed). +COMPASS_CHECKS: tuple[str, ...] = ( + "tool-name-rules", + "oauth-catalog-match", +) + +PHASE2_CHECKS: tuple[str, ...] = LIVE_CHECKS + COMPASS_CHECKS + + +def get_phase2_gates() -> list[Phase2Gate]: + """Instantiate one gate per Phase 2 check, in order.""" + return [Phase2Gate(check) for check in PHASE2_CHECKS] + + +def run_phase2(reports_dir: Path, policy: GatePolicy) -> list[GateResult]: + """Evaluate every Phase 2 gate against the check result files in ``reports_dir``.""" + return [gate.evaluate(reports_dir, policy) for gate in get_phase2_gates()] + + +__all__ = ["COMPASS_CHECKS", "LIVE_CHECKS", "PHASE2_CHECKS", "Phase2Gate", "get_phase2_gates", "run_phase2"] diff --git a/abevalflow/mcp/phase2/base.py b/abevalflow/mcp/phase2/base.py new file mode 100644 index 00000000..827d710e --- /dev/null +++ b/abevalflow/mcp/phase2/base.py @@ -0,0 +1,159 @@ +"""Phase 2 gate: turn a three-state check result file into a GateResult. + +Unlike the Phase 1 security gates (one class per gate, reusing +``SecurityGate.evaluate_scan_json``), Phase 2 has many small, uniform checks that +differ only by name and by a three-state status, so a single data-driven gate is +used instead of a subclass per check. + +Each check result file (written by scripts/mcp/run_phase2_probe.py or run_compass_fetch.py) +has the shape:: + + {"check": ..., "status": "pass"|"fail"|"not_evaluated", "source": ..., + "reason": ..., "findings": [ ... ]} + +Status handling: +* ``pass`` -> passed, score 1.0 +* ``fail`` -> passed only in warn mode; score weighted by findings +* ``not_evaluated`` -> passed, score 1.0 (never penalize; ADR three-state) +* file missing -> not_evaluated (the check did not run) +""" + +from __future__ import annotations + +import json +import logging +from pathlib import Path + +from abevalflow.gates.base import Finding, GateMode, GateResult, GateType, Severity +from abevalflow.observability.decorators import timed_gate +from abevalflow.schemas import GatePolicy + +logger = logging.getLogger(__name__) + +STATUS_PASS = "pass" +STATUS_FAIL = "fail" +STATUS_NOT_EVALUATED = "not_evaluated" + +_SEVERITY_WEIGHT = { + Severity.CRITICAL: 0.0, + Severity.HIGH: 0.25, + Severity.MEDIUM: 0.5, + Severity.LOW: 0.75, + Severity.INFO: 0.9, +} + + +def _weighted_score(findings: list[Finding]) -> float: + if not findings: + return 1.0 + return sum(_SEVERITY_WEIGHT[f.severity] for f in findings) / len(findings) + + +class Phase2Gate: + """Data-driven gate for one Phase 2 check.""" + + def __init__(self, check: str) -> None: + self.check = check + self.result_filename = f"{check}-check.json" + + def _result( + self, + *, + passed: bool, + score: float, + mode: GateMode, + status: str, + reason: str, + source: str, + findings: list[Finding], + ) -> GateResult: + return GateResult( + gate_type=GateType.QUALITY, + gate_name="quality", + policy_key=self.check, + passed=passed, + score=score, + mode=mode, + findings=findings, + details={"check": self.check, "status": status, "source": source, "reason": reason}, + message=f"{self.check}: {status.upper()} - {reason}" if reason else f"{self.check}: {status.upper()}", + ) + + @timed_gate + def evaluate(self, reports_dir: Path, policy: GatePolicy) -> GateResult: + gate_policy = policy.get_gate_policy("quality") + mode = gate_policy.mode + path = reports_dir / self.result_filename + + if not path.is_file(): + return self._result( + passed=True, + score=1.0, + mode=mode, + status=STATUS_NOT_EVALUATED, + reason=f"{self.result_filename} not found (check did not run)", + source="none", + findings=[], + ) + + try: + data = json.loads(path.read_text()) + except (json.JSONDecodeError, OSError) as exc: + # A corrupt result file is a real pipeline defect, so log it loudly and + # score 0.0 - but per the three-state model a "couldn't read" must not + # block outside block mode (mirrors the fail branch below). + logger.error("Failed to read %s: %s", path, exc) + return self._result( + passed=mode != GateMode.BLOCK, + score=0.0, + mode=mode, + status=STATUS_FAIL, + reason=f"could not parse {self.result_filename}: {exc}", + source="none", + findings=[], + ) + + status = data.get("status", STATUS_NOT_EVALUATED) + reason = data.get("reason", "") + source = data.get("source", "probe") + findings = [ + Finding( + severity=_coerce_severity(f.get("severity")), + message=f.get("message", ""), + location=f.get("file_path") or f.get("location"), + rule_id=f.get("rule_id", "unknown"), + details=f, + ) + for f in data.get("findings", []) + ] + + if status == STATUS_PASS: + return self._result( + passed=True, score=1.0, mode=mode, status=status, reason=reason, source=source, findings=findings + ) + if status == STATUS_NOT_EVALUATED: + return self._result( + passed=True, score=1.0, mode=mode, status=status, reason=reason, source=source, findings=findings + ) + # fail + passed = mode != GateMode.BLOCK + # A fail with no findings must not report a perfect score: _weighted_score + # returns 1.0 for an empty list (correct only on the pass path), so a bare + # fail is floored to 0.0 here. + score = _weighted_score(findings) if findings else 0.0 + return self._result( + passed=passed, + score=score, + mode=mode, + status=STATUS_FAIL, + reason=reason, + source=source, + findings=findings, + ) + + +def _coerce_severity(value: object) -> Severity: + try: + return Severity(str(value).lower().strip()) + except ValueError: + return Severity.INFO diff --git a/containers/mcp-eval/Containerfile b/containers/mcp-eval/Containerfile new file mode 100644 index 00000000..152e29e0 --- /dev/null +++ b/containers/mcp-eval/Containerfile @@ -0,0 +1,62 @@ +# MCP server evaluation image for ABEvalFlow Phase 1 (static) + Phase 2 (conformance). +# +# Dedicated image so the two MCP Tekton tasks no longer clone the pipeline repo or +# install tools at runtime. Nothing else pulls this image, so changes here cannot +# affect the shared eval-base pipelines. +# +# Bakes: +# - the MCP eval code and its full import closure (abevalflow.mcp / gates / +# schemas / observability, and scripts.mcp) importable via PYTHONPATH=/opt +# - the scanners/probes it shells out to: gitleaks, semgrep, licensee +# - the minimal Python deps the code imports: pydantic, pyyaml, httpx +# +# Build from repo root (COPY needs repo-root context): +# podman build -t quay.io/ecosystem-appeng/abevalflow-mcp-eval:0.1 \ +# -f containers/mcp-eval/Containerfile . + +# Base tracks :latest (same as Dockerfile.base / agent-eval-harness) so UBI9 CVE +# patches keep flowing. The tools and Python deps that affect runtime behavior are +# pinned below, which is where version drift would actually break the eval code. +FROM registry.access.redhat.com/ubi9/python-312:latest + +# Pinned tool + Python dep versions (avoids the version drift called out in review; +# e.g. an unpinned pydantic v3 would break the schemas the eval code relies on). +ARG GITLEAKS_VERSION=8.30.0 +ARG SEMGREP_VERSION=1.179.0 +ARG LICENSEE_VERSION=9.18.0 +ARG PYDANTIC_VERSION=2.13.5 +ARG PYYAML_VERSION=6.0.3 +ARG HTTPX_VERSION=0.28.1 + +USER 0 + +# MCP eval code — only the verified import closure (abevalflow.mcp pulls in +# gates, schemas, observability; scripts.mcp are the Tekton entry points). +COPY abevalflow/__init__.py /opt/abevalflow/__init__.py +COPY abevalflow/schemas.py /opt/abevalflow/schemas.py +COPY abevalflow/mcp /opt/abevalflow/mcp +COPY abevalflow/gates /opt/abevalflow/gates +COPY abevalflow/observability /opt/abevalflow/observability +COPY scripts/__init__.py /opt/scripts/__init__.py +COPY scripts/mcp /opt/scripts/mcp + +ENV PYTHONPATH=/opt + +# Scanners/probes (shelled out to, not imported) + minimal Python deps (imported). +# licensee's native dep (rugged -> libgit2) needs a build toolchain to compile; +# install it, build the gem, then drop the toolchain to keep the image lean. +RUN curl -LsSf \ + "https://github.com/gitleaks/gitleaks/releases/download/v${GITLEAKS_VERSION}/gitleaks_${GITLEAKS_VERSION}_linux_x64.tar.gz" \ + | tar -xz -C /usr/local/bin gitleaks \ + && pip install --no-cache-dir \ + "semgrep==${SEMGREP_VERSION}" \ + "pydantic==${PYDANTIC_VERSION}" \ + "pyyaml==${PYYAML_VERSION}" \ + "httpx==${HTTPX_VERSION}" \ + && dnf install -y ruby ruby-devel gcc gcc-c++ make cmake pkgconf-pkg-config openssl-devel \ + && gem install licensee -v "${LICENSEE_VERSION}" \ + && dnf remove -y ruby-devel gcc gcc-c++ make cmake pkgconf-pkg-config openssl-devel \ + && dnf clean all + +USER 1001 +WORKDIR /workspace diff --git a/pipeline/tasks/konflux/mcp-phase1-static.yaml b/pipeline/tasks/konflux/mcp-phase1-static.yaml new file mode 100644 index 00000000..84a59118 --- /dev/null +++ b/pipeline/tasks/konflux/mcp-phase1-static.yaml @@ -0,0 +1,104 @@ +apiVersion: tekton.dev/v1 +kind: Task +metadata: + name: mcp-phase1-static + labels: + app.kubernetes.io/name: abevalflow + app.kubernetes.io/component: konflux +spec: + description: >- + MCP evaluation Phase 1 - static / build-time analysis (ADR Approach 4). + Runs three language-agnostic static checks against the MCP server source, + before deployment and without running the server: secrets management + (gitleaks), no user-provided code execution (semgrep), and license + compliance (licensee). Always executes, independent of runtime outcomes; + resolves to pass/fail only. Normalized scan JSON and per-gate GateResults + are written under reports//. + params: + - name: mcp-eval-image + type: string + default: "quay.io/ecosystem-appeng/abevalflow-mcp-eval:0.1" + description: >- + MCP eval image with the scanners (gitleaks, semgrep, licensee) and the + pipeline code (scripts.mcp, abevalflow.mcp) baked in at PYTHONPATH=/opt. + - name: source-subdir + type: string + default: "." + description: >- + Path of the MCP server source within the workspace to scan + (relative to the workspace root). Defaults to the workspace root. + - name: submission-name + type: string + description: Name used for the report directory (reports//). + - name: mode + type: string + default: "block" + description: Gate enforcement mode - block, warn, or disabled. + - name: allowed-licenses + type: string + default: "" + description: >- + Space-separated approved SPDX ids overriding the built-in allow-list + (e.g. "apache-2.0 mit"). Empty uses DEFAULT_ALLOWED_LICENSES. + workspaces: + - name: source + description: Workspace containing the MCP server source to evaluate. + results: + - name: phase1-passed + description: Whether all Phase 1 gates passed (true/false). + - name: report-path + description: Workspace-relative path to the Phase 1 reports directory. + steps: + - name: scan + image: $(params.mcp-eval-image) + # User-controlled params are bound to env vars (Tekton substitutes them as + # literal strings) instead of inlined into the script, so a value containing + # shell metacharacters cannot execute in the task pod. + env: + - name: SOURCE_SUBDIR + value: $(params.source-subdir) + - name: SUBMISSION_NAME + value: $(params.submission-name) + - name: ALLOWED_LICENSES + value: $(params.allowed-licenses) + script: | + #!/usr/bin/env bash + set -euo pipefail + SOURCE_DIR="$(workspaces.source.path)/$SOURCE_SUBDIR" + REPORTS_DIR="$(workspaces.source.path)/reports/$SUBMISSION_NAME" + mkdir -p "$REPORTS_DIR" + + LICENSE_ARGS="" + # shellcheck disable=SC2086 + for spdx in $ALLOWED_LICENSES; do + LICENSE_ARGS="$LICENSE_ARGS --allow $spdx" + done + + echo "=== Phase 1 scanners on $SOURCE_DIR ===" + python -m scripts.mcp.run_secrets_scan "$SOURCE_DIR" --reports-dir "$REPORTS_DIR" + python -m scripts.mcp.run_no_user_code_scan "$SOURCE_DIR" --reports-dir "$REPORTS_DIR" + # shellcheck disable=SC2086 + python -m scripts.mcp.run_license_scan "$SOURCE_DIR" --reports-dir "$REPORTS_DIR" $LICENSE_ARGS + + - name: gate + image: $(params.mcp-eval-image) + env: + - name: SUBMISSION_NAME + value: $(params.submission-name) + - name: MODE + value: $(params.mode) + script: | + #!/usr/bin/env bash + set -euo pipefail + REPORTS_DIR="$(workspaces.source.path)/reports/$SUBMISSION_NAME" + + # Shared gate wrapper (baked in the image): runs the gate module, emits the + # Tekton results from the summary, and propagates the gate's exit code. + bash /opt/scripts/mcp/run_gate.sh \ + scripts.mcp.run_phase1_gates \ + "$REPORTS_DIR" \ + "$MODE" \ + "$REPORTS_DIR/phase1-summary.json" \ + "$(results.phase1-passed.path)" \ + "reports/$SUBMISSION_NAME" \ + "$(results.report-path.path)" diff --git a/pipeline/tasks/konflux/mcp-phase2-conformance.yaml b/pipeline/tasks/konflux/mcp-phase2-conformance.yaml new file mode 100644 index 00000000..01f2b479 --- /dev/null +++ b/pipeline/tasks/konflux/mcp-phase2-conformance.yaml @@ -0,0 +1,113 @@ +apiVersion: tekton.dev/v1 +kind: Task +metadata: + name: mcp-phase2-conformance + labels: + app.kubernetes.io/name: abevalflow + app.kubernetes.io/component: konflux +spec: + description: >- + MCP evaluation Phase 2 - deterministic contract/conformance (ADR Approach 4). + Black-box, no LLM. Probes the deployed, running MCP server over Streamable + HTTP for protocol/schema/HTTP conformance and operational limits, and + consumes the checks Compass already collects as real facts (tool-name rules, + OAuth match). Uses the three-state model + (pass / fail / not_evaluated): an unreachable server or a check that cannot + produce a meaningful signal is reported not_evaluated, never failed. Assumes + the server was already deployed and became ready in a prior pipeline step + (reached at :8080 by convention). + params: + - name: mcp-eval-image + type: string + default: "quay.io/ecosystem-appeng/abevalflow-mcp-eval:0.1" + description: >- + MCP eval image with the probe dependencies (httpx, pydantic, pyyaml) and + the pipeline code (scripts.mcp, abevalflow.mcp) baked in at PYTHONPATH=/opt. + - name: server-url + type: string + description: MCP endpoint URL, e.g. http://:8080/mcp + - name: submission-name + type: string + description: Name used for the report directory (reports//). + - name: mode + type: string + default: "block" + description: Gate enforcement mode - block, warn, or disabled. + - name: timeout + type: string + default: "30" + description: Per-request HTTP timeout (seconds) for the probe. + - name: compass-facts-file + type: string + default: "" + description: >- + Workspace-relative path to a Compass facts JSON for the consumed checks. + Empty skips the Compass-consume step; those checks then report + not_evaluated. (Facts file only; live SoundCheck retrieval is not wired yet.) + workspaces: + - name: source + description: Workspace for the reports directory. + results: + - name: phase2-passed + description: Whether Phase 2 passed (no check FAILED in block mode). + - name: report-path + description: Workspace-relative path to the Phase 2 reports directory. + steps: + - name: probe + image: $(params.mcp-eval-image) + # User-controlled params are bound to env vars (Tekton substitutes them as + # literal strings) instead of inlined into the script, so a value containing + # shell metacharacters cannot execute in the task pod. + env: + - name: SERVER_URL + value: $(params.server-url) + - name: SUBMISSION_NAME + value: $(params.submission-name) + - name: TIMEOUT + value: $(params.timeout) + - name: COMPASS_FACTS_FILE + value: $(params.compass-facts-file) + script: | + #!/usr/bin/env bash + set -euo pipefail + REPORTS_DIR="$(workspaces.source.path)/reports/$SUBMISSION_NAME" + mkdir -p "$REPORTS_DIR" + + echo "=== Phase 2 live probe against $SERVER_URL ===" + python -m scripts.mcp.run_phase2_probe \ + --server-url "$SERVER_URL" \ + --reports-dir "$REPORTS_DIR" \ + --timeout "$TIMEOUT" + + FACTS="$COMPASS_FACTS_FILE" + if [ -n "$FACTS" ]; then + echo "=== Phase 2 Compass consume from $FACTS ===" + python -m scripts.mcp.run_compass_fetch \ + --facts-file "$(workspaces.source.path)/$FACTS" \ + --reports-dir "$REPORTS_DIR" + else + echo "No compass-facts-file given; Compass-consumed checks will report not_evaluated." + fi + + - name: gate + image: $(params.mcp-eval-image) + env: + - name: SUBMISSION_NAME + value: $(params.submission-name) + - name: MODE + value: $(params.mode) + script: | + #!/usr/bin/env bash + set -euo pipefail + REPORTS_DIR="$(workspaces.source.path)/reports/$SUBMISSION_NAME" + + # Shared gate wrapper (baked in the image): runs the gate module, emits the + # Tekton results from the summary, and propagates the gate's exit code. + bash /opt/scripts/mcp/run_gate.sh \ + scripts.mcp.run_phase2_gates \ + "$REPORTS_DIR" \ + "$MODE" \ + "$REPORTS_DIR/phase2-summary.json" \ + "$(results.phase2-passed.path)" \ + "reports/$SUBMISSION_NAME" \ + "$(results.report-path.path)" diff --git a/scripts/mcp/README.md b/scripts/mcp/README.md new file mode 100644 index 00000000..042a86ed --- /dev/null +++ b/scripts/mcp/README.md @@ -0,0 +1,46 @@ +# `scripts/mcp/` — MCP evaluation execution layer + +This package is the **execution / I/O layer** for the MCP-server evaluation +pipeline. These modules shell out to external tools (gitleaks, semgrep, +licensee), probe the live MCP server over HTTP, and read/write report JSON. The +pure pass/fail evaluation logic they feed lives in `abevalflow.mcp`. + +## Naming convention + +Role is encoded in the filename prefix: + +| Prefix | Role | Example | +|--------|------|---------| +| `run_` | Entry point — invoked by a Tekton task as `python -m scripts.mcp.` (or via `run_gate.sh`). | `run_secrets_scan.py` | +| `_` | Helper — imported by the entry points, never run directly. | `_common.py` | + +## Entry points + +| Module | Phase | Invoked by | Emits | +|--------|-------|------------|-------| +| `run_secrets_scan.py` | 1 | phase1 task | `secrets-scan.json` (gitleaks) | +| `run_no_user_code_scan.py` | 1 | phase1 task | `no-user-code-scan.json` (semgrep, bundled offline ruleset) | +| `run_license_scan.py` | 1 | phase1 task | `license-scan.json` (licensee, top-level license only) | +| `run_phase1_gates.py` | 1 | phase1 task (via `run_gate.sh`) | `phase1-summary.json` + gate exit code | +| `run_phase2_probe.py` | 2 | phase2 task | per-check result files (live conformance probe) | +| `run_compass_fetch.py` | 2 | phase2 task | per-check result files (Compass tier-1 facts) | +| `run_phase2_gates.py` | 2 | phase2 task (via `run_gate.sh`) | `phase2-summary.json` + gate exit code | + +`run_gate.sh` is the shared gate wrapper baked into the eval image: both phase +tasks call it to run the phase's gate module, emit the Tekton results +(`passed` flag + report path) from the summary JSON, and propagate the gate's +exit code. + +## Helpers + +| Module | Used by | +|--------|---------| +| `_common.py` | Phase 1 scanners — findings, tool runner, report writer | +| `_phase2.py` | Phase 2 probe + Compass fetch — check result shapes | +| `_mcp_client.py` | Phase 2 probe — minimal MCP client over Streamable HTTP | + +## Layout + +The package is intentionally flat, matching the repo-wide `scripts/` +convention. The `run_`/`_` prefixes keep entry points and helpers legible +without subpackages. diff --git a/scripts/mcp/__init__.py b/scripts/mcp/__init__.py new file mode 100644 index 00000000..740675b8 --- /dev/null +++ b/scripts/mcp/__init__.py @@ -0,0 +1,20 @@ +"""Execution / I/O layer for the MCP evaluation pipeline. + +These modules shell out to external tools, probe the live server, and read/write +report JSON. The pure pass/fail evaluation logic they feed lives in +``abevalflow.mcp``. + +Tekton entry points are ``run_``-prefixed and run as +``python -m scripts.mcp.``: + +- ``run_secrets_scan`` - Phase 1 secrets scan (gitleaks) +- ``run_no_user_code_scan`` - Phase 1 no-user-code scan (semgrep) +- ``run_license_scan`` - Phase 1 license scan (licensee) +- ``run_phase1_gates`` - Phase 1 gate evaluation + summary +- ``run_phase2_probe`` - Phase 2 live conformance probe +- ``run_compass_fetch`` - Phase 2 Compass-fact consume +- ``run_phase2_gates`` - Phase 2 gate evaluation + summary + +Internal helpers (imported by the entry points, never run directly) are +underscore-prefixed: ``_common``, ``_phase2``, ``_mcp_client``. +""" diff --git a/scripts/mcp/_common.py b/scripts/mcp/_common.py new file mode 100644 index 00000000..4adc3bb4 --- /dev/null +++ b/scripts/mcp/_common.py @@ -0,0 +1,101 @@ +"""Shared helpers for MCP Phase 1 scanner scripts. + +All Phase 1 scanners emit the same JSON shape so the Phase 1 gates (which reuse +``SecurityGate.evaluate_scan_json``) can read them uniformly:: + + { + "scanner": "", + "target": "", + "findings": [ + {"severity": "high", "message": "...", "rule_id": "...", "file_path": "..."}, + ... + ] + } + +``severity`` must be one of the values understood by ``abevalflow.gates.base``: +critical / high / medium / low / info. +""" + +from __future__ import annotations + +import json +import logging +import subprocess +from pathlib import Path +from typing import Any + +logger = logging.getLogger(__name__) + +VALID_SEVERITIES = ("critical", "high", "medium", "low", "info") + + +def normalize_severity(value: str | None, default: str = "info") -> str: + """Coerce an arbitrary severity string into a known severity level.""" + if not value: + return default + sev = value.strip().lower() + return sev if sev in VALID_SEVERITIES else default + + +def make_finding( + *, + severity: str, + message: str, + rule_id: str, + file_path: str | None = None, + **extra: Any, +) -> dict[str, Any]: + """Build a single normalized finding dict.""" + finding: dict[str, Any] = { + "severity": normalize_severity(severity), + "message": message, + "rule_id": rule_id, + "file_path": file_path, + } + finding.update(extra) + return finding + + +def write_scan( + scan_path: Path, + scanner: str, + target: Path, + findings: list[dict[str, Any]], + *, + extra: dict[str, Any] | None = None, +) -> None: + """Write a normalized scan report to ``scan_path``. + + ``extra`` merges additional top-level metadata (e.g. a ``coverage`` block) + into the payload. The Phase 1 gates read only ``findings``, so extra keys are + informational and never change pass/fail. The canonical keys always win, so a + caller cannot accidentally clobber ``findings`` via ``extra``. + """ + scan_path.parent.mkdir(parents=True, exist_ok=True) + payload: dict[str, Any] = dict(extra or {}) + payload.update( + { + "scanner": scanner, + "target": str(target), + "findings": findings, + } + ) + scan_path.write_text(json.dumps(payload, indent=2)) + logger.info("Wrote %d findings to %s", len(findings), scan_path) + + +def run_tool(cmd: list[str], *, cwd: Path | None = None) -> subprocess.CompletedProcess[str]: + """Run a scanner CLI, capturing stdout/stderr as text. + + The return code is NOT checked here - most scanners exit non-zero when they + find issues, which is expected. Callers inspect ``returncode`` explicitly to + distinguish "findings present" from "tool failed to run". + """ + logger.info("Running: %s", " ".join(cmd)) + return subprocess.run( # noqa: S603 - cmd is built from trusted, fixed tool args + cmd, + cwd=str(cwd) if cwd else None, + capture_output=True, + text=True, + check=False, + ) diff --git a/scripts/mcp/_mcp_client.py b/scripts/mcp/_mcp_client.py new file mode 100644 index 00000000..2e2aadc6 --- /dev/null +++ b/scripts/mcp/_mcp_client.py @@ -0,0 +1,250 @@ +"""Minimal MCP client over Streamable HTTP for Phase 2 conformance probing. + +Hand-rolled JSON-RPC 2.0 over ``httpx`` (deliberately NOT the ``mcp`` SDK) so the +Phase 2 probe can: + +* send well-formed calls (initialize, tools/list, tools/call), and +* send deliberately malformed / spec-violating payloads and inspect the raw + HTTP response (status code, headers, content-type, body). + +The typed SDK client cannot emit malformed frames and raises on non-conformant +server responses instead of letting us assert on them, which is exactly what the +conformance checks need to observe. ``httpx`` is already a project dependency. + +Streamable HTTP handshake (MCP spec, transport rev 2025-06-18): + 1. POST ``initialize`` with ``Accept: application/json, text/event-stream``. + 2. Capture ``Mcp-Session-Id`` from the response headers; echo it on every + later request. + 3. POST the ``notifications/initialized`` notification (no id) -> 202, no body. + 4. Include ``MCP-Protocol-Version`` on post-init requests. + 5. A request may be answered with ``application/json`` OR ``text/event-stream`` + (SSE); both are handled here. +""" + +from __future__ import annotations + +import json +import logging +from dataclasses import dataclass +from typing import Any + +import httpx + +logger = logging.getLogger(__name__) + +DEFAULT_PROTOCOL_VERSION = "2025-06-18" +ACCEPT_HEADER = "application/json, text/event-stream" +CLIENT_INFO = {"name": "mcp-phase2-probe", "version": "0.1"} + + +@dataclass +class RpcResponse: + """Raw result of a single POST to the MCP endpoint. + + ``message`` is the parsed JSON-RPC object (from a plain JSON body, or the SSE + ``data:`` frame whose id matches the request), or ``None`` when the body is + empty (e.g. a 202 to a notification) or unparseable. + """ + + status_code: int + headers: dict[str, str] + content_type: str + message: dict[str, Any] | list[Any] | None + raw_text: str + + @property + def is_sse(self) -> bool: + return "text/event-stream" in self.content_type + + +def _parse_sse_messages(text: str) -> list[dict[str, Any] | list[Any]]: + """Extract every JSON-RPC message from an SSE stream body, in order.""" + messages: list[dict[str, Any] | list[Any]] = [] + for line in text.splitlines(): + if line.startswith("data:"): + payload = line[len("data:") :].strip() + if not payload: + continue + try: + messages.append(json.loads(payload)) + except json.JSONDecodeError: + continue + return messages + + +def _select_response(messages: list[dict[str, Any] | list[Any]], request_id: Any) -> dict[str, Any] | list[Any] | None: + """Pick the JSON-RPC response matching ``request_id``. + + A server may interleave notifications (no/other id) ahead of the response on + the stream, so selecting the first frame can mis-read a valid response as a + protocol violation. Matching is tried exactly first, then by string value (a + server that echoes the id as "2" rather than 2 still matches). When + ``request_id`` is None (e.g. a notification or a raw malformed frame) or no + frame matches, fall back to the first message. + """ + if request_id is not None: + for msg in messages: + if isinstance(msg, dict) and msg.get("id") == request_id: + return msg + for msg in messages: + if isinstance(msg, dict) and "id" in msg and str(msg.get("id")) == str(request_id): + return msg + return messages[0] if messages else None + + +class MCPConnectionError(RuntimeError): + """Raised when the server cannot be reached at all (used to mark checks not-evaluated).""" + + +class MCPClient: + """Small stateful client for one MCP server endpoint.""" + + def __init__( + self, + endpoint_url: str, + *, + timeout: float = 30.0, + protocol_version: str = DEFAULT_PROTOCOL_VERSION, + ) -> None: + self.endpoint_url = endpoint_url + self.protocol_version = protocol_version + self.negotiated_version: str | None = None + self.session_id: str | None = None + self._client = httpx.Client(timeout=timeout) + + # -- context management ------------------------------------------------- + def __enter__(self) -> MCPClient: + return self + + def __exit__(self, *exc: object) -> None: + self.close() + + def close(self) -> None: + self._client.close() + + # -- low-level POST ----------------------------------------------------- + def post( + self, + body: Any, + *, + include_session: bool = True, + include_version: bool = True, + extra_headers: dict[str, str] | None = None, + raw: bool = False, + ) -> RpcResponse: + """POST a payload and return the raw parsed response. + + ``body`` is JSON-encoded unless ``raw=True``, in which case it is sent + verbatim (used to send malformed frames for negative conformance tests). + Raises :class:`MCPConnectionError` only when the server is unreachable. + """ + headers = {"Accept": ACCEPT_HEADER, "Content-Type": "application/json"} + if include_version: + headers["MCP-Protocol-Version"] = self.negotiated_version or self.protocol_version + if include_session and self.session_id: + headers["Mcp-Session-Id"] = self.session_id + if extra_headers: + headers.update(extra_headers) + + content = body if raw else json.dumps(body) + request_id = body.get("id") if isinstance(body, dict) else None + try: + resp = self._client.post(self.endpoint_url, content=content, headers=headers) + except httpx.HTTPError as exc: + raise MCPConnectionError(str(exc)) from exc + + text = resp.text + content_type = resp.headers.get("content-type", "") + if "text/event-stream" in content_type: + message = _select_response(_parse_sse_messages(text), request_id) + elif text.strip(): + try: + message = json.loads(text) + except json.JSONDecodeError: + message = None + else: + message = None + + return RpcResponse( + status_code=resp.status_code, + headers=dict(resp.headers), + content_type=content_type, + message=message, + raw_text=text, + ) + + # -- MCP protocol helpers ---------------------------------------------- + def initialize(self, *, request_id: Any = 1, params: dict[str, Any] | None = None) -> RpcResponse: + """Send the ``initialize`` request and capture session id + negotiated version.""" + body = { + "jsonrpc": "2.0", + "id": request_id, + "method": "initialize", + "params": params + or { + "protocolVersion": self.protocol_version, + "capabilities": {}, + "clientInfo": CLIENT_INFO, + }, + } + # No Mcp-Session-Id yet, and MCP-Protocol-Version is a post-init header + # (the version is negotiated via the params body here). + resp = self.post(body, include_session=False, include_version=False) + # Session id lives in the response headers (case-insensitive). + for key, value in resp.headers.items(): + if key.lower() == "mcp-session-id": + self.session_id = value + break + if isinstance(resp.message, dict): + result = resp.message.get("result", {}) + if isinstance(result, dict) and result.get("protocolVersion"): + self.negotiated_version = result["protocolVersion"] + return resp + + def send_initialized(self) -> RpcResponse: + """Send the ``notifications/initialized`` notification (expects 202, empty body).""" + return self.post({"jsonrpc": "2.0", "method": "notifications/initialized"}) + + def list_tools(self, *, request_id: Any = 2, cursor: str | None = None) -> RpcResponse: + params: dict[str, Any] = {"cursor": cursor} if cursor else {} + return self.post({"jsonrpc": "2.0", "id": request_id, "method": "tools/list", "params": params}) + + def list_all_tools(self, *, max_pages: int = 50) -> list[dict[str, Any]]: + """Return every tool across all ``tools/list`` pages, following ``nextCursor``. + + Paginating matters for conformance: a server can place a malformed tool on + a later page, so validating only page one would miss it. ``max_pages`` + bounds a server that returns an endless or self-repeating cursor. + """ + tools: list[dict[str, Any]] = [] + cursor: str | None = None + seen_cursors: set[str] = set() + for page in range(max_pages): + resp = self.list_tools(request_id=2 + page, cursor=cursor) + result = resp.message.get("result") if isinstance(resp.message, dict) else None + if not isinstance(result, dict): + break + page_tools = result.get("tools", []) + if isinstance(page_tools, list): + tools.extend(t for t in page_tools if isinstance(t, dict)) + cursor = result.get("nextCursor") + if not cursor or not isinstance(cursor, str) or cursor in seen_cursors: + break + seen_cursors.add(cursor) + return tools + + def call_tool(self, name: str, arguments: dict[str, Any] | None = None, *, request_id: Any = 3) -> RpcResponse: + return self.post( + { + "jsonrpc": "2.0", + "id": request_id, + "method": "tools/call", + "params": {"name": name, "arguments": arguments or {}}, + } + ) + + def handshake(self) -> RpcResponse: + """Convenience: initialize then send the initialized notification.""" + resp = self.initialize() + self.send_initialized() + return resp diff --git a/scripts/mcp/_phase2.py b/scripts/mcp/_phase2.py new file mode 100644 index 00000000..6e0441ec --- /dev/null +++ b/scripts/mcp/_phase2.py @@ -0,0 +1,72 @@ +"""Shared helpers for MCP Phase 2 (deterministic contract/conformance). + +Phase 2 uses a THREE-STATE model (pass / fail / not_evaluated) - unlike Phase 1, +which is always pass/fail. ``not_evaluated`` is used when a check cannot produce a +meaningful signal (server unreachable, no OpenAPI contract, a capability the +server does not expose), so a server is never penalized for something it was not +actually subjected to (ADR: three-state reporting). + +Every Phase 2 check - whether a live probe or a value consumed from Compass - +writes the same JSON shape so the Phase 2 gates can read them uniformly:: + + { + "check": "protocol-compliance", + "status": "pass" | "fail" | "not_evaluated", + "source": "probe" | "compass", + "reason": "human-readable summary", + "findings": [ {"severity": ..., "message": ..., "rule_id": ...}, ... ] + } +""" + +from __future__ import annotations + +import json +import logging +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any + +logger = logging.getLogger(__name__) + +STATUS_PASS = "pass" +STATUS_FAIL = "fail" +STATUS_NOT_EVALUATED = "not_evaluated" +VALID_STATUSES = (STATUS_PASS, STATUS_FAIL, STATUS_NOT_EVALUATED) + + +@dataclass +class CheckOutcome: + """Result of a single Phase 2 check.""" + + check: str + status: str + reason: str = "" + source: str = "probe" + findings: list[dict[str, Any]] = field(default_factory=list) + + def __post_init__(self) -> None: + if self.status not in VALID_STATUSES: + raise ValueError(f"invalid status {self.status!r}; expected one of {VALID_STATUSES}") + + def to_dict(self) -> dict[str, Any]: + return { + "check": self.check, + "status": self.status, + "source": self.source, + "reason": self.reason, + "findings": self.findings, + } + + +def result_filename(check: str) -> str: + """Filename a check's result is written to (and read back by its gate).""" + return f"{check}-check.json" + + +def write_check_result(reports_dir: Path, outcome: CheckOutcome) -> Path: + """Write one check outcome to ``/-check.json``.""" + reports_dir.mkdir(parents=True, exist_ok=True) + path = reports_dir / result_filename(outcome.check) + path.write_text(json.dumps(outcome.to_dict(), indent=2)) + logger.info("Wrote %s check result (%s) to %s", outcome.check, outcome.status, path) + return path diff --git a/scripts/mcp/rules/no_user_code.yml b/scripts/mcp/rules/no_user_code.yml new file mode 100644 index 00000000..5e8e4376 --- /dev/null +++ b/scripts/mcp/rules/no_user_code.yml @@ -0,0 +1,81 @@ +# Semgrep ruleset for MCP Phase 1 "does not execute user-provided code". +# +# Flags dynamic / arbitrary code-execution sinks across the languages MCP +# servers are commonly written in. Deliberately narrow (high-signal sinks only) +# to keep false positives low and to run fully offline (no --config p/registry +# network fetch). Extend per language as needed. +# +# Severity ERROR -> mapped to HIGH by scripts/mcp/no_user_code_scan.py. +rules: + # ---- Python ------------------------------------------------------------- + - id: python-dynamic-exec + languages: [python] + severity: ERROR + message: Dynamic code execution via eval/exec/compile. + patterns: + - pattern-either: + - pattern: eval(...) + - pattern: exec(...) + - pattern: compile(...) + + - id: python-os-system + languages: [python] + severity: ERROR + message: Shell command execution via os.system. + pattern: os.system(...) + + - id: python-subprocess-shell + languages: [python] + severity: ERROR + message: subprocess call with shell=True executes a shell command string. + patterns: + - pattern-either: + - pattern: subprocess.$FN(..., shell=True, ...) + - pattern: subprocess.Popen(..., shell=True, ...) + + # ---- JavaScript / TypeScript ------------------------------------------- + - id: js-dynamic-eval + languages: [javascript, typescript] + severity: ERROR + message: Dynamic code execution via eval or the Function constructor. + patterns: + - pattern-either: + - pattern: eval(...) + - pattern: new Function(...) + - pattern: Function(...) + + - id: js-child-process-exec + languages: [javascript, typescript] + severity: ERROR + message: Shell command execution via child_process.exec. + patterns: + - pattern-either: + - pattern: child_process.exec(...) + - pattern: $CP.exec(...) + - pattern: exec(...) + + # ---- Go ---------------------------------------------------------------- + - id: go-exec-command + languages: [go] + severity: ERROR + message: External command execution via os/exec. + patterns: + - pattern-either: + - pattern: exec.Command(...) + - pattern: exec.CommandContext(...) + + # ---- Java -------------------------------------------------------------- + - id: java-runtime-exec + languages: [java] + severity: ERROR + message: External command execution via Runtime.exec / ProcessBuilder. + patterns: + - pattern-either: + - pattern: (Runtime $R).exec(...) + - pattern: new ProcessBuilder(...) + + - id: java-script-engine + languages: [java] + severity: ERROR + message: Dynamic script evaluation via javax.script ScriptEngine. + pattern: (ScriptEngine $E).eval(...) diff --git a/scripts/mcp/run_compass_fetch.py b/scripts/mcp/run_compass_fetch.py new file mode 100644 index 00000000..9ea00de8 --- /dev/null +++ b/scripts/mcp/run_compass_fetch.py @@ -0,0 +1,178 @@ +#!/usr/bin/env python3 +"""MCP Phase 2 Compass-consumed checks. + +A few Phase 2 checks are already collected by Compass as real automated facts, so +the pipeline consumes them rather than re-probing (avoids duplicating what +Compass does well). This script maps those Compass SoundCheck facts into the same +three-state per-check result files the Phase 2 gates read. + +Consumed checks and their source facts: + tool-name-rules <- mcp:default/tools $.allToolNamesValid + oauth-catalog-match<- mcp:default/security $.oauth.enforced + $.entityAuth.authServerMatch + $.scopeMatch.allScopesMatch + +Only these two are consumed: the Compass tier-1 *automated* facts worth surfacing +in our aggregate. Metadata Compass owns directly (schema-present, documentation, +data-source approval) is not re-gated here, per the ADR ("consumed as existing +Compass Check Results"). + +Facts are supplied from a local JSON file. Live SoundCheck retrieval is not wired +yet: the read endpoint and access permissions are still being settled with the +Compass team, so ``--facts-file`` is the only path until then. + +Usage: + python -m scripts.mcp.run_compass_fetch --facts-file facts.json --reports-dir +""" + +from __future__ import annotations + +import argparse +import json +import logging +import sys +from pathlib import Path +from typing import Any + +from scripts.mcp._common import make_finding +from scripts.mcp._phase2 import ( + STATUS_FAIL, + STATUS_NOT_EVALUATED, + STATUS_PASS, + CheckOutcome, + write_check_result, +) + +logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") +logger = logging.getLogger(__name__) + +FACT_TOOLS = "mcp:default/tools" +FACT_SECURITY = "mcp:default/security" + + +def _get(facts: dict[str, Any], fact_ref: str, *path: str) -> Any: + """Walk ``facts[fact_ref]`` down ``path``; return None if any hop is missing.""" + node: Any = facts.get(fact_ref) + for key in path: + if not isinstance(node, dict): + return None + node = node.get(key) + return node + + +def _bool_outcome(check: str, value: Any, ok_reason: str, bad_reason: str) -> CheckOutcome: + """pass/fail from a boolean fact; not_evaluated when the fact is absent.""" + if value is None: + return CheckOutcome(check, STATUS_NOT_EVALUATED, "Source Compass fact not present.", source="compass") + if value is True: + return CheckOutcome(check, STATUS_PASS, ok_reason, source="compass") + finding = make_finding(severity="high", message=bad_reason, rule_id=check) + return CheckOutcome(check, STATUS_FAIL, bad_reason, source="compass", findings=[finding]) + + +def map_facts(facts: dict[str, Any]) -> list[CheckOutcome]: + """Map Compass facts to the consumed Phase 2 check outcomes.""" + outcomes: list[CheckOutcome] = [] + + outcomes.append( + _bool_outcome( + "tool-name-rules", + _get(facts, FACT_TOOLS, "allToolNamesValid"), + "Compass reports all tool names valid.", + "Compass reports one or more tool names invalid.", + ) + ) + + # OAuth matches catalog = enforced AND auth-server matches AND scopes match. + # Distinguish an explicit False (a real violation) from an absent fact (None): + # a missing subfield must NOT be read as a violation, or a partially-reported + # entity would falsely FAIL and block the phase. Only when every subfield is + # present and True can we assert the conjunction; any absent subfield with no + # explicit False leaves us unable to evaluate. + oauth_fields = [ + ("oauth-enforced", _get(facts, FACT_SECURITY, "oauth", "enforced"), "OAuth not enforced per Compass"), + ( + "oauth-server-match", + _get(facts, FACT_SECURITY, "entityAuth", "authServerMatch"), + "OAuth auth server does not match catalog per Compass", + ), + ( + "oauth-scope-match", + _get(facts, FACT_SECURITY, "scopeMatch", "allScopesMatch"), + "OAuth scopes do not match catalog per Compass", + ), + ] + violations = [ + make_finding(severity="high", message=msg, rule_id=rule) for rule, value, msg in oauth_fields if value is False + ] + missing = [rule for rule, value, _ in oauth_fields if value is None] + if violations: + outcomes.append( + CheckOutcome( + "oauth-catalog-match", + STATUS_FAIL, + "OAuth configuration does not match the catalog.", + source="compass", + findings=violations, + ) + ) + elif not missing: + outcomes.append( + CheckOutcome( + "oauth-catalog-match", STATUS_PASS, "OAuth enforced and matches the catalog.", source="compass" + ) + ) + elif len(missing) == len(oauth_fields): + outcomes.append( + CheckOutcome( + "oauth-catalog-match", STATUS_NOT_EVALUATED, "No Compass OAuth facts present.", source="compass" + ) + ) + else: + outcomes.append( + CheckOutcome( + "oauth-catalog-match", + STATUS_NOT_EVALUATED, + f"Incomplete Compass OAuth facts (missing: {', '.join(missing)}); cannot assert OAuth-matches-catalog.", + source="compass", + ) + ) + + return outcomes + + +def main() -> int: + parser = argparse.ArgumentParser(description="MCP Phase 2 Compass-consumed checks") + parser.add_argument( + "--reports-dir", type=Path, required=True, help="Directory to write -check.json files into" + ) + parser.add_argument("--facts-file", type=Path, required=True, help="Local JSON file of Compass facts") + args = parser.parse_args() + + facts: dict[str, Any] = {} + if not args.facts_file.is_file(): + logger.warning("Facts file not found: %s; consumed checks reported not_evaluated.", args.facts_file) + else: + try: + loaded = json.loads(args.facts_file.read_text()) + except (json.JSONDecodeError, OSError) as exc: + logger.warning( + "Could not read facts file %s (%s); consumed checks reported not_evaluated.", args.facts_file, exc + ) + else: + if isinstance(loaded, dict): + facts = loaded + else: + logger.warning( + "Facts file %s is not a JSON object; consumed checks reported not_evaluated.", args.facts_file + ) + + outcomes = map_facts(facts) + for outcome in outcomes: + write_check_result(args.reports_dir, outcome) + logger.info("Compass consume complete: %d checks", len(outcomes)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/mcp/run_gate.sh b/scripts/mcp/run_gate.sh new file mode 100644 index 00000000..76d545b6 --- /dev/null +++ b/scripts/mcp/run_gate.sh @@ -0,0 +1,37 @@ +#!/usr/bin/env bash +# Shared Phase 1 / Phase 2 gate wrapper, baked into the MCP eval image. +# +# Both phase gate steps do the same thing: run the phase's gate module, emit the +# Tekton results (passed flag + report path) from the summary JSON, then exit +# with the gate's own code so a failing gate fails the step. Keeping it here means +# the two tasks stay consistent when this behavior changes. +# +# Usage: +# run_gate.sh \ +# +set -uo pipefail + +GATE_MODULE="$1" # e.g. scripts.mcp.run_phase1_gates +REPORTS_DIR="$2" # absolute reports dir for this submission/phase +MODE="$3" # block | warn | disabled +SUMMARY="$4" # phaseN-summary.json written by the gate module +PASSED_RESULT_PATH="$5" # Tekton result path for the passed flag +REPORT_REL="$6" # workspace-relative report path (result value) +REPORT_PATH_RESULT="$7" # Tekton result path for the report path + +# Never abort before writing results: capture the gate's exit code, emit the +# Tekton results, then propagate the code so the step fails if a gate failed. +python -m "$GATE_MODULE" --reports-dir "$REPORTS_DIR" --mode "$MODE" +GATE_EXIT=$? + +# Emit the passed flag from the summary. A missing summary or one without a +# 'passed' key is itself a failure: fail loudly rather than writing an empty +# result that a downstream `when` would read as neither true nor false. +if ! python3 -c "import json,sys;print(str(json.load(open(sys.argv[1]))['passed']).lower(),end='')" \ + "$SUMMARY" > "$PASSED_RESULT_PATH"; then + echo "run_gate: gate summary $SUMMARY is missing or has no 'passed' key" >&2 + exit 1 +fi +printf '%s' "$REPORT_REL" > "$REPORT_PATH_RESULT" + +exit "$GATE_EXIT" diff --git a/scripts/mcp/run_license_scan.py b/scripts/mcp/run_license_scan.py new file mode 100644 index 00000000..9e92b291 --- /dev/null +++ b/scripts/mcp/run_license_scan.py @@ -0,0 +1,127 @@ +#!/usr/bin/env python3 +"""MCP Phase 1 license-compliance scan. + +Runs licensee to detect the repository's TOP-LEVEL declared license (from the +LICENSE file and/or package manifests) and checks it against an approved-license +allow-list. Emits license-scan.json for LicenseGate. Transitive/dependency +license scanning is out of scope (ADR Phase 1). + +Usage: + python -m scripts.mcp.run_license_scan --reports-dir \ + [--allow apache-2.0 --allow mit ...] +""" + +from __future__ import annotations + +import argparse +import json +import logging +import sys +from pathlib import Path + +from abevalflow.mcp.phase1.license import DEFAULT_ALLOWED_LICENSES +from scripts.mcp._common import make_finding, run_tool, write_scan + +logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") +logger = logging.getLogger(__name__) + +SCAN_FILENAME = "license-scan.json" +SCANNER = "licensee" + + +def evaluate_licenses( + detected: list[dict], + allowed: set[str], + matched_files: list[dict], +) -> list[dict]: + """Produce findings for a missing or disallowed top-level license. + + A repo passes when at least one detected license is in the allow-list. + licensee emits canonical SPDX ids (e.g. "Apache-2.0"); compare + case-insensitively by lowercasing both sides. + """ + location = matched_files[0].get("filename") if matched_files else None + + spdx_ids = [ + str(lic.get("spdx_id", "")).lower() + for lic in detected + if lic.get("spdx_id") and str(lic.get("spdx_id")).lower() != "noassertion" + ] + + if not spdx_ids: + return [ + make_finding( + severity="high", + message="No top-level declared license could be identified.", + rule_id="license-missing", + file_path=location, + ) + ] + + if any(spdx in allowed for spdx in spdx_ids): + return [] + + return [ + make_finding( + severity="high", + message=(f"Declared license(s) {', '.join(spdx_ids)} not in allow-list ({', '.join(sorted(allowed))})."), + rule_id="license-not-allowed", + file_path=location, + detected=spdx_ids, + ) + ] + + +def main() -> int: + parser = argparse.ArgumentParser(description="MCP Phase 1 license scan (licensee)") + parser.add_argument("target_dir", type=Path, help="MCP server repo to scan") + parser.add_argument( + "--reports-dir", + type=Path, + required=True, + help="Directory to write license-scan.json into", + ) + parser.add_argument( + "--allow", + action="append", + default=None, + metavar="SPDX_ID", + help="Approved SPDX id (repeatable). Defaults to DEFAULT_ALLOWED_LICENSES.", + ) + args = parser.parse_args() + + if not args.target_dir.is_dir(): + logger.error("Not a directory: %s", args.target_dir) + return 1 + + allowed = {a.lower() for a in (args.allow or DEFAULT_ALLOWED_LICENSES)} + scan_path = args.reports_dir / SCAN_FILENAME + + result = run_tool(["licensee", "detect", "--json", str(args.target_dir)]) + + # licensee exits non-zero when it finds no license - that is a valid result we + # must still evaluate (-> a "license-missing" finding + gate decision), not a + # tool failure. It still emits JSON in that case, so a non-zero exit with NO + # output (or unparseable output) is a genuine tool error, not "no license". + stdout = (result.stdout or "").strip() + if not stdout and result.returncode != 0: + logger.error("licensee failed (exit %d): %s", result.returncode, (result.stderr or "").strip()) + return 1 + try: + data = json.loads(stdout or "{}") + except json.JSONDecodeError as exc: + logger.error("licensee failed (exit %d): %s", result.returncode, (result.stderr or stdout or str(exc)).strip()) + return 1 + + findings = evaluate_licenses( + data.get("licenses", []), + allowed, + data.get("matched_files", []), + ) + write_scan(scan_path, SCANNER, args.target_dir, findings) + logger.info("License scan complete: %d finding(s)", len(findings)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/mcp/run_no_user_code_scan.py b/scripts/mcp/run_no_user_code_scan.py new file mode 100644 index 00000000..c8404e5d --- /dev/null +++ b/scripts/mcp/run_no_user_code_scan.py @@ -0,0 +1,180 @@ +#!/usr/bin/env python3 +"""MCP Phase 1 "does not execute user-provided code" scan. + +Runs semgrep with the bundled offline ruleset (rules/no_user_code.yml) over an +MCP server repo and normalizes results into no-user-code-scan.json for +NoUserCodeGate. The ruleset covers Python/JS/TS/Go/Java; a file in an uncovered +language is not scanned, so the scan records a ``coverage`` block and warns when +zero files were scanned (a zero-finding pass then means "nothing to scan", not +"clean"). Keep the ruleset offline - do not use ``--config auto`` (network). + +Submissions must not be able to quietly suppress their own findings: inline +``# nosem`` / ``# nosemgrep`` comments are disabled (``--disable-nosem``). semgrep +still honors a committed ``.semgrepignore``, so its mere presence is flagged as a +high-severity finding - which fails a blocking gate - rather than being trusted. + +Usage: + python -m scripts.mcp.run_no_user_code_scan --reports-dir +""" + +from __future__ import annotations + +import argparse +import json +import logging +import sys +from pathlib import Path + +from scripts.mcp._common import make_finding, run_tool, write_scan + +logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") +logger = logging.getLogger(__name__) + +SCAN_FILENAME = "no-user-code-scan.json" +SCANNER = "semgrep" +# Offline multi-language ruleset (Python/JS/TS/Go/Java). See module docstring. +RULES_PATH = Path(__file__).parent / "rules" / "no_user_code.yml" + +# semgrep severity -> our severity. +_SEVERITY_MAP = {"ERROR": "high", "WARNING": "medium", "INFO": "low"} + +# File extensions -> the ruleset's covered languages, for the coverage report. +_EXT_LANG = { + ".py": "python", + ".pyi": "python", + ".js": "javascript", + ".jsx": "javascript", + ".mjs": "javascript", + ".cjs": "javascript", + ".ts": "typescript", + ".tsx": "typescript", + ".mts": "typescript", + ".cts": "typescript", + ".go": "go", + ".java": "java", +} + + +def coverage_summary(target_dir: Path, scanned: list[str]) -> dict: + """Summarize what semgrep actually inspected. + + ``scanned`` is semgrep's ``paths.scanned`` (only files in a covered language, so + uncovered languages are already excluded). ``target_files_total`` counts every + file, so a reader can see the scanned fraction. + """ + languages = sorted({_EXT_LANG.get(Path(p).suffix.lower(), "other") for p in scanned}) + total = sum(1 for p in target_dir.rglob("*") if p.is_file() and ".git" not in p.parts) + return { + "files_scanned": len(scanned), + "languages": languages, + "target_files_total": total, + } + + +def normalize(semgrep_results: list[dict]) -> list[dict]: + """Map semgrep JSON results to normalized findings.""" + findings: list[dict] = [] + for res in semgrep_results: + extra = res.get("extra", {}) + sev = _SEVERITY_MAP.get(str(extra.get("severity", "")).upper(), "medium") + findings.append( + make_finding( + severity=sev, + message=extra.get("message", "Dynamic/arbitrary code execution pattern"), + rule_id=res.get("check_id", "unknown"), + file_path=res.get("path"), + line=res.get("start", {}).get("line"), + ) + ) + return findings + + +def main() -> int: + parser = argparse.ArgumentParser(description="MCP Phase 1 no-user-code-execution scan (semgrep)") + parser.add_argument("target_dir", type=Path, help="MCP server repo to scan") + parser.add_argument( + "--reports-dir", + type=Path, + required=True, + help="Directory to write no-user-code-scan.json into", + ) + args = parser.parse_args() + + if not args.target_dir.is_dir(): + logger.error("Not a directory: %s", args.target_dir) + return 1 + if not RULES_PATH.is_file(): + logger.error("Ruleset missing: %s", RULES_PATH) + return 1 + + scan_path = args.reports_dir / SCAN_FILENAME + + # semgrep honors a committed .semgrepignore, which could silently drop files + # from the scan. We cannot disable that per-file, so flag its presence as a + # high-severity finding (a blocking gate then fails rather than over-passing on + # a scan the submission narrowed). + findings: list[dict] = [] + for ignore in sorted(args.target_dir.rglob(".semgrepignore")): + if ignore.is_file(): + rel = ignore.relative_to(args.target_dir) + findings.append( + make_finding( + severity="high", + message=( + f"Submission contains a .semgrepignore ({rel}); semgrep honors it and may " + "exclude files from the scan, so its presence fails this gate." + ), + rule_id="semgrep-ignore-present", + file_path=str(rel), + ) + ) + + result = run_tool( + [ + "semgrep", + "scan", + "--config", + str(RULES_PATH), + "--json", + "--quiet", + "--metrics=off", # stay fully offline (no telemetry call) + "--disable-version-check", # no phone-home version check (stay offline) + "--disable-nosem", # ignore submission `# nosem` / `# nosemgrep` suppression comments + "--no-git-ignore", # scan every file in the artifact, not just tracked + str(args.target_dir), + ] + ) + + # semgrep exits 0 on success (with or without findings); non-zero = tool error. + if result.returncode != 0: + logger.error("semgrep failed (exit %d): %s", result.returncode, result.stderr.strip()) + return 1 + + try: + data = json.loads(result.stdout or "{}") + except json.JSONDecodeError as exc: + logger.error("Could not parse semgrep JSON: %s", exc) + return 1 + + findings += normalize(data.get("results", [])) + coverage = coverage_summary(args.target_dir, data.get("paths", {}).get("scanned", [])) + write_scan(scan_path, SCANNER, args.target_dir, findings, extra={"coverage": coverage}) + if coverage["files_scanned"] == 0: + logger.warning( + "no-user-code: semgrep scanned 0 files in a covered language (%d files in artifact); " + "this PASS does not show the server avoids user-code execution - add rules for its language", + coverage["target_files_total"], + ) + else: + logger.info( + "No-user-code scan complete: %d finding(s); scanned %d/%d files (%s)", + len(findings), + coverage["files_scanned"], + coverage["target_files_total"], + ", ".join(coverage["languages"]), + ) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/mcp/run_phase1_gates.py b/scripts/mcp/run_phase1_gates.py new file mode 100644 index 00000000..fb3b0ec0 --- /dev/null +++ b/scripts/mcp/run_phase1_gates.py @@ -0,0 +1,94 @@ +#!/usr/bin/env python3 +"""Run the MCP Phase 1 gates over the normalized scan reports. + +Reads the three scan JSON files produced by the Phase 1 scanners, evaluates each +gate, and writes the results back into the reports directory: + +- ``-result.json`` per gate (the full GateResult) +- ``phase1-summary.json`` (overall pass/fail + per-gate summary) + +Reporting stub: the three-state Evaluation-results-database writer is added by the +pipeline layer later; Phase 1 checks are pass/fail only. + +Usage: + python -m scripts.mcp.run_phase1_gates --reports-dir [--mode block|warn] +""" + +from __future__ import annotations + +import argparse +import json +import logging +import sys +from pathlib import Path + +from abevalflow.mcp.phase1 import run_phase1 +from abevalflow.schemas import GateMode, GatePolicy + +logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") +logger = logging.getLogger(__name__) + +SUMMARY_FILENAME = "phase1-summary.json" + + +def main() -> int: + parser = argparse.ArgumentParser(description="Run MCP Phase 1 gates") + parser.add_argument( + "--reports-dir", + type=Path, + required=True, + help="Directory holding the Phase 1 scan JSON files", + ) + parser.add_argument( + "--mode", + choices=[m.value for m in GateMode], + default=GateMode.BLOCK.value, + help="Gate enforcement mode (default: block)", + ) + args = parser.parse_args() + + if not args.reports_dir.is_dir(): + logger.error("Not a directory: %s", args.reports_dir) + return 1 + + policy = GatePolicy(default_mode=GateMode(args.mode)) + results = run_phase1(args.reports_dir, policy) + + summary: list[dict] = [] + for result in results: + gate_key = result.get_policy_key() + (args.reports_dir / f"{gate_key}-result.json").write_text(result.model_dump_json(indent=2)) + summary.append( + { + "gate": gate_key, + "passed": result.passed, + "score": result.score, + "findings": len(result.findings), + "message": result.message, + } + ) + logger.info( + "%s: passed=%s score=%.2f findings=%d", + gate_key, + result.passed, + result.score, + len(result.findings), + ) + + overall_passed = all(r.passed for r in results) + (args.reports_dir / SUMMARY_FILENAME).write_text( + json.dumps( + {"phase": "phase1", "passed": overall_passed, "gates": summary}, + indent=2, + ) + ) + + logger.info("Phase 1 overall: %s", "PASS" if overall_passed else "FAIL") + # Exit non-zero in block mode when any gate failed, so the Tekton step fails. + if not overall_passed and args.mode == GateMode.BLOCK.value: + return 1 + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/mcp/run_phase2_gates.py b/scripts/mcp/run_phase2_gates.py new file mode 100644 index 00000000..b4db026b --- /dev/null +++ b/scripts/mcp/run_phase2_gates.py @@ -0,0 +1,93 @@ +#!/usr/bin/env python3 +"""Run the MCP Phase 2 gates over the three-state check result files. + +Reads the ``-check.json`` files produced by the Phase 2 probe and the +Compass-consume step, evaluates each gate, and writes results into the reports +directory: + +- ``-result.json`` per gate (the full GateResult) +- ``phase2-summary.json`` (overall pass/fail + per-check summary with status) + +Three-state aware: a ``not_evaluated`` check never fails the phase. The step +fails (non-zero exit) only when a check genuinely FAILED in block mode. Reporting +to the Evaluation-results database is still handled by the pipeline layer later; +this writes local JSON only. + +Usage: + python -m scripts.mcp.run_phase2_gates --reports-dir [--mode block|warn] +""" + +from __future__ import annotations + +import argparse +import json +import logging +import sys +from pathlib import Path + +from abevalflow.mcp.phase2 import run_phase2 +from abevalflow.schemas import GateMode, GatePolicy + +logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") +logger = logging.getLogger(__name__) + +SUMMARY_FILENAME = "phase2-summary.json" + + +def main() -> int: + parser = argparse.ArgumentParser(description="Run MCP Phase 2 gates") + parser.add_argument( + "--reports-dir", type=Path, required=True, help="Directory holding the Phase 2 check result files" + ) + parser.add_argument( + "--mode", + choices=[m.value for m in GateMode], + default=GateMode.BLOCK.value, + help="Gate enforcement mode (default: block)", + ) + args = parser.parse_args() + + if not args.reports_dir.is_dir(): + logger.error("Not a directory: %s", args.reports_dir) + return 1 + + policy = GatePolicy(default_mode=GateMode(args.mode)) + results = run_phase2(args.reports_dir, policy) + + summary: list[dict] = [] + for result in results: + check_key = result.get_policy_key() + status = result.details.get("status", "unknown") + (args.reports_dir / f"{check_key}-result.json").write_text(result.model_dump_json(indent=2)) + summary.append( + { + "check": check_key, + "status": status, + "passed": result.passed, + "score": result.score, + "findings": len(result.findings), + "source": result.details.get("source"), + "message": result.message, + } + ) + logger.info("%s: status=%s passed=%s score=%.2f", check_key, status, result.passed, result.score) + + overall_passed = all(r.passed for r in results) + counts = { + "pass": sum(1 for s in summary if s["status"] == "pass"), + "fail": sum(1 for s in summary if s["status"] == "fail"), + "not_evaluated": sum(1 for s in summary if s["status"] == "not_evaluated"), + } + (args.reports_dir / SUMMARY_FILENAME).write_text( + json.dumps({"phase": "phase2", "passed": overall_passed, "counts": counts, "checks": summary}, indent=2) + ) + + logger.info("Phase 2 overall: %s (%s)", "PASS" if overall_passed else "FAIL", counts) + # A gate is only not-passed when a real check FAILED in block mode + # (not_evaluated and warn-mode failures keep passed=True), so this exits + # non-zero exactly when the step should fail. + return 0 if overall_passed else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/mcp/run_phase2_probe.py b/scripts/mcp/run_phase2_probe.py new file mode 100644 index 00000000..41a3b3a4 --- /dev/null +++ b/scripts/mcp/run_phase2_probe.py @@ -0,0 +1,413 @@ +#!/usr/bin/env python3 +"""MCP Phase 2 live conformance probe. + +Connects to a running MCP server over Streamable HTTP and runs the deterministic +(no-LLM) contract/conformance checks that require a live server, writing one +three-state result file per check for the Phase 2 gates to read. + +Checks that Compass already collects as real automated facts (tool-name rules, +OAuth match) or as metadata (schema present, documentation) are NOT probed here - +they are consumed via ``scripts/mcp/run_compass_fetch.py`` instead. + +Usage: + python -m scripts.mcp.run_phase2_probe --server-url http://:8080/mcp \ + --reports-dir [--timeout 30] +""" + +from __future__ import annotations + +import argparse +import logging +import sys +import time +from pathlib import Path + +from scripts.mcp._common import make_finding +from scripts.mcp._mcp_client import MCPClient, MCPConnectionError, RpcResponse +from scripts.mcp._phase2 import ( + STATUS_FAIL, + STATUS_NOT_EVALUATED, + STATUS_PASS, + CheckOutcome, + write_check_result, +) + +logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") +logger = logging.getLogger(__name__) + + +# --------------------------------------------------------------------------- +# Individual checks. Each returns a CheckOutcome. +# --------------------------------------------------------------------------- + + +def check_http_support(client: MCPClient, init: RpcResponse) -> CheckOutcome: + """Server responds correctly over HTTP with a parseable JSON-RPC message.""" + findings = [] + if init.status_code != 200: + findings.append( + make_finding( + severity="high", + message=f"initialize returned HTTP {init.status_code}, expected 200", + rule_id="http-status", + ) + ) + if init.message is None: + findings.append( + make_finding( + severity="high", + message="initialize response body was not a parseable JSON-RPC message (json or SSE)", + rule_id="http-body", + ) + ) + if findings: + return CheckOutcome( + "http-support", STATUS_FAIL, "Server did not respond correctly over HTTP.", findings=findings + ) + return CheckOutcome("http-support", STATUS_PASS, f"Server responded over HTTP (content-type: {init.content_type}).") + + +def check_protocol_compliance(client: MCPClient, init: RpcResponse) -> CheckOutcome: + """initialize result, capability negotiation, JSON-RPC 2.0 correctness.""" + findings = [] + msg = init.message if isinstance(init.message, dict) else {} + + if msg.get("jsonrpc") != "2.0": + findings.append( + make_finding( + severity="high", + message=f"initialize response jsonrpc={msg.get('jsonrpc')!r}, expected '2.0'", + rule_id="jsonrpc-version", + ) + ) + if msg.get("id") != 1: + findings.append( + make_finding( + severity="high", + message=f"initialize response id={msg.get('id')!r} did not echo request id 1", + rule_id="id-echo", + ) + ) + + result = msg.get("result", {}) if isinstance(msg.get("result"), dict) else {} + for required in ("protocolVersion", "capabilities", "serverInfo"): + if required not in result: + findings.append( + make_finding( + severity="high", + message=f"InitializeResult missing '{required}' (capability negotiation)", + rule_id="init-result", + ) + ) + + # Unknown method must yield a JSON-RPC error (ideally -32601 method not found). + try: + unknown = client.post({"jsonrpc": "2.0", "id": 90, "method": "no/such/method", "params": {}}) + err = unknown.message.get("error") if isinstance(unknown.message, dict) else None + if not err: + findings.append( + make_finding( + severity="medium", + message="unknown method did not return a JSON-RPC error object", + rule_id="error-unknown-method", + ) + ) + elif not isinstance(err, dict): + # A non-object "error" (e.g. a bare string) is itself a JSON-RPC + # violation - and calling .get() on it would crash the probe. + findings.append( + make_finding( + severity="medium", + message=f"unknown method returned a non-object error ({type(err).__name__}): {err!r}", + rule_id="error-not-object", + ) + ) + elif err.get("code") != -32601: + findings.append( + make_finding( + severity="low", + message=f"unknown method returned error code {err.get('code')}, expected -32601", + rule_id="error-code", + ) + ) + + # Malformed request (bad jsonrpc version) must not be accepted as success. + bad = client.post({"jsonrpc": "1.0", "id": 91, "method": "tools/list", "params": {}}) + bad_err = bad.message.get("error") if isinstance(bad.message, dict) else None + if bad.status_code == 200 and not bad_err: + findings.append( + make_finding( + severity="medium", + message="malformed request (jsonrpc 1.0) was accepted without an error", + rule_id="error-malformed", + ) + ) + except MCPConnectionError as exc: + return CheckOutcome("protocol-compliance", STATUS_NOT_EVALUATED, f"Server became unreachable mid-probe: {exc}") + + if findings: + return CheckOutcome( + "protocol-compliance", STATUS_FAIL, "Protocol/JSON-RPC 2.0 conformance violations found.", findings=findings + ) + return CheckOutcome( + "protocol-compliance", + STATUS_PASS, + "initialize, capability negotiation, and JSON-RPC 2.0 error handling conform.", + ) + + +def _tools_from(tools_resp: RpcResponse) -> list[dict]: + if isinstance(tools_resp.message, dict): + result = tools_resp.message.get("result", {}) + if isinstance(result, dict): + tools = result.get("tools", []) + if isinstance(tools, list): + return [t for t in tools if isinstance(t, dict)] + return [] + + +def _schema_is_object(schema: object) -> bool: + """A well-formed JSON-Schema object node. + + Requires the declared ``type`` to permit ``object`` (JSON Schema allows a type + array like ``["object", "null"]``, so that is accepted; a conflicting type like + ``{"type": "array", "properties": {}}`` is rejected), or, when no type is + declared, at least one object keyword. Object keywords, when present, must have + the right shape - ``properties`` / ``patternProperties`` must be mappings - so + ``{"type": "object", "properties": []}`` is rejected too. + """ + if not isinstance(schema, dict): + return False + declared_type = schema.get("type") + # type may be a string or, per JSON Schema, a list of allowed types. + type_permits_object = declared_type is None or "object" in ( + declared_type if isinstance(declared_type, list) else [declared_type] + ) + if declared_type is not None and not type_permits_object: + return False + object_keywords = ("properties", "patternProperties", "additionalProperties") + if declared_type is None and not any(k in schema for k in object_keywords): + return False + for mapping_keyword in ("properties", "patternProperties"): + value = schema.get(mapping_keyword) + if value is not None and not isinstance(value, dict): + return False + return True + + +def check_schema_conformance(client: MCPClient, tools_resp: RpcResponse) -> CheckOutcome: + """Every tool declares a well-formed input schema, plus a well-formed output + schema when one is present. + + inputSchema is required by the MCP tool type, so a missing/non-object one is a + fail. outputSchema is optional in the MCP spec, so it is validated only when + present (its absence is not a violation). Whether actual tool RESPONSES conform + to the declared schema is behavioral and is left to Phase 3. + """ + tools = _tools_from(tools_resp) + if not tools: + return CheckOutcome("schema-conformance", STATUS_NOT_EVALUATED, "tools/list returned no tools to validate.") + findings = [] + for tool in tools: + name = tool.get("name", "") + schema = tool.get("inputSchema") + if not isinstance(schema, dict): + findings.append( + make_finding( + severity="high", message=f"tool '{name}' has no object inputSchema", rule_id="schema-missing" + ) + ) + elif not _schema_is_object(schema): + findings.append( + make_finding( + severity="medium", + message=f"tool '{name}' inputSchema is not a JSON-Schema object (no type/properties)", + rule_id="schema-shape", + ) + ) + # outputSchema is optional; validate its shape only when the tool declares one. + out_schema = tool.get("outputSchema") + if out_schema is not None and not _schema_is_object(out_schema): + findings.append( + make_finding( + severity="medium", + message=f"tool '{name}' outputSchema is present but not a JSON-Schema object", + rule_id="output-schema-shape", + ) + ) + if findings: + return CheckOutcome( + "schema-conformance", STATUS_FAIL, "One or more tool schemas are not well-formed.", findings=findings + ) + return CheckOutcome("schema-conformance", STATUS_PASS, f"All {len(tools)} tool schemas are well-formed.") + + +_HINT_FIELDS = ("readOnlyHint", "destructiveHint", "idempotentHint", "openWorldHint") + + +def check_tool_annotations(client: MCPClient, tools_resp: RpcResponse) -> CheckOutcome: + """Validate tool behavior-hint annotations when present. + + Annotations are OPTIONAL in the MCP spec, so their absence is not a + conformance violation: a server that declares none is not_evaluated here + (nothing to verify), never failed. Only a *malformed* annotation (a hint + field present with a non-boolean value) is a real fail. Honesty of the hints + is behavioral and is left to Phase 3. + """ + tools = _tools_from(tools_resp) + if not tools: + return CheckOutcome("tool-annotations", STATUS_NOT_EVALUATED, "tools/list returned no tools to inspect.") + + findings = [] + annotated = 0 + for tool in tools: + name = tool.get("name", "") + annotations = tool.get("annotations") + if not isinstance(annotations, dict): + continue + present_hints = [h for h in _HINT_FIELDS if h in annotations] + if present_hints: + annotated += 1 + for hint in present_hints: + if not isinstance(annotations[hint], bool): + findings.append( + make_finding( + severity="medium", + message=f"tool '{name}' annotation '{hint}' is not a boolean", + rule_id="annotation-malformed", + ) + ) + + if findings: + return CheckOutcome( + "tool-annotations", STATUS_FAIL, "One or more tool annotations are malformed.", findings=findings + ) + if annotated == 0: + return CheckOutcome( + "tool-annotations", + STATUS_NOT_EVALUATED, + "No tools declare behavior-hint annotations (optional in the MCP spec); nothing to verify.", + ) + return CheckOutcome("tool-annotations", STATUS_PASS, f"All {annotated} annotated tool(s) have well-formed hints.") + + +def check_rate_limiting(client: MCPClient, *, burst: int = 20) -> CheckOutcome: + """Send a burst of requests; pass if the server throttles (HTTP 429).""" + try: + statuses = [client.list_tools(request_id=1000 + i).status_code for i in range(burst)] + except MCPConnectionError as exc: + return CheckOutcome("rate-limiting", STATUS_NOT_EVALUATED, f"Server unreachable during burst: {exc}") + if 429 in statuses: + return CheckOutcome( + "rate-limiting", STATUS_PASS, f"Server returned HTTP 429 under a burst of {burst} requests." + ) + return CheckOutcome( + "rate-limiting", + STATUS_NOT_EVALUATED, + f"No throttling observed under a burst of {burst}; cannot tell 'no limit' from a limit above the burst.", + ) + + +def check_response_size_limit(client: MCPClient) -> CheckOutcome: + """Response-size capping is not determinable black-box without a large-output tool + declared cap.""" + return CheckOutcome( + "response-size-limit", + STATUS_NOT_EVALUATED, + "Requires a tool designed to elicit large output and a declared size cap; not determinable black-box.", + ) + + +def check_mandatory_timeouts(client: MCPClient) -> CheckOutcome: + """Server-side timeout enforcement is not observable black-box without a hanging tool.""" + return CheckOutcome( + "mandatory-timeouts", + STATUS_NOT_EVALUATED, + "Server-side timeout enforcement is not observable black-box without a tool that intentionally hangs.", + ) + + +def check_openapi_conformance(client: MCPClient) -> CheckOutcome: + """OpenAPI conformance - deferred (not_evaluated). + + The ADR wants requests/responses validated against the server's OpenAPI + contract (ADR lines 103, 209, 257), but two prerequisites are missing: no MCP + repo publishes a contract at the standardized location yet, and the + OpenAPI-operation <-> MCP-tool/method mapping is undefined. Wiring a validator + before both exist would invent that convention, so this stays not_evaluated + (never a fail) until they are settled. + """ + return CheckOutcome( + "openapi-conformance", + STATUS_NOT_EVALUATED, + "No OpenAPI contract provided and the contract<->tool mapping is unspecified; " + "deferred until an MCP repo publishes a contract at the standardized location.", + ) + + +def probe_all(server_url: str, *, timeout: float = 30.0) -> list[CheckOutcome]: + """Run every live Phase 2 check against ``server_url``. + + If the server is unreachable, all checks are reported not_evaluated (ADR: + a readiness/reachability failure must never be scored as a fail). + """ + live_names = [ + "http-support", + "protocol-compliance", + "schema-conformance", + "tool-annotations", + "rate-limiting", + "response-size-limit", + "mandatory-timeouts", + "openapi-conformance", + ] + with MCPClient(server_url, timeout=timeout) as client: + try: + init = client.initialize() + client.send_initialized() + # Aggregate every tools/list page (follows nextCursor) so schema and + # annotation checks see tools beyond page one. Wrap the full set in a + # synthetic response so the per-check helpers keep one interface. + all_tools = client.list_all_tools() + tools_resp = RpcResponse( + status_code=200, + headers={}, + content_type="application/json", + message={"jsonrpc": "2.0", "id": 2, "result": {"tools": all_tools}}, + raw_text="", + ) + except MCPConnectionError as exc: + reason = f"Server unreachable at {server_url}: {exc}" + logger.warning(reason) + return [CheckOutcome(name, STATUS_NOT_EVALUATED, reason) for name in live_names] + + return [ + check_http_support(client, init), + check_protocol_compliance(client, init), + check_schema_conformance(client, tools_resp), + check_tool_annotations(client, tools_resp), + check_rate_limiting(client), + check_response_size_limit(client), + check_mandatory_timeouts(client), + check_openapi_conformance(client), + ] + + +def main() -> int: + parser = argparse.ArgumentParser(description="MCP Phase 2 live conformance probe") + parser.add_argument("--server-url", required=True, help="MCP endpoint URL, e.g. http://:8080/mcp") + parser.add_argument( + "--reports-dir", type=Path, required=True, help="Directory to write -check.json files into" + ) + parser.add_argument("--timeout", type=float, default=30.0, help="Per-request HTTP timeout in seconds") + args = parser.parse_args() + + start = time.monotonic() + outcomes = probe_all(args.server_url, timeout=args.timeout) + for outcome in outcomes: + write_check_result(args.reports_dir, outcome) + logger.info("Phase 2 probe complete: %d checks in %.1fs", len(outcomes), time.monotonic() - start) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/mcp/run_secrets_scan.py b/scripts/mcp/run_secrets_scan.py new file mode 100644 index 00000000..c57720e9 --- /dev/null +++ b/scripts/mcp/run_secrets_scan.py @@ -0,0 +1,105 @@ +#!/usr/bin/env python3 +"""MCP Phase 1 secrets scan. + +Runs gitleaks over an MCP server repository (static working-tree scan, no live +verification) and normalizes its output into secrets-scan.json for SecretsGate. + +Usage: + python -m scripts.mcp.run_secrets_scan --reports-dir +""" + +from __future__ import annotations + +import argparse +import json +import logging +import sys +import tempfile +from pathlib import Path + +from scripts.mcp._common import make_finding, run_tool, write_scan + +logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") +logger = logging.getLogger(__name__) + +SCAN_FILENAME = "secrets-scan.json" +SCANNER = "gitleaks" + +# gitleaks has no severity field; every hit is a candidate hardcoded credential, +# so all findings are HIGH (the gate blocks on HIGH/CRITICAL in block mode). + + +def normalize(gitleaks_findings: list[dict]) -> list[dict]: + """Map gitleaks JSON records to normalized findings.""" + findings: list[dict] = [] + for rec in gitleaks_findings: + rule_id = rec.get("RuleID", "unknown") + findings.append( + make_finding( + severity="high", + message=rec.get("Description", "Hardcoded secret detected"), + rule_id=rule_id, + file_path=rec.get("File"), + line=rec.get("StartLine"), + ) + ) + return findings + + +def main() -> int: + parser = argparse.ArgumentParser(description="MCP Phase 1 secrets scan (gitleaks)") + parser.add_argument("target_dir", type=Path, help="MCP server repo to scan") + parser.add_argument( + "--reports-dir", + type=Path, + required=True, + help="Directory to write secrets-scan.json into", + ) + args = parser.parse_args() + + if not args.target_dir.is_dir(): + logger.error("Not a directory: %s", args.target_dir) + return 1 + + scan_path = args.reports_dir / SCAN_FILENAME + + with tempfile.NamedTemporaryFile("r", suffix=".json", delete=False) as tmp: + report_path = Path(tmp.name) + + # gitleaks `dir` scans the working tree (no git history required). + result = run_tool( + [ + "gitleaks", + "dir", + str(args.target_dir), + "--report-format", + "json", + "--report-path", + str(report_path), + "--no-banner", + "--exit-code", + "0", # never fail the process on findings; the gate decides pass/fail + ] + ) + + # Exit codes: 0 clean/normal. Any output on a crash goes to stderr. + if result.returncode != 0: + logger.error("gitleaks failed (exit %d): %s", result.returncode, result.stderr.strip()) + return 1 + + try: + raw = json.loads(report_path.read_text() or "[]") + except (json.JSONDecodeError, OSError) as exc: + logger.error("Could not read gitleaks report: %s", exc) + return 1 + finally: + report_path.unlink(missing_ok=True) + + findings = normalize(raw if isinstance(raw, list) else []) + write_scan(scan_path, SCANNER, args.target_dir, findings) + logger.info("Secrets scan complete: %d finding(s)", len(findings)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/test_mcp_phase1.py b/tests/test_mcp_phase1.py new file mode 100644 index 00000000..67fe202f --- /dev/null +++ b/tests/test_mcp_phase1.py @@ -0,0 +1,311 @@ +"""Tests for MCP Phase 1 static / build-time checks. + +Two layers: + +- Unit tests exercise the normalization functions and the gate pass/fail logic + with synthetic scan JSON. No external tools required - safe in any CI. +- Integration tests build a fixture repo in a temp dir and run the real scanner + (gitleaks / semgrep / licensee) end-to-end. Skipped when the tool is absent. +""" + +from __future__ import annotations + +import json +import secrets +import shutil +import string +from pathlib import Path + +import pytest + +from abevalflow.mcp.phase1 import ( + DEFAULT_ALLOWED_LICENSES, + LicenseGate, + NoUserCodeGate, + SecretsGate, + run_phase1, +) +from abevalflow.schemas import GateMode, GatePolicy +from scripts.mcp import run_license_scan as license_scan +from scripts.mcp import run_no_user_code_scan as no_user_code_scan +from scripts.mcp import run_secrets_scan as secrets_scan + +BLOCK = GatePolicy(default_mode=GateMode.BLOCK) + + +def _write(reports_dir: Path, filename: str, findings: list[dict]) -> None: + (reports_dir / filename).write_text(json.dumps({"findings": findings})) + + +# --------------------------------------------------------------------------- +# Normalization unit tests (no external tools) +# --------------------------------------------------------------------------- + + +def test_secrets_normalize_maps_fields_and_severity(): + raw = [ + {"RuleID": "aws-access-token", "Description": "AWS key", "File": "a.py", "StartLine": 3}, + {"RuleID": "generic-api-key", "Description": "generic", "File": "b.py", "StartLine": 7}, + ] + findings = secrets_scan.normalize(raw) + assert [f["rule_id"] for f in findings] == ["aws-access-token", "generic-api-key"] + assert findings[0]["severity"] == "high" + assert findings[1]["severity"] == "high" # every gitleaks hit is HIGH + assert findings[0]["file_path"] == "a.py" + + +def test_no_user_code_normalize_maps_semgrep_severity(): + raw = [ + { + "check_id": "python-dynamic-exec", + "path": "bad.py", + "start": {"line": 2}, + "extra": {"severity": "ERROR", "message": "eval used"}, + } + ] + findings = no_user_code_scan.normalize(raw) + assert findings[0]["severity"] == "high" + assert findings[0]["rule_id"] == "python-dynamic-exec" + assert findings[0]["file_path"] == "bad.py" + + +def test_license_allowed_produces_no_findings(): + detected = [{"spdx_id": "MIT"}] + findings = license_scan.evaluate_licenses(detected, {"mit", "apache-2.0"}, []) + assert findings == [] + + +def test_license_disallowed_flags_high(): + detected = [{"spdx_id": "GPL-3.0"}] + matched = [{"filename": "LICENSE"}] + findings = license_scan.evaluate_licenses(detected, {"mit"}, matched) + assert len(findings) == 1 + assert findings[0]["severity"] == "high" + assert findings[0]["rule_id"] == "license-not-allowed" + assert findings[0]["file_path"] == "LICENSE" + + +def test_license_missing_flags_high(): + findings = license_scan.evaluate_licenses([], {"mit"}, []) + assert findings[0]["rule_id"] == "license-missing" + + +# --------------------------------------------------------------------------- +# Gate logic tests (synthetic scan JSON) +# --------------------------------------------------------------------------- + + +def test_secrets_gate_fails_on_high_in_block_mode(tmp_path): + _write(tmp_path, "secrets-scan.json", [{"severity": "high", "message": "leak", "rule_id": "r"}]) + result = SecretsGate().evaluate(tmp_path, BLOCK) + assert result.passed is False + + +def test_secrets_gate_passes_when_clean(tmp_path): + _write(tmp_path, "secrets-scan.json", []) + result = SecretsGate().evaluate(tmp_path, BLOCK) + assert result.passed is True + assert result.score == 1.0 + + +def test_run_phase1_all_clean_passes(tmp_path): + _write(tmp_path, "secrets-scan.json", []) + _write(tmp_path, "no-user-code-scan.json", []) + _write(tmp_path, "license-scan.json", []) + results = run_phase1(tmp_path, BLOCK) + assert len(results) == 3 + assert all(r.passed for r in results) + assert {r.policy_key for r in results} == {"mcp-secrets", "mcp-no-user-code", "mcp-license"} + + +def test_run_phase1_one_dirty_fails_that_gate(tmp_path): + _write(tmp_path, "secrets-scan.json", []) + _write(tmp_path, "no-user-code-scan.json", [{"severity": "high", "message": "eval", "rule_id": "x"}]) + _write(tmp_path, "license-scan.json", []) + results = {r.policy_key: r for r in run_phase1(tmp_path, BLOCK)} + assert results["mcp-secrets"].passed is True + assert results["mcp-no-user-code"].passed is False + assert results["mcp-license"].passed is True + + +# --------------------------------------------------------------------------- +# Scanner main() behavior with a faked tool (no real binary) +# --------------------------------------------------------------------------- + + +def _fake_result(returncode=0, stdout="", stderr=""): + import types + + return types.SimpleNamespace(returncode=returncode, stdout=stdout, stderr=stderr) + + +def test_license_scan_treats_nonzero_exit_with_json_as_valid(tmp_path, monkeypatch): + # licensee exits non-zero when no license is found; that JSON must still be + # parsed into a license-missing finding, not treated as a tool failure. + repo = tmp_path / "repo" + repo.mkdir() + reports = tmp_path / "reports" + reports.mkdir() + monkeypatch.setattr( + license_scan, + "run_tool", + lambda *a, **k: _fake_result(returncode=1, stdout='{"licenses": [], "matched_files": []}'), + ) + rc = _run_scanner(license_scan, [str(repo), "--reports-dir", str(reports)]) + assert rc == 0 + data = json.loads((reports / "license-scan.json").read_text()) + assert [f["rule_id"] for f in data["findings"]] == ["license-missing"] + + +def test_license_scan_fails_only_on_unparseable_output(tmp_path, monkeypatch): + repo = tmp_path / "repo" + repo.mkdir() + reports = tmp_path / "reports" + reports.mkdir() + monkeypatch.setattr( + license_scan, + "run_tool", + lambda *a, **k: _fake_result(returncode=1, stdout="not json", stderr="boom"), + ) + assert _run_scanner(license_scan, [str(repo), "--reports-dir", str(reports)]) == 1 + + +def test_license_scan_nonzero_exit_with_no_output_is_tool_error(tmp_path, monkeypatch): + # A genuine licensee crash (non-zero exit, empty stdout) must be a tool error, + # not silently reclassified as "license-missing". + repo = tmp_path / "repo" + repo.mkdir() + reports = tmp_path / "reports" + reports.mkdir() + monkeypatch.setattr( + license_scan, + "run_tool", + lambda *a, **k: _fake_result(returncode=2, stdout="", stderr="segfault"), + ) + assert _run_scanner(license_scan, [str(repo), "--reports-dir", str(reports)]) == 1 + assert not (reports / "license-scan.json").exists() + + +def test_no_user_code_scan_flags_committed_semgrepignore(tmp_path, monkeypatch): + repo = tmp_path / "repo" + repo.mkdir() + (repo / ".semgrepignore").write_text("secrets/\n") + reports = tmp_path / "reports" + reports.mkdir() + monkeypatch.setattr( + no_user_code_scan, + "run_tool", + lambda *a, **k: _fake_result(returncode=0, stdout='{"results": [], "paths": {"scanned": []}}'), + ) + rc = _run_scanner(no_user_code_scan, [str(repo), "--reports-dir", str(reports)]) + assert rc == 0 + data = json.loads((reports / "no-user-code-scan.json").read_text()) + assert "semgrep-ignore-present" in {f["rule_id"] for f in data["findings"]} + + +# --------------------------------------------------------------------------- +# Integration tests (real tools, skipped when absent) +# --------------------------------------------------------------------------- + +_MIT = """MIT License + +Copyright (c) 2026 Example + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. +""" + + +@pytest.mark.skipif(not shutil.which("gitleaks"), reason="gitleaks not installed") +def test_integration_secrets_detects_planted_key(tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + # Synthetic key generated at runtime - no hardcoded credential in this source + # (only the public "AKIA" prefix). gitleaks matches by pattern, so a generated + # AKIA+16 key still trips its AWS rule; the full value lives only in the temp file. + access_key = "AKIA" + "".join(secrets.choice(string.ascii_uppercase + string.digits) for _ in range(16)) + secret_key = "".join(secrets.choice(string.ascii_letters + string.digits) for _ in range(40)) + (repo / "config.py").write_text(f'AWS_ACCESS_KEY_ID = "{access_key}"\naws_secret = "{secret_key}"\n') + reports = tmp_path / "reports" + reports.mkdir() + rc = _run_scanner(secrets_scan, [str(repo), "--reports-dir", str(reports)]) + assert rc == 0 + result = SecretsGate().evaluate(reports, BLOCK) + assert result.passed is False + assert result.findings + + +@pytest.mark.skipif(not shutil.which("semgrep"), reason="semgrep not installed") +def test_integration_no_user_code_detects_eval(tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + (repo / "bad.py").write_text("def handler(x):\n return eval(x)\n") + reports = tmp_path / "reports" + reports.mkdir() + rc = _run_scanner(no_user_code_scan, [str(repo), "--reports-dir", str(reports)]) + assert rc == 0 + result = NoUserCodeGate().evaluate(reports, BLOCK) + assert result.passed is False + + +@pytest.mark.skipif(not shutil.which("semgrep"), reason="semgrep not installed") +def test_integration_no_user_code_clean_passes(tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + (repo / "good.py").write_text("def handler(x):\n return x.upper()\n") + reports = tmp_path / "reports" + reports.mkdir() + rc = _run_scanner(no_user_code_scan, [str(repo), "--reports-dir", str(reports)]) + assert rc == 0 + result = NoUserCodeGate().evaluate(reports, BLOCK) + assert result.passed is True + + +@pytest.mark.skipif(not shutil.which("licensee"), reason="licensee not installed") +def test_integration_license_mit_passes(tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + (repo / "LICENSE").write_text(_MIT) + reports = tmp_path / "reports" + reports.mkdir() + rc = _run_scanner( + license_scan, + [str(repo), "--reports-dir", str(reports), *_allow_args()], + ) + assert rc == 0 + result = LicenseGate().evaluate(reports, BLOCK) + assert result.passed is True + + +def _allow_args() -> list[str]: + args: list[str] = [] + for spdx in DEFAULT_ALLOWED_LICENSES: + args += ["--allow", spdx] + return args + + +def _run_scanner(module, argv: list[str]) -> int: + """Invoke a scanner module's main() with a patched argv.""" + import sys + + old = sys.argv + sys.argv = [module.__name__, *argv] + try: + return module.main() + finally: + sys.argv = old diff --git a/tests/test_mcp_phase2.py b/tests/test_mcp_phase2.py new file mode 100644 index 00000000..c50ba330 --- /dev/null +++ b/tests/test_mcp_phase2.py @@ -0,0 +1,435 @@ +"""Tests for MCP Phase 2 (deterministic contract/conformance). + +Unit tests use synthetic JSON-RPC responses and a fake client, so they need no +running server and are CI-safe. The integration test runs the real probe against +a live MCP server and is skipped unless MCP_TEST_SERVER_URL is set (e.g. a local +generic-mock-mcp-server). +""" + +from __future__ import annotations + +import json +import os + +import pytest + +from abevalflow.gates.base import GateMode +from abevalflow.mcp.phase2 import PHASE2_CHECKS, run_phase2 +from abevalflow.mcp.phase2.base import Phase2Gate +from abevalflow.schemas import GatePolicy +from scripts.mcp import run_compass_fetch as compass_fetch +from scripts.mcp import run_phase2_gates +from scripts.mcp import run_phase2_probe as phase2_probe +from scripts.mcp._mcp_client import RpcResponse, _parse_sse_messages, _select_response +from scripts.mcp._phase2 import STATUS_FAIL, STATUS_NOT_EVALUATED, STATUS_PASS, CheckOutcome, write_check_result + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + + +def _rpc(message, *, status=200, content_type="application/json", headers=None): + return RpcResponse( + status_code=status, + headers=headers or {}, + content_type=content_type, + message=message, + raw_text=json.dumps(message) if message is not None else "", + ) + + +class FakeClient: + """Routes .post()/.list_tools() to canned RpcResponses by JSON-RPC method.""" + + def __init__(self, by_method): + self._by_method = by_method + + def post(self, body, **_kw): + method = body.get("method") if isinstance(body, dict) else None + resp = self._by_method.get(method) + if resp is None: + return _rpc({"jsonrpc": "2.0", "id": body.get("id"), "error": {"code": -32601, "message": "not found"}}) + return resp + + def list_tools(self, request_id=2, cursor=None): + return self._by_method["tools/list"] + + +# --------------------------------------------------------------------------- +# mcp_client +# --------------------------------------------------------------------------- + + +def test_parse_sse_extracts_json_rpc(): + body = 'event: message\ndata: {"jsonrpc": "2.0", "id": 1, "result": {}}\n\n' + messages = _parse_sse_messages(body) + assert messages == [{"jsonrpc": "2.0", "id": 1, "result": {}}] + assert _select_response(messages, 1) == {"jsonrpc": "2.0", "id": 1, "result": {}} + + +def test_parse_sse_returns_empty_without_data(): + assert _parse_sse_messages("event: ping\n\n") == [] + assert _select_response([], 1) is None + + +def test_select_response_skips_interleaved_notification(): + # A notification (no id) is streamed before the id-matched response; the + # matching frame must be chosen, not the first. + body = ( + 'data: {"jsonrpc": "2.0", "method": "notifications/progress"}\n\n' + 'data: {"jsonrpc": "2.0", "id": 7, "result": {"ok": true}}\n\n' + ) + messages = _parse_sse_messages(body) + assert _select_response(messages, 7) == {"jsonrpc": "2.0", "id": 7, "result": {"ok": True}} + + +def test_select_response_falls_back_to_first_when_no_match(): + messages = [{"jsonrpc": "2.0", "method": "notifications/progress"}] + assert _select_response(messages, 99) == messages[0] + assert _select_response(messages, None) == messages[0] + + +def test_select_response_matches_string_id_against_int_request(): + # A server that echoes the id as "2" still matches an int request id 2, + # rather than falling through to an interleaved notification frame. + messages = [ + {"jsonrpc": "2.0", "method": "notifications/progress"}, + {"jsonrpc": "2.0", "id": "2", "result": {"tools": []}}, + ] + assert _select_response(messages, 2) == {"jsonrpc": "2.0", "id": "2", "result": {"tools": []}} + + +def test_list_all_tools_follows_pagination(): + from scripts.mcp._mcp_client import MCPClient + + pages = { + None: {"tools": [{"name": "a"}], "nextCursor": "c1"}, + "c1": {"tools": [{"name": "b"}], "nextCursor": "c2"}, + "c2": {"tools": [{"name": "c"}]}, # no nextCursor -> last page + } + client = MCPClient.__new__(MCPClient) # skip __init__ (no real socket) + + def fake_list_tools(*, request_id=2, cursor=None): + return _rpc({"jsonrpc": "2.0", "id": request_id, "result": pages[cursor]}) + + client.list_tools = fake_list_tools + names = [t["name"] for t in client.list_all_tools()] + assert names == ["a", "b", "c"] + + +def test_list_all_tools_breaks_on_repeating_cursor(): + from scripts.mcp._mcp_client import MCPClient + + client = MCPClient.__new__(MCPClient) + + def fake_list_tools(*, request_id=2, cursor=None): + # Always returns the same cursor -> would loop forever without dedup. + return _rpc({"jsonrpc": "2.0", "id": request_id, "result": {"tools": [{"name": "x"}], "nextCursor": "same"}}) + + client.list_tools = fake_list_tools + tools = client.list_all_tools(max_pages=100) + # Two pages consumed (initial None cursor, then "same"), then dedup breaks it. + assert len(tools) == 2 + + +# --------------------------------------------------------------------------- +# Probe checks (synthetic responses) +# --------------------------------------------------------------------------- + + +def _init_ok(): + return _rpc( + { + "jsonrpc": "2.0", + "id": 1, + "result": {"protocolVersion": "2025-06-18", "capabilities": {}, "serverInfo": {"name": "x"}}, + } + ) + + +def test_http_support_pass_and_fail(): + assert phase2_probe.check_http_support(None, _init_ok()).status == STATUS_PASS + bad = _rpc(None, status=500, content_type="text/plain") + out = phase2_probe.check_http_support(None, bad) + assert out.status == STATUS_FAIL + assert len(out.findings) == 2 # bad status + unparseable body + + +def test_protocol_compliance_pass(): + client = FakeClient( + { + "no/such/method": _rpc({"jsonrpc": "2.0", "id": 90, "error": {"code": -32601, "message": "nope"}}), + "tools/list": _rpc({"jsonrpc": "2.0", "id": 91, "error": {"code": -32600, "message": "bad"}}), + } + ) + out = phase2_probe.check_protocol_compliance(client, _init_ok()) + assert out.status == STATUS_PASS, out.findings + + +def test_protocol_compliance_flags_missing_result_and_bad_id(): + init = _rpc({"jsonrpc": "2.0", "id": 999, "result": {"capabilities": {}}}) # wrong id, missing fields + client = FakeClient( + { + "no/such/method": _rpc({"jsonrpc": "2.0", "id": 90, "error": {"code": -32601}}), + "tools/list": _rpc({"jsonrpc": "2.0", "id": 91, "error": {"code": -32600}}), + } + ) + out = phase2_probe.check_protocol_compliance(client, init) + assert out.status == STATUS_FAIL + rule_ids = {f["rule_id"] for f in out.findings} + assert "id-echo" in rule_ids + assert "init-result" in rule_ids + + +def test_schema_conformance_pass_and_fail(): + good = _rpc({"result": {"tools": [{"name": "a", "inputSchema": {"type": "object", "properties": {}}}]}}) + assert phase2_probe.check_schema_conformance(None, good).status == STATUS_PASS + + bad = _rpc({"result": {"tools": [{"name": "b"}]}}) # no inputSchema + out = phase2_probe.check_schema_conformance(None, bad) + assert out.status == STATUS_FAIL + + empty = _rpc({"result": {"tools": []}}) + assert phase2_probe.check_schema_conformance(None, empty).status == STATUS_NOT_EVALUATED + + +def test_schema_conformance_output_schema_when_present(): + # Valid input + valid output -> pass. + both = _rpc( + {"result": {"tools": [{"name": "a", "inputSchema": {"type": "object"}, "outputSchema": {"type": "object"}}]}} + ) + assert phase2_probe.check_schema_conformance(None, both).status == STATUS_PASS + + # Present-but-malformed output schema -> fail (optional, but must be well-formed when declared). + bad_out = _rpc({"result": {"tools": [{"name": "b", "inputSchema": {"type": "object"}, "outputSchema": "nope"}]}}) + out = phase2_probe.check_schema_conformance(None, bad_out) + assert out.status == STATUS_FAIL + assert out.findings[0]["rule_id"] == "output-schema-shape" + + # No output schema at all -> not a violation (optional in the MCP spec). + no_out = _rpc({"result": {"tools": [{"name": "c", "inputSchema": {"type": "object"}}]}}) + assert phase2_probe.check_schema_conformance(None, no_out).status == STATUS_PASS + + +def test_schema_is_object_rejects_conflicting_type_and_bad_properties(): + # Declared type conflicts with object -> not an object schema. + assert phase2_probe._schema_is_object({"type": "array", "properties": {}}) is False + # type object but properties is not a mapping -> malformed. + assert phase2_probe._schema_is_object({"type": "object", "properties": []}) is False + # No type but an object keyword present -> accepted. + assert phase2_probe._schema_is_object({"properties": {}}) is True + # No type and no object keyword -> not an object schema. + assert phase2_probe._schema_is_object({"description": "x"}) is False + # Not a dict at all. + assert phase2_probe._schema_is_object("nope") is False + # JSON Schema allows a type array; one that permits "object" is accepted. + assert phase2_probe._schema_is_object({"type": ["object", "null"], "properties": {}}) is True + # A type array that does NOT include "object" is rejected. + assert phase2_probe._schema_is_object({"type": ["array", "null"]}) is False + + +def test_protocol_compliance_flags_non_object_error(): + # A bare-string "error" is a JSON-RPC violation and must not crash the probe. + client = FakeClient( + { + "no/such/method": _rpc({"jsonrpc": "2.0", "id": 90, "error": "boom"}), + "tools/list": _rpc({"jsonrpc": "2.0", "id": 91, "error": {"code": -32600}}), + } + ) + out = phase2_probe.check_protocol_compliance(client, _init_ok()) + assert out.status == STATUS_FAIL + assert "error-not-object" in {f["rule_id"] for f in out.findings} + + +def test_tool_annotations_absent_is_not_evaluated(): + # Annotations are optional in the MCP spec: none declared -> not_evaluated. + resp = _rpc({"result": {"tools": [{"name": "a"}, {"name": "b"}]}}) + assert phase2_probe.check_tool_annotations(None, resp).status == STATUS_NOT_EVALUATED + + +def test_tool_annotations_wellformed_passes(): + resp = _rpc({"result": {"tools": [{"name": "a", "annotations": {"readOnlyHint": True}}, {"name": "b"}]}}) + assert phase2_probe.check_tool_annotations(None, resp).status == STATUS_PASS + + +def test_tool_annotations_malformed_fails(): + resp = _rpc({"result": {"tools": [{"name": "a", "annotations": {"readOnlyHint": "yes"}}]}}) + out = phase2_probe.check_tool_annotations(None, resp) + assert out.status == STATUS_FAIL + assert out.findings[0]["rule_id"] == "annotation-malformed" + + +def test_rate_limiting_pass_on_429(): + class Burst: + def list_tools(self, request_id=0): + return _rpc({}, status=429 if request_id == 1005 else 200) + + assert phase2_probe.check_rate_limiting(Burst(), burst=10).status == STATUS_PASS + + +def test_rate_limiting_not_evaluated_without_throttle(): + class Burst: + def list_tools(self, request_id=0): + return _rpc({}, status=200) + + assert phase2_probe.check_rate_limiting(Burst(), burst=5).status == STATUS_NOT_EVALUATED + + +def test_deferred_checks_are_not_evaluated(): + assert phase2_probe.check_response_size_limit(None).status == STATUS_NOT_EVALUATED + assert phase2_probe.check_mandatory_timeouts(None).status == STATUS_NOT_EVALUATED + assert phase2_probe.check_openapi_conformance(None).status == STATUS_NOT_EVALUATED + + +# --------------------------------------------------------------------------- +# Compass consume +# --------------------------------------------------------------------------- + + +def test_map_facts_pass(): + facts = { + "mcp:default/tools": {"allToolNamesValid": True}, + "mcp:default/security": { + "oauth": {"enforced": True}, + "entityAuth": {"authServerMatch": True}, + "scopeMatch": {"allScopesMatch": True}, + }, + } + by_check = {o.check: o for o in compass_fetch.map_facts(facts)} + assert set(by_check) == {"tool-name-rules", "oauth-catalog-match"} + assert by_check["tool-name-rules"].status == STATUS_PASS + assert by_check["oauth-catalog-match"].status == STATUS_PASS + assert all(o.source == "compass" for o in by_check.values()) + + +def test_map_facts_fail_and_missing(): + facts = {"mcp:default/tools": {"allToolNamesValid": False}} + by_check = {o.check: o for o in compass_fetch.map_facts(facts)} + assert by_check["tool-name-rules"].status == STATUS_FAIL + assert by_check["oauth-catalog-match"].status == STATUS_NOT_EVALUATED # no OAuth facts + + +def test_map_facts_oauth_partial_is_not_evaluated(): + # Only one OAuth subfield present -> cannot assert the conjunction -> not_evaluated, + # never a fail (a partially-reported entity must not be blocked). + facts = {"mcp:default/security": {"oauth": {"enforced": True}}} + out = {o.check: o for o in compass_fetch.map_facts(facts)}["oauth-catalog-match"] + assert out.status == STATUS_NOT_EVALUATED + assert out.findings == [] + + +def test_map_facts_oauth_explicit_false_fails_only_for_false(): + # An explicit False is a real violation; an absent sibling is not. + facts = { + "mcp:default/security": { + "oauth": {"enforced": True}, + "entityAuth": {"authServerMatch": False}, + # scopeMatch absent + } + } + out = {o.check: o for o in compass_fetch.map_facts(facts)}["oauth-catalog-match"] + assert out.status == STATUS_FAIL + assert {f["rule_id"] for f in out.findings} == {"oauth-server-match"} + + +def test_missing_facts_file_degrades_to_not_evaluated(monkeypatch, tmp_path): + monkeypatch.setattr( + "sys.argv", + ["compass_fetch", "--facts-file", str(tmp_path / "nope.json"), "--reports-dir", str(tmp_path)], + ) + assert compass_fetch.main() == 0 + for check in ("tool-name-rules", "oauth-catalog-match"): + data = json.loads((tmp_path / f"{check}-check.json").read_text()) + assert data["status"] == STATUS_NOT_EVALUATED + + +def test_malformed_facts_file_degrades_to_not_evaluated(monkeypatch, tmp_path): + bad = tmp_path / "bad.json" + bad.write_text("{ not valid json ") + monkeypatch.setattr( + "sys.argv", + ["compass_fetch", "--facts-file", str(bad), "--reports-dir", str(tmp_path)], + ) + assert compass_fetch.main() == 0 + data = json.loads((tmp_path / "tool-name-rules-check.json").read_text()) + assert data["status"] == STATUS_NOT_EVALUATED + + +# --------------------------------------------------------------------------- +# Gates + runner (three-state) +# --------------------------------------------------------------------------- + + +def _write(tmp_path, check, status, findings=None): + write_check_result(tmp_path, CheckOutcome(check, status, "reason", findings=findings or [])) + + +def test_gate_pass_fail_not_evaluated(tmp_path): + _write(tmp_path, "http-support", STATUS_PASS) + _write(tmp_path, "protocol-compliance", STATUS_FAIL, [{"severity": "high", "message": "m", "rule_id": "r"}]) + _write(tmp_path, "openapi-conformance", STATUS_NOT_EVALUATED) + # remaining checks have no file -> not_evaluated, must not fail + + policy = GatePolicy(default_mode=GateMode.BLOCK) + results = {r.get_policy_key(): r for r in run_phase2(tmp_path, policy)} + + assert len(results) == len(PHASE2_CHECKS) + assert results["http-support"].passed is True + assert results["protocol-compliance"].passed is False # fail in block mode + assert results["openapi-conformance"].passed is True # not_evaluated never penalized + assert results["mandatory-timeouts"].details["status"] == STATUS_NOT_EVALUATED # missing file + + +def test_corrupt_result_file_blocks_only_in_block_mode(tmp_path): + (tmp_path / "http-support-check.json").write_text("{ not valid json ") + gate = Phase2Gate("http-support") + warn = gate.evaluate(tmp_path, GatePolicy(default_mode=GateMode.WARN)) + assert warn.details["status"] == STATUS_FAIL + assert warn.passed is True # warn never blocks, even on a corrupt file + block = gate.evaluate(tmp_path, GatePolicy(default_mode=GateMode.BLOCK)) + assert block.passed is False + + +def test_fail_without_findings_scores_zero(tmp_path): + # A bare fail (no findings) must not report a perfect score. + write_check_result(tmp_path, CheckOutcome("http-support", STATUS_FAIL, "boom", findings=[])) + res = Phase2Gate("http-support").evaluate(tmp_path, GatePolicy(default_mode=GateMode.BLOCK)) + assert res.passed is False + assert res.score == 0.0 + + +def test_not_evaluated_does_not_fail_runner(tmp_path, monkeypatch): + _write(tmp_path, "http-support", STATUS_PASS) + for check in PHASE2_CHECKS: + if check != "http-support": + _write(tmp_path, check, STATUS_NOT_EVALUATED) + monkeypatch.setattr("sys.argv", ["run_phase2_gates", "--reports-dir", str(tmp_path), "--mode", "block"]) + assert run_phase2_gates.main() == 0 + summary = json.loads((tmp_path / "phase2-summary.json").read_text()) + assert summary["passed"] is True + assert summary["counts"]["not_evaluated"] == len(PHASE2_CHECKS) - 1 + + +def test_real_fail_exits_nonzero_in_block(tmp_path, monkeypatch): + for check in PHASE2_CHECKS: + _write(tmp_path, check, STATUS_PASS) + _write(tmp_path, "protocol-compliance", STATUS_FAIL, [{"severity": "high", "message": "m", "rule_id": "r"}]) + monkeypatch.setattr("sys.argv", ["run_phase2_gates", "--reports-dir", str(tmp_path), "--mode", "block"]) + assert run_phase2_gates.main() == 1 + + +# --------------------------------------------------------------------------- +# Integration (real server) +# --------------------------------------------------------------------------- + +_SERVER_URL = os.environ.get("MCP_TEST_SERVER_URL") + + +@pytest.mark.skipif(not _SERVER_URL, reason="set MCP_TEST_SERVER_URL to a running MCP server") +def test_probe_against_live_server(tmp_path): + # Smoke against a real server, e.g. generic-mock-mcp-server run with + # --transport streamable-http on :8080 (endpoint http://localhost:8080/mcp). + outcomes = phase2_probe.probe_all(_SERVER_URL, timeout=10.0) + by_check = {o.check: o for o in outcomes} + assert by_check["http-support"].status == STATUS_PASS + assert by_check["protocol-compliance"].status in (STATUS_PASS, STATUS_FAIL)