From 66d00c6dcbc1a135c6bdd6230ce8ce99819c392d Mon Sep 17 00:00:00 2001 From: operetz Date: Wed, 30 Sep 2026 13:39:05 +0300 Subject: [PATCH 1/9] add Phase 1 static evaluation --- abevalflow/mcp/__init__.py | 12 ++ abevalflow/mcp/phase1/__init__.py | 67 +++++++++++ abevalflow/mcp/phase1/license.py | 46 ++++++++ abevalflow/mcp/phase1/no_user_code.py | 36 ++++++ abevalflow/mcp/phase1/secrets.py | 34 ++++++ pipeline/tasks/phases/mcp-phase1.yaml | 157 ++++++++++++++++++++++++++ scripts/mcp/__init__.py | 6 + scripts/mcp/_common.py | 90 +++++++++++++++ scripts/mcp/license_scan.py | 122 ++++++++++++++++++++ scripts/mcp/no_user_code_scan.py | 103 +++++++++++++++++ scripts/mcp/rules/no_user_code.yml | 81 +++++++++++++ scripts/mcp/run_phase1_gates.py | 95 ++++++++++++++++ scripts/mcp/secrets_scan.py | 107 ++++++++++++++++++ 13 files changed, 956 insertions(+) create mode 100644 abevalflow/mcp/__init__.py create mode 100644 abevalflow/mcp/phase1/__init__.py create mode 100644 abevalflow/mcp/phase1/license.py create mode 100644 abevalflow/mcp/phase1/no_user_code.py create mode 100644 abevalflow/mcp/phase1/secrets.py create mode 100644 pipeline/tasks/phases/mcp-phase1.yaml create mode 100644 scripts/mcp/__init__.py create mode 100644 scripts/mcp/_common.py create mode 100644 scripts/mcp/license_scan.py create mode 100644 scripts/mcp/no_user_code_scan.py create mode 100644 scripts/mcp/rules/no_user_code.yml create mode 100644 scripts/mcp/run_phase1_gates.py create mode 100644 scripts/mcp/secrets_scan.py diff --git a/abevalflow/mcp/__init__.py b/abevalflow/mcp/__init__.py new file mode 100644 index 0000000..540afb8 --- /dev/null +++ b/abevalflow/mcp/__init__.py @@ -0,0 +1,12 @@ +"""MCP server evaluation pipeline. + +Isolated from the skill evaluation pipeline (see ADR: Evaluation Strategy for +MCP Servers as Standalone Software Components, Approach 4). Evaluates an MCP +server as a black box across three sequenced phases: + +- Phase 1 - static / build-time analysis (no running server). +- Phase 2 - deterministic contract/conformance testing (live, no LLM). +- Phase 3 - behavioral testing (mcpchecker, conditional on a task suite). + +Only Phase 1 is implemented here so far. +""" diff --git a/abevalflow/mcp/phase1/__init__.py b/abevalflow/mcp/phase1/__init__.py new file mode 100644 index 0000000..9823ca2 --- /dev/null +++ b/abevalflow/mcp/phase1/__init__.py @@ -0,0 +1,67 @@ +"""MCP Phase 1 - static / build-time analysis. + +Runs three static checks against the MCP server's built artifact/source, before +any deployment and without running the server (ADR Approach 4, Phase 1): + +- Secrets management (SecretsGate) +- No user-provided code execution (NoUserCodeGate) +- License compliance (LicenseGate) + +Each gate subclasses SecurityGate and reads a normalized ``{"findings": [...]}`` +scan JSON emitted by the corresponding scanner in ``scripts/mcp/``. Phase 1 is +kept isolated from the skill pipeline's global security-gate registry +(``abevalflow.gates.security``): these gates are instantiated only via this +module's :func:`run_phase1`, so onboarding an MCP server does not pull skill +scanners and vice versa. + +Phase 1 always executes, independent of runtime outcomes, so its checks resolve +to pass/fail only (never the not-evaluated state used for Phase 2/3). +""" + +from __future__ import annotations + +from pathlib import Path + +from abevalflow.gates.base import GateResult +from abevalflow.gates.security.base import SecurityGate +from abevalflow.mcp.phase1.license import DEFAULT_ALLOWED_LICENSES, LicenseGate +from abevalflow.mcp.phase1.no_user_code import NoUserCodeGate +from abevalflow.mcp.phase1.secrets import SecretsGate +from abevalflow.schemas import GatePolicy + +# Ordered list of the Phase 1 gate classes, in execution order. +PHASE1_GATES: tuple[type[SecurityGate], ...] = ( + SecretsGate, + NoUserCodeGate, + LicenseGate, +) + + +def get_phase1_gates() -> list[SecurityGate]: + """Instantiate all Phase 1 gates, in order.""" + return [gate_cls() for gate_cls in PHASE1_GATES] + + +def run_phase1(reports_dir: Path, policy: GatePolicy) -> list[GateResult]: + """Evaluate all Phase 1 gates against the scan reports. + + Args: + reports_dir: Path to ``reports/{submission-name}/`` holding the + normalized scan JSON files written by ``scripts/mcp/*``. + policy: Gate policy controlling mode (disabled/warn/block) per gate. + + Returns: + One GateResult per Phase 1 gate, in :data:`PHASE1_GATES` order. + """ + return [gate.evaluate(reports_dir, policy) for gate in get_phase1_gates()] + + +__all__ = [ + "DEFAULT_ALLOWED_LICENSES", + "PHASE1_GATES", + "LicenseGate", + "NoUserCodeGate", + "SecretsGate", + "get_phase1_gates", + "run_phase1", +] diff --git a/abevalflow/mcp/phase1/license.py b/abevalflow/mcp/phase1/license.py new file mode 100644 index 0000000..2c70171 --- /dev/null +++ b/abevalflow/mcp/phase1/license.py @@ -0,0 +1,46 @@ +"""Phase 1 license-compliance gate. + +Reads license-scan.json produced by scripts/mcp/license_scan.py (a licensee run +compared against the approved-license allow-list) and converts it into a +standardized GateResult. Fails in block mode when the top-level declared license +is missing or not in the allow-list (emitted as a HIGH finding by the scanner). + +ADR Phase 1: "License compliance - verify the MCP server's top-level declared +license against a list of approved/supported licenses ... Scoped to the +top-level declared license only; transitive dependency license scanning is out +of scope." +""" + +from __future__ import annotations + +from pathlib import Path + +from abevalflow.gates.base import GateResult +from abevalflow.gates.security.base import SecurityGate +from abevalflow.observability.decorators import timed_gate +from abevalflow.schemas import GatePolicy + +# Approved SPDX license identifiers for the top-level declared license. +# Compared case-insensitively (see license_scan.py::evaluate_licenses). +DEFAULT_ALLOWED_LICENSES: tuple[str, ...] = ( + "apache-2.0", + "mit", + "bsd-2-clause", + "bsd-3-clause", +) + + +class LicenseGate(SecurityGate): + """Top-level declared-license allow-list gate (licensee-backed).""" + + name = "mcp-license" + scan_filename = "license-scan.json" + + @timed_gate + def evaluate( + self, + reports_dir: Path, + policy: GatePolicy, + ) -> GateResult: + """Evaluate the license allow-list scan.""" + return self.evaluate_scan_json(reports_dir, policy) diff --git a/abevalflow/mcp/phase1/no_user_code.py b/abevalflow/mcp/phase1/no_user_code.py new file mode 100644 index 0000000..3b9199c --- /dev/null +++ b/abevalflow/mcp/phase1/no_user_code.py @@ -0,0 +1,36 @@ +"""Phase 1 no-user-code-execution gate. + +Reads no-user-code-scan.json produced by scripts/mcp/no_user_code_scan.py (a +normalized semgrep run) and converts it into a standardized GateResult. Fails +in block mode when any HIGH/CRITICAL dynamic/arbitrary-execution pattern is +present (e.g. eval/exec, os.system, subprocess shell=True, Function(), +Runtime.exec). + +ADR Phase 1: "Does not execute user-provided code - static analysis of the +artifact for dynamic/arbitrary code execution." +""" + +from __future__ import annotations + +from pathlib import Path + +from abevalflow.gates.base import GateResult +from abevalflow.gates.security.base import SecurityGate +from abevalflow.observability.decorators import timed_gate +from abevalflow.schemas import GatePolicy + + +class NoUserCodeGate(SecurityGate): + """Static dynamic-execution pattern scan gate (semgrep-backed).""" + + name = "mcp-no-user-code" + scan_filename = "no-user-code-scan.json" + + @timed_gate + def evaluate( + self, + reports_dir: Path, + policy: GatePolicy, + ) -> GateResult: + """Evaluate the normalized semgrep dynamic-execution scan.""" + return self.evaluate_scan_json(reports_dir, policy) diff --git a/abevalflow/mcp/phase1/secrets.py b/abevalflow/mcp/phase1/secrets.py new file mode 100644 index 0000000..4bde3a4 --- /dev/null +++ b/abevalflow/mcp/phase1/secrets.py @@ -0,0 +1,34 @@ +"""Phase 1 secrets-management gate. + +Reads secrets-scan.json produced by scripts/mcp/secrets_scan.py (a normalized +gitleaks run) and converts it into a standardized GateResult. Fails in block +mode when any HIGH/CRITICAL hardcoded-credential finding is present. + +ADR Phase 1: "Secrets management - static scan of the artifact for hardcoded +credentials." +""" + +from __future__ import annotations + +from pathlib import Path + +from abevalflow.gates.base import GateResult +from abevalflow.gates.security.base import SecurityGate +from abevalflow.observability.decorators import timed_gate +from abevalflow.schemas import GatePolicy + + +class SecretsGate(SecurityGate): + """Static hardcoded-credential scan gate (gitleaks-backed).""" + + name = "mcp-secrets" + scan_filename = "secrets-scan.json" + + @timed_gate + def evaluate( + self, + reports_dir: Path, + policy: GatePolicy, + ) -> GateResult: + """Evaluate the normalized gitleaks secrets scan.""" + return self.evaluate_scan_json(reports_dir, policy) diff --git a/pipeline/tasks/phases/mcp-phase1.yaml b/pipeline/tasks/phases/mcp-phase1.yaml new file mode 100644 index 0000000..cfc7192 --- /dev/null +++ b/pipeline/tasks/phases/mcp-phase1.yaml @@ -0,0 +1,157 @@ +apiVersion: tekton.dev/v1 +kind: Task +metadata: + name: mcp-phase1-static + namespace: ab-eval-flow +spec: + description: >- + MCP evaluation Phase 1 - static / build-time analysis (ADR Approach 4). + Runs three language-agnostic static checks against the MCP server source, + before deployment and without running the server: secrets management + (gitleaks), no user-provided code execution (semgrep), and license + compliance (licensee). Always executes, independent of runtime outcomes; + resolves to pass/fail only. Normalized scan JSON and per-gate GateResults + are written under reports//. + params: + - name: source-subdir + type: string + default: "." + description: >- + Path of the MCP server source within the workspace to scan + (relative to the workspace root). Defaults to the workspace root. + - name: submission-name + type: string + description: Name used for the report directory (reports//). + - name: mode + type: string + default: "block" + description: Gate enforcement mode - block, warn, or disabled. + - name: allowed-licenses + type: string + default: "" + description: >- + Space-separated approved SPDX ids overriding the built-in allow-list + (e.g. "apache-2.0 mit"). Empty uses DEFAULT_ALLOWED_LICENSES. + - name: pipeline-repo-url + type: string + default: "https://github.com/RHEcosystemAppEng/agentic_eval_flow.git" + description: URL of the Agentic Eval Flow pipeline repository (scanner scripts). + - name: pipeline-repo-revision + type: string + default: "main" + description: Branch or SHA of the pipeline repo to use for scripts. + workspaces: + - name: source + description: Workspace containing the MCP server source to evaluate. + results: + - name: phase1-passed + description: Whether all Phase 1 gates passed (true/false). + - name: secrets-passed + description: Whether the secrets-management gate passed. + - name: no-user-code-passed + description: Whether the no-user-code-execution gate passed. + - name: license-passed + description: Whether the license-compliance gate passed. + - name: report-path + description: Workspace-relative path to the Phase 1 reports directory. + steps: + - name: clone-pipeline-repo + image: registry.access.redhat.com/ubi9/python-311:9.6 + script: | + #!/usr/bin/env bash + set -euo pipefail + PIPELINE_DIR="$(workspaces.source.path)/_pipeline" + if [ -d "$PIPELINE_DIR/.git" ]; then + echo "Pipeline repo already cloned, skipping" + exit 0 + fi + git clone --depth 1 --branch "$(params.pipeline-repo-revision)" \ + "$(params.pipeline-repo-url)" "$PIPELINE_DIR" + echo "Cloned pipeline repo at $(params.pipeline-repo-revision)" + + - name: scan + image: registry.access.redhat.com/ubi9/python-311:9.6 + # Tekton steps run as separate containers sharing only the workspaces, not + # the container root filesystem. Tools must be installed in the SAME step + # that runs the scanners, so install and scan live together here. + env: + - name: SEMGREP_ENABLE_VERSION_CHECK + value: "0" # no phone-home version check; stay offline + script: | + #!/usr/bin/env bash + set -euo pipefail + PIPELINE_DIR="$(workspaces.source.path)/_pipeline" + SOURCE_DIR="$(workspaces.source.path)/$(params.source-subdir)" + REPORTS_DIR="$(workspaces.source.path)/reports/$(params.submission-name)" + mkdir -p "$REPORTS_DIR" + + echo "=== Installing gitleaks (secrets) ===" + GITLEAKS_VERSION="8.30.0" + curl -sL \ + "https://github.com/gitleaks/gitleaks/releases/download/v${GITLEAKS_VERSION}/gitleaks_${GITLEAKS_VERSION}_linux_x64.tar.gz" \ + | tar -xz -C /usr/local/bin gitleaks + gitleaks version + + echo "=== Installing semgrep (no-user-code) ===" + pip install --quiet --no-cache-dir semgrep + semgrep --version + + echo "=== Installing licensee (license) ===" + dnf install -y --quiet ruby ruby-devel gcc make redhat-rpm-config 2>&1 | tail -3 + gem install --quiet licensee + licensee version + + cd "$PIPELINE_DIR" + export PYTHONPATH="$PIPELINE_DIR" + pip install --quiet --no-cache-dir pydantic pyyaml + + LICENSE_ARGS="" + for spdx in $(params.allowed-licenses); do + LICENSE_ARGS="$LICENSE_ARGS --allow $spdx" + done + + echo "=== Phase 1 scanners on $SOURCE_DIR ===" + python -m scripts.mcp.secrets_scan "$SOURCE_DIR" --reports-dir "$REPORTS_DIR" + python -m scripts.mcp.no_user_code_scan "$SOURCE_DIR" --reports-dir "$REPORTS_DIR" + # shellcheck disable=SC2086 + python -m scripts.mcp.license_scan "$SOURCE_DIR" --reports-dir "$REPORTS_DIR" $LICENSE_ARGS + + - name: gate + image: registry.access.redhat.com/ubi9/python-311:9.6 + script: | + #!/usr/bin/env bash + set -euo pipefail + PIPELINE_DIR="$(workspaces.source.path)/_pipeline" + REPORTS_DIR="$(workspaces.source.path)/reports/$(params.submission-name)" + REPORT_REL="reports/$(params.submission-name)" + + cd "$PIPELINE_DIR" + export PYTHONPATH="$PIPELINE_DIR" + pip install --quiet --no-cache-dir pydantic pyyaml + + # Never abort before writing results: capture the exit code, emit + # Tekton results from the summary, then fail the step if a gate failed. + set +e + python -m scripts.mcp.run_phase1_gates \ + --reports-dir "$REPORTS_DIR" --mode "$(params.mode)" + GATE_EXIT=$? + set -e + + SUMMARY="$REPORTS_DIR/phase1-summary.json" + emit() { + python3 -c " + import json, sys + data = json.load(open('$SUMMARY')) + gates = {g['gate']: g['passed'] for g in data['gates']} + print(str(gates.get(sys.argv[1], False)).lower(), end='') + " "$1" + } + + python3 -c "import json;print(str(json.load(open('$SUMMARY'))['passed']).lower(),end='')" \ + > "$(results.phase1-passed.path)" + emit mcp-secrets > "$(results.secrets-passed.path)" + emit mcp-no-user-code > "$(results.no-user-code-passed.path)" + emit mcp-license > "$(results.license-passed.path)" + echo -n "$REPORT_REL" > "$(results.report-path.path)" + + exit $GATE_EXIT diff --git a/scripts/mcp/__init__.py b/scripts/mcp/__init__.py new file mode 100644 index 0000000..70540c6 --- /dev/null +++ b/scripts/mcp/__init__.py @@ -0,0 +1,6 @@ +"""Scanner scripts for the MCP Phase 1 static / build-time checks. + +Each module runs an external, language-agnostic static-analysis tool against an +MCP server repository and normalizes its output into the shared +``{"findings": [...]}`` schema that ``abevalflow.mcp.phase1`` gates read. +""" diff --git a/scripts/mcp/_common.py b/scripts/mcp/_common.py new file mode 100644 index 0000000..424380a --- /dev/null +++ b/scripts/mcp/_common.py @@ -0,0 +1,90 @@ +"""Shared helpers for MCP Phase 1 scanner scripts. + +All Phase 1 scanners emit the same JSON shape so the Phase 1 gates (which reuse +``SecurityGate.evaluate_scan_json``) can read them uniformly:: + + { + "scanner": "", + "target": "", + "findings": [ + {"severity": "high", "message": "...", "rule_id": "...", "file_path": "..."}, + ... + ] + } + +``severity`` must be one of the values understood by ``abevalflow.gates.base``: +critical / high / medium / low / info. +""" + +from __future__ import annotations + +import json +import logging +import subprocess +from pathlib import Path +from typing import Any + +logger = logging.getLogger(__name__) + +VALID_SEVERITIES = ("critical", "high", "medium", "low", "info") + + +def normalize_severity(value: str | None, default: str = "info") -> str: + """Coerce an arbitrary severity string into a known severity level.""" + if not value: + return default + sev = value.strip().lower() + return sev if sev in VALID_SEVERITIES else default + + +def make_finding( + *, + severity: str, + message: str, + rule_id: str, + file_path: str | None = None, + **extra: Any, +) -> dict[str, Any]: + """Build a single normalized finding dict.""" + finding: dict[str, Any] = { + "severity": normalize_severity(severity), + "message": message, + "rule_id": rule_id, + "file_path": file_path, + } + finding.update(extra) + return finding + + +def write_scan( + scan_path: Path, + scanner: str, + target: Path, + findings: list[dict[str, Any]], +) -> None: + """Write a normalized scan report to ``scan_path``.""" + scan_path.parent.mkdir(parents=True, exist_ok=True) + payload = { + "scanner": scanner, + "target": str(target), + "findings": findings, + } + scan_path.write_text(json.dumps(payload, indent=2)) + logger.info("Wrote %d findings to %s", len(findings), scan_path) + + +def run_tool(cmd: list[str], *, cwd: Path | None = None) -> subprocess.CompletedProcess[str]: + """Run a scanner CLI, capturing stdout/stderr as text. + + The return code is NOT checked here - most scanners exit non-zero when they + find issues, which is expected. Callers inspect ``returncode`` explicitly to + distinguish "findings present" from "tool failed to run". + """ + logger.info("Running: %s", " ".join(cmd)) + return subprocess.run( # noqa: S603 - cmd is built from trusted, fixed tool args + cmd, + cwd=str(cwd) if cwd else None, + capture_output=True, + text=True, + check=False, + ) diff --git a/scripts/mcp/license_scan.py b/scripts/mcp/license_scan.py new file mode 100644 index 0000000..8b7a55f --- /dev/null +++ b/scripts/mcp/license_scan.py @@ -0,0 +1,122 @@ +#!/usr/bin/env python3 +"""MCP Phase 1 license-compliance scan. + +Runs licensee to detect the repository's TOP-LEVEL declared license (from the +LICENSE file and/or package manifests) and checks it against an approved-license +allow-list. Emits license-scan.json for LicenseGate. Transitive/dependency +license scanning is out of scope (ADR Phase 1). + +Usage: + python -m scripts.mcp.license_scan --reports-dir \ + [--allow apache-2.0 --allow mit ...] +""" + +from __future__ import annotations + +import argparse +import json +import logging +import sys +from pathlib import Path + +from abevalflow.mcp.phase1.license import DEFAULT_ALLOWED_LICENSES +from scripts.mcp._common import make_finding, run_tool, write_scan + +logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") +logger = logging.getLogger(__name__) + +SCAN_FILENAME = "license-scan.json" +SCANNER = "licensee" + + +def evaluate_licenses( + detected: list[dict], + allowed: set[str], + matched_files: list[dict], +) -> list[dict]: + """Produce findings for a missing or disallowed top-level license. + + A repo passes when at least one detected license is in the allow-list. + licensee emits canonical SPDX ids (e.g. "Apache-2.0"); compare + case-insensitively by lowercasing both sides. + """ + location = matched_files[0].get("filename") if matched_files else None + + spdx_ids = [ + str(lic.get("spdx_id", "")).lower() + for lic in detected + if lic.get("spdx_id") and str(lic.get("spdx_id")).lower() != "noassertion" + ] + + if not spdx_ids: + return [ + make_finding( + severity="high", + message="No top-level declared license could be identified.", + rule_id="license-missing", + file_path=location, + ) + ] + + if any(spdx in allowed for spdx in spdx_ids): + return [] + + return [ + make_finding( + severity="high", + message=(f"Declared license(s) {', '.join(spdx_ids)} not in allow-list ({', '.join(sorted(allowed))})."), + rule_id="license-not-allowed", + file_path=location, + detected=spdx_ids, + ) + ] + + +def main() -> int: + parser = argparse.ArgumentParser(description="MCP Phase 1 license scan (licensee)") + parser.add_argument("target_dir", type=Path, help="MCP server repo to scan") + parser.add_argument( + "--reports-dir", + type=Path, + required=True, + help="Directory to write license-scan.json into", + ) + parser.add_argument( + "--allow", + action="append", + default=None, + metavar="SPDX_ID", + help="Approved SPDX id (repeatable). Defaults to DEFAULT_ALLOWED_LICENSES.", + ) + args = parser.parse_args() + + if not args.target_dir.is_dir(): + logger.error("Not a directory: %s", args.target_dir) + return 1 + + allowed = {a.lower() for a in (args.allow or DEFAULT_ALLOWED_LICENSES)} + scan_path = args.reports_dir / SCAN_FILENAME + + result = run_tool(["licensee", "detect", "--json", str(args.target_dir)]) + if result.returncode != 0: + logger.error("licensee failed (exit %d): %s", result.returncode, result.stderr.strip()) + return 1 + + try: + data = json.loads(result.stdout or "{}") + except json.JSONDecodeError as exc: + logger.error("Could not parse licensee JSON: %s", exc) + return 1 + + findings = evaluate_licenses( + data.get("licenses", []), + allowed, + data.get("matched_files", []), + ) + write_scan(scan_path, SCANNER, args.target_dir, findings) + logger.info("License scan complete: %d finding(s)", len(findings)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/mcp/no_user_code_scan.py b/scripts/mcp/no_user_code_scan.py new file mode 100644 index 0000000..1eb1afc --- /dev/null +++ b/scripts/mcp/no_user_code_scan.py @@ -0,0 +1,103 @@ +#!/usr/bin/env python3 +"""MCP Phase 1 "does not execute user-provided code" scan. + +Runs semgrep with the bundled offline ruleset (rules/no_user_code.yml) over an +MCP server repository and normalizes results into no-user-code-scan.json for +NoUserCodeGate. + +Usage: + python -m scripts.mcp.no_user_code_scan --reports-dir +""" + +from __future__ import annotations + +import argparse +import json +import logging +import sys +from pathlib import Path + +from scripts.mcp._common import make_finding, run_tool, write_scan + +logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") +logger = logging.getLogger(__name__) + +SCAN_FILENAME = "no-user-code-scan.json" +SCANNER = "semgrep" +RULES_PATH = Path(__file__).parent / "rules" / "no_user_code.yml" + +# semgrep severity -> our severity. +_SEVERITY_MAP = {"ERROR": "high", "WARNING": "medium", "INFO": "low"} + + +def normalize(semgrep_results: list[dict]) -> list[dict]: + """Map semgrep JSON results to normalized findings.""" + findings: list[dict] = [] + for res in semgrep_results: + extra = res.get("extra", {}) + sev = _SEVERITY_MAP.get(str(extra.get("severity", "")).upper(), "medium") + findings.append( + make_finding( + severity=sev, + message=extra.get("message", "Dynamic/arbitrary code execution pattern"), + rule_id=res.get("check_id", "unknown"), + file_path=res.get("path"), + line=res.get("start", {}).get("line"), + ) + ) + return findings + + +def main() -> int: + parser = argparse.ArgumentParser(description="MCP Phase 1 no-user-code-execution scan (semgrep)") + parser.add_argument("target_dir", type=Path, help="MCP server repo to scan") + parser.add_argument( + "--reports-dir", + type=Path, + required=True, + help="Directory to write no-user-code-scan.json into", + ) + args = parser.parse_args() + + if not args.target_dir.is_dir(): + logger.error("Not a directory: %s", args.target_dir) + return 1 + if not RULES_PATH.is_file(): + logger.error("Ruleset missing: %s", RULES_PATH) + return 1 + + scan_path = args.reports_dir / SCAN_FILENAME + + result = run_tool( + [ + "semgrep", + "scan", + "--config", + str(RULES_PATH), + "--json", + "--quiet", + "--metrics=off", # stay fully offline (no telemetry call) + "--no-git-ignore", # scan every file in the artifact, not just tracked + str(args.target_dir), + ] + ) + + # semgrep exits 0 on success (with or without findings); non-zero = tool error. + if result.returncode != 0: + logger.error("semgrep failed (exit %d): %s", result.returncode, result.stderr.strip()) + return 1 + + try: + data = json.loads(result.stdout or "{}") + except json.JSONDecodeError as exc: + logger.error("Could not parse semgrep JSON: %s", exc) + return 1 + + findings = normalize(data.get("results", [])) + write_scan(scan_path, SCANNER, args.target_dir, findings) + logger.info("No-user-code scan complete: %d finding(s)", len(findings)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/mcp/rules/no_user_code.yml b/scripts/mcp/rules/no_user_code.yml new file mode 100644 index 0000000..5e8e437 --- /dev/null +++ b/scripts/mcp/rules/no_user_code.yml @@ -0,0 +1,81 @@ +# Semgrep ruleset for MCP Phase 1 "does not execute user-provided code". +# +# Flags dynamic / arbitrary code-execution sinks across the languages MCP +# servers are commonly written in. Deliberately narrow (high-signal sinks only) +# to keep false positives low and to run fully offline (no --config p/registry +# network fetch). Extend per language as needed. +# +# Severity ERROR -> mapped to HIGH by scripts/mcp/no_user_code_scan.py. +rules: + # ---- Python ------------------------------------------------------------- + - id: python-dynamic-exec + languages: [python] + severity: ERROR + message: Dynamic code execution via eval/exec/compile. + patterns: + - pattern-either: + - pattern: eval(...) + - pattern: exec(...) + - pattern: compile(...) + + - id: python-os-system + languages: [python] + severity: ERROR + message: Shell command execution via os.system. + pattern: os.system(...) + + - id: python-subprocess-shell + languages: [python] + severity: ERROR + message: subprocess call with shell=True executes a shell command string. + patterns: + - pattern-either: + - pattern: subprocess.$FN(..., shell=True, ...) + - pattern: subprocess.Popen(..., shell=True, ...) + + # ---- JavaScript / TypeScript ------------------------------------------- + - id: js-dynamic-eval + languages: [javascript, typescript] + severity: ERROR + message: Dynamic code execution via eval or the Function constructor. + patterns: + - pattern-either: + - pattern: eval(...) + - pattern: new Function(...) + - pattern: Function(...) + + - id: js-child-process-exec + languages: [javascript, typescript] + severity: ERROR + message: Shell command execution via child_process.exec. + patterns: + - pattern-either: + - pattern: child_process.exec(...) + - pattern: $CP.exec(...) + - pattern: exec(...) + + # ---- Go ---------------------------------------------------------------- + - id: go-exec-command + languages: [go] + severity: ERROR + message: External command execution via os/exec. + patterns: + - pattern-either: + - pattern: exec.Command(...) + - pattern: exec.CommandContext(...) + + # ---- Java -------------------------------------------------------------- + - id: java-runtime-exec + languages: [java] + severity: ERROR + message: External command execution via Runtime.exec / ProcessBuilder. + patterns: + - pattern-either: + - pattern: (Runtime $R).exec(...) + - pattern: new ProcessBuilder(...) + + - id: java-script-engine + languages: [java] + severity: ERROR + message: Dynamic script evaluation via javax.script ScriptEngine. + pattern: (ScriptEngine $E).eval(...) diff --git a/scripts/mcp/run_phase1_gates.py b/scripts/mcp/run_phase1_gates.py new file mode 100644 index 0000000..7803110 --- /dev/null +++ b/scripts/mcp/run_phase1_gates.py @@ -0,0 +1,95 @@ +#!/usr/bin/env python3 +"""Run the MCP Phase 1 gates over the normalized scan reports. + +Reads the three scan JSON files produced by the Phase 1 scanners, evaluates each +gate, and writes the results back into the reports directory: + +- ``-result.json`` per gate (the full GateResult) +- ``phase1-summary.json`` (overall pass/fail + per-gate summary) + +This is the Phase 1 reporting stub. The real Evaluation-results-database writer +(three-state pass/fail/not-evaluated) is added by the pipeline layer later; Phase +1 always executes, so its checks resolve to pass/fail only. + +Usage: + python -m scripts.mcp.run_phase1_gates --reports-dir [--mode block|warn] +""" + +from __future__ import annotations + +import argparse +import json +import logging +import sys +from pathlib import Path + +from abevalflow.mcp.phase1 import run_phase1 +from abevalflow.schemas import GateMode, GatePolicy + +logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") +logger = logging.getLogger(__name__) + +SUMMARY_FILENAME = "phase1-summary.json" + + +def main() -> int: + parser = argparse.ArgumentParser(description="Run MCP Phase 1 gates") + parser.add_argument( + "--reports-dir", + type=Path, + required=True, + help="Directory holding the Phase 1 scan JSON files", + ) + parser.add_argument( + "--mode", + choices=[m.value for m in GateMode], + default=GateMode.BLOCK.value, + help="Gate enforcement mode (default: block)", + ) + args = parser.parse_args() + + if not args.reports_dir.is_dir(): + logger.error("Not a directory: %s", args.reports_dir) + return 1 + + policy = GatePolicy(default_mode=GateMode(args.mode)) + results = run_phase1(args.reports_dir, policy) + + summary: list[dict] = [] + for result in results: + gate_key = result.get_policy_key() + (args.reports_dir / f"{gate_key}-result.json").write_text(result.model_dump_json(indent=2)) + summary.append( + { + "gate": gate_key, + "passed": result.passed, + "score": result.score, + "findings": len(result.findings), + "message": result.message, + } + ) + logger.info( + "%s: passed=%s score=%.2f findings=%d", + gate_key, + result.passed, + result.score, + len(result.findings), + ) + + overall_passed = all(r.passed for r in results) + (args.reports_dir / SUMMARY_FILENAME).write_text( + json.dumps( + {"phase": "phase1", "passed": overall_passed, "gates": summary}, + indent=2, + ) + ) + + logger.info("Phase 1 overall: %s", "PASS" if overall_passed else "FAIL") + # Exit non-zero in block mode when any gate failed, so the Tekton step fails. + if not overall_passed and args.mode == GateMode.BLOCK.value: + return 1 + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/mcp/secrets_scan.py b/scripts/mcp/secrets_scan.py new file mode 100644 index 0000000..0d1e728 --- /dev/null +++ b/scripts/mcp/secrets_scan.py @@ -0,0 +1,107 @@ +#!/usr/bin/env python3 +"""MCP Phase 1 secrets scan. + +Runs gitleaks over an MCP server repository (static working-tree scan, no live +verification) and normalizes its output into secrets-scan.json for SecretsGate. + +Usage: + python -m scripts.mcp.secrets_scan --reports-dir +""" + +from __future__ import annotations + +import argparse +import json +import logging +import sys +import tempfile +from pathlib import Path + +from scripts.mcp._common import make_finding, run_tool, write_scan + +logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") +logger = logging.getLogger(__name__) + +SCAN_FILENAME = "secrets-scan.json" +SCANNER = "gitleaks" + +# gitleaks has no severity field. Every gitleaks hit is a candidate hardcoded +# credential, so all findings are HIGH: the gate blocks in block mode only on +# HIGH/CRITICAL, and a leaked secret must never pass Phase 1 just because it was +# caught by the broad generic rule rather than a vendor-specific one. + + +def normalize(gitleaks_findings: list[dict]) -> list[dict]: + """Map gitleaks JSON records to normalized findings.""" + findings: list[dict] = [] + for rec in gitleaks_findings: + rule_id = rec.get("RuleID", "unknown") + findings.append( + make_finding( + severity="high", + message=rec.get("Description", "Hardcoded secret detected"), + rule_id=rule_id, + file_path=rec.get("File"), + line=rec.get("StartLine"), + ) + ) + return findings + + +def main() -> int: + parser = argparse.ArgumentParser(description="MCP Phase 1 secrets scan (gitleaks)") + parser.add_argument("target_dir", type=Path, help="MCP server repo to scan") + parser.add_argument( + "--reports-dir", + type=Path, + required=True, + help="Directory to write secrets-scan.json into", + ) + args = parser.parse_args() + + if not args.target_dir.is_dir(): + logger.error("Not a directory: %s", args.target_dir) + return 1 + + scan_path = args.reports_dir / SCAN_FILENAME + + with tempfile.NamedTemporaryFile("r", suffix=".json", delete=False) as tmp: + report_path = Path(tmp.name) + + # gitleaks `dir` scans the working tree (no git history required). + result = run_tool( + [ + "gitleaks", + "dir", + str(args.target_dir), + "--report-format", + "json", + "--report-path", + str(report_path), + "--no-banner", + "--exit-code", + "0", # never fail the process on findings; the gate decides pass/fail + ] + ) + + # Exit codes: 0 clean/normal. Any output on a crash goes to stderr. + if result.returncode != 0: + logger.error("gitleaks failed (exit %d): %s", result.returncode, result.stderr.strip()) + return 1 + + try: + raw = json.loads(report_path.read_text() or "[]") + except (json.JSONDecodeError, OSError) as exc: + logger.error("Could not read gitleaks report: %s", exc) + return 1 + finally: + report_path.unlink(missing_ok=True) + + findings = normalize(raw if isinstance(raw, list) else []) + write_scan(scan_path, SCANNER, args.target_dir, findings) + logger.info("Secrets scan complete: %d finding(s)", len(findings)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) From b6b7d58657f1f46abdccd4a347357010d147e51d Mon Sep 17 00:00:00 2001 From: operetz Date: Wed, 30 Sep 2026 16:59:57 +0300 Subject: [PATCH 2/9] add Phase 2 deterministic conformance evaluation --- abevalflow/mcp/phase2/__init__.py | 61 +++++ abevalflow/mcp/phase2/base.py | 152 ++++++++++++ pipeline/tasks/phases/mcp-phase2.yaml | 129 ++++++++++ scripts/mcp/_phase2.py | 72 ++++++ scripts/mcp/compass_fetch.py | 160 ++++++++++++ scripts/mcp/mcp_client.py | 203 +++++++++++++++ scripts/mcp/phase2_probe.py | 339 ++++++++++++++++++++++++++ scripts/mcp/run_phase2_gates.py | 93 +++++++ 8 files changed, 1209 insertions(+) create mode 100644 abevalflow/mcp/phase2/__init__.py create mode 100644 abevalflow/mcp/phase2/base.py create mode 100644 pipeline/tasks/phases/mcp-phase2.yaml create mode 100644 scripts/mcp/_phase2.py create mode 100644 scripts/mcp/compass_fetch.py create mode 100644 scripts/mcp/mcp_client.py create mode 100644 scripts/mcp/phase2_probe.py create mode 100644 scripts/mcp/run_phase2_gates.py diff --git a/abevalflow/mcp/phase2/__init__.py b/abevalflow/mcp/phase2/__init__.py new file mode 100644 index 0000000..e4c0dd0 --- /dev/null +++ b/abevalflow/mcp/phase2/__init__.py @@ -0,0 +1,61 @@ +"""MCP evaluation Phase 2 - deterministic contract/conformance (live, no LLM). + +Phase 2 is black-box tested against the deployed, running MCP server (ADR +Approach 4). It is deterministic and uses NO LLM. Results follow the three-state +model (pass / fail / not_evaluated). + +Two sources feed the same set of gates: +* **Live probes** (scripts/mcp/phase2_probe.py) - checks that need a running + server: protocol compliance, schema conformance, tool annotations, HTTP + support, rate limiting, response-size limit, mandatory timeouts, OpenAPI + conformance. +* **Compass-consumed** (scripts/mcp/compass_fetch.py) - the Compass tier-1 + automated facts consumed rather than re-probed: tool-name rules and + OAuth-matches-catalog. Metadata Compass also owns (schema-present, + documentation, data-source approval) is left to Compass to report directly. + +Like Phase 1, this package is kept isolated from the skill pipeline: the gates +are instantiated only here via ``run_phase2`` and are never registered in the +global security-gate registry. +""" + +from __future__ import annotations + +from pathlib import Path + +from abevalflow.gates.base import GateResult +from abevalflow.mcp.phase2.base import Phase2Gate +from abevalflow.schemas import GatePolicy + +# Live probe checks (need a running server). +LIVE_CHECKS: tuple[str, ...] = ( + "http-support", + "protocol-compliance", + "schema-conformance", + "tool-annotations", + "rate-limiting", + "response-size-limit", + "mandatory-timeouts", + "openapi-conformance", +) + +# Checks consumed from existing Compass facts (not re-probed). +COMPASS_CHECKS: tuple[str, ...] = ( + "tool-name-rules", + "oauth-catalog-match", +) + +PHASE2_CHECKS: tuple[str, ...] = LIVE_CHECKS + COMPASS_CHECKS + + +def get_phase2_gates() -> list[Phase2Gate]: + """Instantiate one gate per Phase 2 check, in order.""" + return [Phase2Gate(check) for check in PHASE2_CHECKS] + + +def run_phase2(reports_dir: Path, policy: GatePolicy) -> list[GateResult]: + """Evaluate every Phase 2 gate against the check result files in ``reports_dir``.""" + return [gate.evaluate(reports_dir, policy) for gate in get_phase2_gates()] + + +__all__ = ["COMPASS_CHECKS", "LIVE_CHECKS", "PHASE2_CHECKS", "Phase2Gate", "get_phase2_gates", "run_phase2"] diff --git a/abevalflow/mcp/phase2/base.py b/abevalflow/mcp/phase2/base.py new file mode 100644 index 0000000..c3d798b --- /dev/null +++ b/abevalflow/mcp/phase2/base.py @@ -0,0 +1,152 @@ +"""Phase 2 gate: turn a three-state check result file into a GateResult. + +Unlike the Phase 1 security gates (one class per gate, reusing +``SecurityGate.evaluate_scan_json``), Phase 2 has many small, uniform checks that +differ only by name and by a three-state status, so a single data-driven gate is +used instead of a subclass per check. + +Each check result file (written by scripts/mcp/phase2_probe.py or compass_fetch.py) +has the shape:: + + {"check": ..., "status": "pass"|"fail"|"not_evaluated", "source": ..., + "reason": ..., "findings": [ ... ]} + +Status handling: +* ``pass`` -> passed, score 1.0 +* ``fail`` -> passed only in warn mode; score weighted by findings +* ``not_evaluated`` -> passed, score 1.0 (never penalize; ADR three-state) +* file missing -> not_evaluated (the check did not run) +""" + +from __future__ import annotations + +import json +import logging +from pathlib import Path + +from abevalflow.gates.base import Finding, GateMode, GateResult, GateType, Severity +from abevalflow.observability.decorators import timed_gate +from abevalflow.schemas import GatePolicy + +logger = logging.getLogger(__name__) + +STATUS_PASS = "pass" +STATUS_FAIL = "fail" +STATUS_NOT_EVALUATED = "not_evaluated" + +_SEVERITY_WEIGHT = { + Severity.CRITICAL: 0.0, + Severity.HIGH: 0.25, + Severity.MEDIUM: 0.5, + Severity.LOW: 0.75, + Severity.INFO: 0.9, +} + + +def _weighted_score(findings: list[Finding]) -> float: + if not findings: + return 1.0 + return sum(_SEVERITY_WEIGHT[f.severity] for f in findings) / len(findings) + + +class Phase2Gate: + """Data-driven gate for one Phase 2 check.""" + + def __init__(self, check: str) -> None: + self.check = check + self.result_filename = f"{check}-check.json" + + def _result( + self, + *, + passed: bool, + score: float, + mode: GateMode, + status: str, + reason: str, + source: str, + findings: list[Finding], + ) -> GateResult: + return GateResult( + gate_type=GateType.QUALITY, + gate_name="quality", + policy_key=self.check, + passed=passed, + score=score, + mode=mode, + findings=findings, + details={"check": self.check, "status": status, "source": source, "reason": reason}, + message=f"{self.check}: {status.upper()} - {reason}" if reason else f"{self.check}: {status.upper()}", + ) + + @timed_gate + def evaluate(self, reports_dir: Path, policy: GatePolicy) -> GateResult: + gate_policy = policy.get_gate_policy("quality") + mode = gate_policy.mode + path = reports_dir / self.result_filename + + if not path.is_file(): + return self._result( + passed=True, + score=1.0, + mode=mode, + status=STATUS_NOT_EVALUATED, + reason=f"{self.result_filename} not found (check did not run)", + source="none", + findings=[], + ) + + try: + data = json.loads(path.read_text()) + except (json.JSONDecodeError, OSError) as exc: + logger.error("Failed to read %s: %s", path, exc) + return self._result( + passed=False, + score=0.0, + mode=mode, + status=STATUS_FAIL, + reason=f"could not parse {self.result_filename}: {exc}", + source="none", + findings=[], + ) + + status = data.get("status", STATUS_NOT_EVALUATED) + reason = data.get("reason", "") + source = data.get("source", "probe") + findings = [ + Finding( + severity=_coerce_severity(f.get("severity")), + message=f.get("message", ""), + location=f.get("file_path") or f.get("location"), + rule_id=f.get("rule_id", "unknown"), + details=f, + ) + for f in data.get("findings", []) + ] + + if status == STATUS_PASS: + return self._result( + passed=True, score=1.0, mode=mode, status=status, reason=reason, source=source, findings=findings + ) + if status == STATUS_NOT_EVALUATED: + return self._result( + passed=True, score=1.0, mode=mode, status=status, reason=reason, source=source, findings=findings + ) + # fail + passed = mode != GateMode.BLOCK + return self._result( + passed=passed, + score=_weighted_score(findings), + mode=mode, + status=STATUS_FAIL, + reason=reason, + source=source, + findings=findings, + ) + + +def _coerce_severity(value: object) -> Severity: + try: + return Severity(str(value).lower().strip()) + except ValueError: + return Severity.INFO diff --git a/pipeline/tasks/phases/mcp-phase2.yaml b/pipeline/tasks/phases/mcp-phase2.yaml new file mode 100644 index 0000000..13e07f9 --- /dev/null +++ b/pipeline/tasks/phases/mcp-phase2.yaml @@ -0,0 +1,129 @@ +apiVersion: tekton.dev/v1 +kind: Task +metadata: + name: mcp-phase2-conformance + namespace: ab-eval-flow +spec: + description: >- + MCP evaluation Phase 2 - deterministic contract/conformance (ADR Approach 4). + Black-box, no LLM. Probes the deployed, running MCP server over Streamable + HTTP for protocol/schema/HTTP conformance and operational limits, and + consumes the checks Compass already collects as real facts (tool-name rules, + OAuth match). Uses the three-state model + (pass / fail / not_evaluated): an unreachable server or a check that cannot + produce a meaningful signal is reported not_evaluated, never failed. Assumes + the server was already deployed and became ready in a prior pipeline step + (reached at :8080 by convention). + params: + - name: server-url + type: string + description: MCP endpoint URL, e.g. http://:8080/mcp + - name: submission-name + type: string + description: Name used for the report directory (reports//). + - name: mode + type: string + default: "block" + description: Gate enforcement mode - block, warn, or disabled. + - name: timeout + type: string + default: "30" + description: Per-request HTTP timeout (seconds) for the probe. + - name: compass-facts-file + type: string + default: "" + description: >- + Workspace-relative path to a Compass facts JSON for the consumed checks. + Empty skips the Compass-consume step; those checks then report + not_evaluated. (Facts file only; live SoundCheck retrieval is not wired yet.) + - name: pipeline-repo-url + type: string + default: "https://github.com/RHEcosystemAppEng/agentic_eval_flow.git" + description: URL of the Agentic Eval Flow pipeline repository (probe scripts). + - name: pipeline-repo-revision + type: string + default: "main" + description: Branch or SHA of the pipeline repo to use for scripts. + workspaces: + - name: source + description: Workspace for the pipeline checkout and the reports directory. + results: + - name: phase2-passed + description: Whether Phase 2 passed (no check FAILED in block mode). + - name: not-evaluated-count + description: Number of Phase 2 checks reported not_evaluated. + - name: report-path + description: Workspace-relative path to the Phase 2 reports directory. + steps: + - name: clone-pipeline-repo + image: registry.access.redhat.com/ubi9/python-311:9.6 + script: | + #!/usr/bin/env bash + set -euo pipefail + PIPELINE_DIR="$(workspaces.source.path)/_pipeline" + if [ -d "$PIPELINE_DIR/.git" ]; then + echo "Pipeline repo already cloned, skipping" + exit 0 + fi + git clone --depth 1 --branch "$(params.pipeline-repo-revision)" \ + "$(params.pipeline-repo-url)" "$PIPELINE_DIR" + echo "Cloned pipeline repo at $(params.pipeline-repo-revision)" + + - name: probe + image: registry.access.redhat.com/ubi9/python-311:9.6 + script: | + #!/usr/bin/env bash + set -euo pipefail + PIPELINE_DIR="$(workspaces.source.path)/_pipeline" + REPORTS_DIR="$(workspaces.source.path)/reports/$(params.submission-name)" + mkdir -p "$REPORTS_DIR" + + cd "$PIPELINE_DIR" + export PYTHONPATH="$PIPELINE_DIR" + pip install --quiet --no-cache-dir httpx pydantic pyyaml + + echo "=== Phase 2 live probe against $(params.server-url) ===" + python -m scripts.mcp.phase2_probe \ + --server-url "$(params.server-url)" \ + --reports-dir "$REPORTS_DIR" \ + --timeout "$(params.timeout)" + + FACTS="$(params.compass-facts-file)" + if [ -n "$FACTS" ]; then + echo "=== Phase 2 Compass consume from $FACTS ===" + python -m scripts.mcp.compass_fetch \ + --facts-file "$(workspaces.source.path)/$FACTS" \ + --reports-dir "$REPORTS_DIR" + else + echo "No compass-facts-file given; Compass-consumed checks will report not_evaluated." + fi + + - name: gate + image: registry.access.redhat.com/ubi9/python-311:9.6 + script: | + #!/usr/bin/env bash + set -euo pipefail + PIPELINE_DIR="$(workspaces.source.path)/_pipeline" + REPORTS_DIR="$(workspaces.source.path)/reports/$(params.submission-name)" + REPORT_REL="reports/$(params.submission-name)" + + cd "$PIPELINE_DIR" + export PYTHONPATH="$PIPELINE_DIR" + pip install --quiet --no-cache-dir httpx pydantic pyyaml + + # Never abort before writing results: capture the exit code, emit the + # Tekton results from the summary, then fail the step if a check failed. + set +e + python -m scripts.mcp.run_phase2_gates \ + --reports-dir "$REPORTS_DIR" --mode "$(params.mode)" + GATE_EXIT=$? + set -e + + SUMMARY="$REPORTS_DIR/phase2-summary.json" + python3 -c "import json;print(str(json.load(open('$SUMMARY'))['passed']).lower(),end='')" \ + > "$(results.phase2-passed.path)" + python3 -c "import json;print(json.load(open('$SUMMARY'))['counts']['not_evaluated'],end='')" \ + > "$(results.not-evaluated-count.path)" + echo -n "$REPORT_REL" > "$(results.report-path.path)" + + exit $GATE_EXIT diff --git a/scripts/mcp/_phase2.py b/scripts/mcp/_phase2.py new file mode 100644 index 0000000..6e0441e --- /dev/null +++ b/scripts/mcp/_phase2.py @@ -0,0 +1,72 @@ +"""Shared helpers for MCP Phase 2 (deterministic contract/conformance). + +Phase 2 uses a THREE-STATE model (pass / fail / not_evaluated) - unlike Phase 1, +which is always pass/fail. ``not_evaluated`` is used when a check cannot produce a +meaningful signal (server unreachable, no OpenAPI contract, a capability the +server does not expose), so a server is never penalized for something it was not +actually subjected to (ADR: three-state reporting). + +Every Phase 2 check - whether a live probe or a value consumed from Compass - +writes the same JSON shape so the Phase 2 gates can read them uniformly:: + + { + "check": "protocol-compliance", + "status": "pass" | "fail" | "not_evaluated", + "source": "probe" | "compass", + "reason": "human-readable summary", + "findings": [ {"severity": ..., "message": ..., "rule_id": ...}, ... ] + } +""" + +from __future__ import annotations + +import json +import logging +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any + +logger = logging.getLogger(__name__) + +STATUS_PASS = "pass" +STATUS_FAIL = "fail" +STATUS_NOT_EVALUATED = "not_evaluated" +VALID_STATUSES = (STATUS_PASS, STATUS_FAIL, STATUS_NOT_EVALUATED) + + +@dataclass +class CheckOutcome: + """Result of a single Phase 2 check.""" + + check: str + status: str + reason: str = "" + source: str = "probe" + findings: list[dict[str, Any]] = field(default_factory=list) + + def __post_init__(self) -> None: + if self.status not in VALID_STATUSES: + raise ValueError(f"invalid status {self.status!r}; expected one of {VALID_STATUSES}") + + def to_dict(self) -> dict[str, Any]: + return { + "check": self.check, + "status": self.status, + "source": self.source, + "reason": self.reason, + "findings": self.findings, + } + + +def result_filename(check: str) -> str: + """Filename a check's result is written to (and read back by its gate).""" + return f"{check}-check.json" + + +def write_check_result(reports_dir: Path, outcome: CheckOutcome) -> Path: + """Write one check outcome to ``/-check.json``.""" + reports_dir.mkdir(parents=True, exist_ok=True) + path = reports_dir / result_filename(outcome.check) + path.write_text(json.dumps(outcome.to_dict(), indent=2)) + logger.info("Wrote %s check result (%s) to %s", outcome.check, outcome.status, path) + return path diff --git a/scripts/mcp/compass_fetch.py b/scripts/mcp/compass_fetch.py new file mode 100644 index 0000000..094c098 --- /dev/null +++ b/scripts/mcp/compass_fetch.py @@ -0,0 +1,160 @@ +#!/usr/bin/env python3 +"""MCP Phase 2 Compass-consumed checks. + +A few Phase 2 checks are already collected by Compass as real automated facts, so +the pipeline consumes them rather than re-probing (avoids duplicating what +Compass does well). This script maps those Compass SoundCheck facts into the same +three-state per-check result files the Phase 2 gates read. + +Consumed checks and their source facts: + tool-name-rules <- mcp:default/tools $.allToolNamesValid + oauth-catalog-match<- mcp:default/security $.oauth.enforced + $.entityAuth.authServerMatch + $.scopeMatch.allScopesMatch + +Only these two are consumed: they are the Compass tier-1 *automated* facts worth +surfacing in our aggregate. Metadata reads Compass also owns (schema-present, +documentation, data-source approval) are left to Compass to report directly, per +the ADR ("consumed as existing Compass Check Results"); re-gating them here would +just duplicate Compass. + +Facts are supplied from a local JSON file. Live SoundCheck retrieval is not wired +yet: the read endpoint and access permissions are still being settled with the +Compass team, so ``--facts-file`` is the only path until then. + +Usage: + python -m scripts.mcp.compass_fetch --facts-file facts.json --reports-dir +""" + +from __future__ import annotations + +import argparse +import json +import logging +import sys +from pathlib import Path +from typing import Any + +from scripts.mcp._common import make_finding +from scripts.mcp._phase2 import ( + STATUS_FAIL, + STATUS_NOT_EVALUATED, + STATUS_PASS, + CheckOutcome, + write_check_result, +) + +logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") +logger = logging.getLogger(__name__) + +FACT_TOOLS = "mcp:default/tools" +FACT_SECURITY = "mcp:default/security" + + +def _get(facts: dict[str, Any], fact_ref: str, *path: str) -> Any: + """Walk ``facts[fact_ref]`` down ``path``; return None if any hop is missing.""" + node: Any = facts.get(fact_ref) + for key in path: + if not isinstance(node, dict): + return None + node = node.get(key) + return node + + +def _bool_outcome(check: str, value: Any, ok_reason: str, bad_reason: str) -> CheckOutcome: + """pass/fail from a boolean fact; not_evaluated when the fact is absent.""" + if value is None: + return CheckOutcome(check, STATUS_NOT_EVALUATED, "Source Compass fact not present.", source="compass") + if value is True: + return CheckOutcome(check, STATUS_PASS, ok_reason, source="compass") + finding = make_finding(severity="high", message=bad_reason, rule_id=check) + return CheckOutcome(check, STATUS_FAIL, bad_reason, source="compass", findings=[finding]) + + +def map_facts(facts: dict[str, Any]) -> list[CheckOutcome]: + """Map Compass facts to the consumed Phase 2 check outcomes.""" + outcomes: list[CheckOutcome] = [] + + outcomes.append( + _bool_outcome( + "tool-name-rules", + _get(facts, FACT_TOOLS, "allToolNamesValid"), + "Compass reports all tool names valid.", + "Compass reports one or more tool names invalid.", + ) + ) + + # OAuth matches catalog = enforced AND auth-server matches AND scopes match. + enforced = _get(facts, FACT_SECURITY, "oauth", "enforced") + server_match = _get(facts, FACT_SECURITY, "entityAuth", "authServerMatch") + scope_match = _get(facts, FACT_SECURITY, "scopeMatch", "allScopesMatch") + if enforced is None and server_match is None and scope_match is None: + outcomes.append( + CheckOutcome( + "oauth-catalog-match", STATUS_NOT_EVALUATED, "No Compass OAuth facts present.", source="compass" + ) + ) + else: + findings = [] + if enforced is not True: + findings.append( + make_finding(severity="high", message="OAuth not enforced per Compass", rule_id="oauth-enforced") + ) + if server_match is not True: + findings.append( + make_finding( + severity="high", + message="OAuth auth server does not match catalog per Compass", + rule_id="oauth-server-match", + ) + ) + if scope_match is not True: + findings.append( + make_finding( + severity="high", + message="OAuth scopes do not match catalog per Compass", + rule_id="oauth-scope-match", + ) + ) + if findings: + outcomes.append( + CheckOutcome( + "oauth-catalog-match", + STATUS_FAIL, + "OAuth configuration does not match the catalog.", + source="compass", + findings=findings, + ) + ) + else: + outcomes.append( + CheckOutcome( + "oauth-catalog-match", STATUS_PASS, "OAuth enforced and matches the catalog.", source="compass" + ) + ) + + return outcomes + + +def main() -> int: + parser = argparse.ArgumentParser(description="MCP Phase 2 Compass-consumed checks") + parser.add_argument( + "--reports-dir", type=Path, required=True, help="Directory to write -check.json files into" + ) + parser.add_argument("--facts-file", type=Path, required=True, help="Local JSON file of Compass facts") + args = parser.parse_args() + + if not args.facts_file.is_file(): + logger.error("Facts file not found: %s", args.facts_file) + return 1 + facts = json.loads(args.facts_file.read_text()) + + outcomes = map_facts(facts) + for outcome in outcomes: + write_check_result(args.reports_dir, outcome) + logger.info("Compass consume complete: %d checks", len(outcomes)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/mcp/mcp_client.py b/scripts/mcp/mcp_client.py new file mode 100644 index 0000000..da2d188 --- /dev/null +++ b/scripts/mcp/mcp_client.py @@ -0,0 +1,203 @@ +"""Minimal MCP client over Streamable HTTP for Phase 2 conformance probing. + +Hand-rolled JSON-RPC 2.0 over ``httpx`` (deliberately NOT the ``mcp`` SDK) so the +Phase 2 probe can: + +* send well-formed calls (initialize, tools/list, tools/call), and +* send deliberately malformed / spec-violating payloads and inspect the raw + HTTP response (status code, headers, content-type, body). + +The typed SDK client cannot emit malformed frames and raises on non-conformant +server responses instead of letting us assert on them, which is exactly what the +conformance checks need to observe. ``httpx`` is already a project dependency. + +Streamable HTTP handshake (MCP spec, transport rev 2025-06-18): + 1. POST ``initialize`` with ``Accept: application/json, text/event-stream``. + 2. Capture ``Mcp-Session-Id`` from the response headers; echo it on every + later request. + 3. POST the ``notifications/initialized`` notification (no id) -> 202, no body. + 4. Include ``MCP-Protocol-Version`` on post-init requests. + 5. A request may be answered with ``application/json`` OR ``text/event-stream`` + (SSE); both are handled here. +""" + +from __future__ import annotations + +import json +import logging +from dataclasses import dataclass +from typing import Any + +import httpx + +logger = logging.getLogger(__name__) + +DEFAULT_PROTOCOL_VERSION = "2025-06-18" +ACCEPT_HEADER = "application/json, text/event-stream" +CLIENT_INFO = {"name": "mcp-phase2-probe", "version": "0.1"} + + +@dataclass +class RpcResponse: + """Raw result of a single POST to the MCP endpoint. + + ``message`` is the parsed JSON-RPC object (from a plain JSON body or scraped + from the first SSE ``data:`` frame), or ``None`` when the body is empty (e.g. + a 202 to a notification) or unparseable. + """ + + status_code: int + headers: dict[str, str] + content_type: str + message: dict[str, Any] | list[Any] | None + raw_text: str + + @property + def is_sse(self) -> bool: + return "text/event-stream" in self.content_type + + +def _parse_sse(text: str) -> dict[str, Any] | list[Any] | None: + """Extract the first JSON-RPC message from an SSE stream body.""" + for line in text.splitlines(): + if line.startswith("data:"): + payload = line[len("data:") :].strip() + if not payload: + continue + try: + return json.loads(payload) + except json.JSONDecodeError: + continue + return None + + +class MCPConnectionError(RuntimeError): + """Raised when the server cannot be reached at all (used to mark checks not-evaluated).""" + + +class MCPClient: + """Small stateful client for one MCP server endpoint.""" + + def __init__( + self, + endpoint_url: str, + *, + timeout: float = 30.0, + protocol_version: str = DEFAULT_PROTOCOL_VERSION, + ) -> None: + self.endpoint_url = endpoint_url + self.protocol_version = protocol_version + self.negotiated_version: str | None = None + self.session_id: str | None = None + self._client = httpx.Client(timeout=timeout) + + # -- context management ------------------------------------------------- + def __enter__(self) -> MCPClient: + return self + + def __exit__(self, *exc: object) -> None: + self.close() + + def close(self) -> None: + self._client.close() + + # -- low-level POST ----------------------------------------------------- + def post( + self, + body: Any, + *, + include_session: bool = True, + include_version: bool = True, + extra_headers: dict[str, str] | None = None, + raw: bool = False, + ) -> RpcResponse: + """POST a payload and return the raw parsed response. + + ``body`` is JSON-encoded unless ``raw=True``, in which case it is sent + verbatim (used to send malformed frames for negative conformance tests). + Raises :class:`MCPConnectionError` only when the server is unreachable. + """ + headers = {"Accept": ACCEPT_HEADER, "Content-Type": "application/json"} + if include_version: + headers["MCP-Protocol-Version"] = self.negotiated_version or self.protocol_version + if include_session and self.session_id: + headers["Mcp-Session-Id"] = self.session_id + if extra_headers: + headers.update(extra_headers) + + content = body if raw else json.dumps(body) + try: + resp = self._client.post(self.endpoint_url, content=content, headers=headers) + except httpx.HTTPError as exc: + raise MCPConnectionError(str(exc)) from exc + + text = resp.text + content_type = resp.headers.get("content-type", "") + if "text/event-stream" in content_type: + message = _parse_sse(text) + elif text.strip(): + try: + message = json.loads(text) + except json.JSONDecodeError: + message = None + else: + message = None + + return RpcResponse( + status_code=resp.status_code, + headers=dict(resp.headers), + content_type=content_type, + message=message, + raw_text=text, + ) + + # -- MCP protocol helpers ---------------------------------------------- + def initialize(self, *, request_id: Any = 1, params: dict[str, Any] | None = None) -> RpcResponse: + """Send the ``initialize`` request and capture session id + negotiated version.""" + body = { + "jsonrpc": "2.0", + "id": request_id, + "method": "initialize", + "params": params + or { + "protocolVersion": self.protocol_version, + "capabilities": {}, + "clientInfo": CLIENT_INFO, + }, + } + # No Mcp-Session-Id yet, and MCP-Protocol-Version is a post-init header + # (the version is negotiated via the params body here). + resp = self.post(body, include_session=False, include_version=False) + # Session id lives in the response headers (case-insensitive). + for key, value in resp.headers.items(): + if key.lower() == "mcp-session-id": + self.session_id = value + break + if isinstance(resp.message, dict): + result = resp.message.get("result", {}) + if isinstance(result, dict) and result.get("protocolVersion"): + self.negotiated_version = result["protocolVersion"] + return resp + + def send_initialized(self) -> RpcResponse: + """Send the ``notifications/initialized`` notification (expects 202, empty body).""" + return self.post({"jsonrpc": "2.0", "method": "notifications/initialized"}) + + def list_tools(self, *, request_id: Any = 2) -> RpcResponse: + return self.post({"jsonrpc": "2.0", "id": request_id, "method": "tools/list", "params": {}}) + + def call_tool(self, name: str, arguments: dict[str, Any] | None = None, *, request_id: Any = 3) -> RpcResponse: + return self.post( + { + "jsonrpc": "2.0", + "id": request_id, + "method": "tools/call", + "params": {"name": name, "arguments": arguments or {}}, + } + ) + + def handshake(self) -> RpcResponse: + """Convenience: initialize then send the initialized notification.""" + resp = self.initialize() + self.send_initialized() + return resp diff --git a/scripts/mcp/phase2_probe.py b/scripts/mcp/phase2_probe.py new file mode 100644 index 0000000..28d858d --- /dev/null +++ b/scripts/mcp/phase2_probe.py @@ -0,0 +1,339 @@ +#!/usr/bin/env python3 +"""MCP Phase 2 live conformance probe. + +Connects to a running MCP server over Streamable HTTP and runs the deterministic +(no-LLM) contract/conformance checks that require a live server, writing one +three-state result file per check for the Phase 2 gates to read. + +Checks that Compass already collects as real automated facts (tool-name rules, +OAuth match) or as metadata (schema present, documentation) are NOT probed here - +they are consumed via ``scripts/mcp/compass_fetch.py`` instead. + +Usage: + python -m scripts.mcp.phase2_probe --server-url http://:8080/mcp \ + --reports-dir [--timeout 30] +""" + +from __future__ import annotations + +import argparse +import logging +import sys +import time +from pathlib import Path + +from scripts.mcp._common import make_finding +from scripts.mcp._phase2 import ( + STATUS_FAIL, + STATUS_NOT_EVALUATED, + STATUS_PASS, + CheckOutcome, + write_check_result, +) +from scripts.mcp.mcp_client import MCPClient, MCPConnectionError, RpcResponse + +logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") +logger = logging.getLogger(__name__) + + +# --------------------------------------------------------------------------- +# Individual checks. Each returns a CheckOutcome. +# --------------------------------------------------------------------------- + + +def check_http_support(client: MCPClient, init: RpcResponse) -> CheckOutcome: + """Server responds correctly over HTTP with a parseable JSON-RPC message.""" + findings = [] + if init.status_code != 200: + findings.append( + make_finding( + severity="high", + message=f"initialize returned HTTP {init.status_code}, expected 200", + rule_id="http-status", + ) + ) + if init.message is None: + findings.append( + make_finding( + severity="high", + message="initialize response body was not a parseable JSON-RPC message (json or SSE)", + rule_id="http-body", + ) + ) + if findings: + return CheckOutcome( + "http-support", STATUS_FAIL, "Server did not respond correctly over HTTP.", findings=findings + ) + return CheckOutcome("http-support", STATUS_PASS, f"Server responded over HTTP (content-type: {init.content_type}).") + + +def check_protocol_compliance(client: MCPClient, init: RpcResponse) -> CheckOutcome: + """initialize result, capability negotiation, JSON-RPC 2.0 correctness.""" + findings = [] + msg = init.message if isinstance(init.message, dict) else {} + + if msg.get("jsonrpc") != "2.0": + findings.append( + make_finding( + severity="high", + message=f"initialize response jsonrpc={msg.get('jsonrpc')!r}, expected '2.0'", + rule_id="jsonrpc-version", + ) + ) + if msg.get("id") != 1: + findings.append( + make_finding( + severity="high", + message=f"initialize response id={msg.get('id')!r} did not echo request id 1", + rule_id="id-echo", + ) + ) + + result = msg.get("result", {}) if isinstance(msg.get("result"), dict) else {} + for required in ("protocolVersion", "capabilities", "serverInfo"): + if required not in result: + findings.append( + make_finding( + severity="high", + message=f"InitializeResult missing '{required}' (capability negotiation)", + rule_id="init-result", + ) + ) + + # Unknown method must yield a JSON-RPC error (ideally -32601 method not found). + try: + unknown = client.post({"jsonrpc": "2.0", "id": 90, "method": "no/such/method", "params": {}}) + err = unknown.message.get("error") if isinstance(unknown.message, dict) else None + if not err: + findings.append( + make_finding( + severity="medium", + message="unknown method did not return a JSON-RPC error object", + rule_id="error-unknown-method", + ) + ) + elif err.get("code") != -32601: + findings.append( + make_finding( + severity="low", + message=f"unknown method returned error code {err.get('code')}, expected -32601", + rule_id="error-code", + ) + ) + + # Malformed request (bad jsonrpc version) must not be accepted as success. + bad = client.post({"jsonrpc": "1.0", "id": 91, "method": "tools/list", "params": {}}) + bad_err = bad.message.get("error") if isinstance(bad.message, dict) else None + if bad.status_code == 200 and not bad_err: + findings.append( + make_finding( + severity="medium", + message="malformed request (jsonrpc 1.0) was accepted without an error", + rule_id="error-malformed", + ) + ) + except MCPConnectionError as exc: + return CheckOutcome("protocol-compliance", STATUS_NOT_EVALUATED, f"Server became unreachable mid-probe: {exc}") + + if findings: + return CheckOutcome( + "protocol-compliance", STATUS_FAIL, "Protocol/JSON-RPC 2.0 conformance violations found.", findings=findings + ) + return CheckOutcome( + "protocol-compliance", + STATUS_PASS, + "initialize, capability negotiation, and JSON-RPC 2.0 error handling conform.", + ) + + +def _tools_from(tools_resp: RpcResponse) -> list[dict]: + if isinstance(tools_resp.message, dict): + result = tools_resp.message.get("result", {}) + if isinstance(result, dict): + tools = result.get("tools", []) + if isinstance(tools, list): + return [t for t in tools if isinstance(t, dict)] + return [] + + +def check_schema_conformance(client: MCPClient, tools_resp: RpcResponse) -> CheckOutcome: + """Every tool declares a well-formed input schema (object with type/properties).""" + tools = _tools_from(tools_resp) + if not tools: + return CheckOutcome("schema-conformance", STATUS_NOT_EVALUATED, "tools/list returned no tools to validate.") + findings = [] + for tool in tools: + name = tool.get("name", "") + schema = tool.get("inputSchema") + if not isinstance(schema, dict): + findings.append( + make_finding( + severity="high", message=f"tool '{name}' has no object inputSchema", rule_id="schema-missing" + ) + ) + continue + if schema.get("type") != "object" and "properties" not in schema: + findings.append( + make_finding( + severity="medium", + message=f"tool '{name}' inputSchema is not a JSON-Schema object (no type/properties)", + rule_id="schema-shape", + ) + ) + if findings: + return CheckOutcome( + "schema-conformance", STATUS_FAIL, "One or more tool schemas are not well-formed.", findings=findings + ) + return CheckOutcome("schema-conformance", STATUS_PASS, f"All {len(tools)} tool input schemas are well-formed.") + + +_HINT_FIELDS = ("readOnlyHint", "destructiveHint", "idempotentHint", "openWorldHint") + + +def check_tool_annotations(client: MCPClient, tools_resp: RpcResponse) -> CheckOutcome: + """Validate tool behavior-hint annotations when present. + + Annotations are OPTIONAL in the MCP spec, so their absence is not a + conformance violation: a server that declares none is not_evaluated here + (nothing to verify), never failed. Only a *malformed* annotation (a hint + field present with a non-boolean value) is a real fail. Honesty of the hints + is behavioral and is left to Phase 3. + """ + tools = _tools_from(tools_resp) + if not tools: + return CheckOutcome("tool-annotations", STATUS_NOT_EVALUATED, "tools/list returned no tools to inspect.") + + findings = [] + annotated = 0 + for tool in tools: + name = tool.get("name", "") + annotations = tool.get("annotations") + if not isinstance(annotations, dict): + continue + present_hints = [h for h in _HINT_FIELDS if h in annotations] + if present_hints: + annotated += 1 + for hint in present_hints: + if not isinstance(annotations[hint], bool): + findings.append( + make_finding( + severity="medium", + message=f"tool '{name}' annotation '{hint}' is not a boolean", + rule_id="annotation-malformed", + ) + ) + + if findings: + return CheckOutcome( + "tool-annotations", STATUS_FAIL, "One or more tool annotations are malformed.", findings=findings + ) + if annotated == 0: + return CheckOutcome( + "tool-annotations", + STATUS_NOT_EVALUATED, + "No tools declare behavior-hint annotations (optional in the MCP spec); nothing to verify.", + ) + return CheckOutcome("tool-annotations", STATUS_PASS, f"All {annotated} annotated tool(s) have well-formed hints.") + + +def check_rate_limiting(client: MCPClient, *, burst: int = 20) -> CheckOutcome: + """Send a burst of requests; pass if the server throttles (HTTP 429).""" + try: + statuses = [client.list_tools(request_id=1000 + i).status_code for i in range(burst)] + except MCPConnectionError as exc: + return CheckOutcome("rate-limiting", STATUS_NOT_EVALUATED, f"Server unreachable during burst: {exc}") + if 429 in statuses: + return CheckOutcome( + "rate-limiting", STATUS_PASS, f"Server returned HTTP 429 under a burst of {burst} requests." + ) + return CheckOutcome( + "rate-limiting", + STATUS_NOT_EVALUATED, + f"No throttling observed under a burst of {burst}; cannot tell 'no limit' from a limit above the burst.", + ) + + +def check_response_size_limit(client: MCPClient) -> CheckOutcome: + """Response-size capping is not determinable black-box without a large-output tool + declared cap.""" + return CheckOutcome( + "response-size-limit", + STATUS_NOT_EVALUATED, + "Requires a tool designed to elicit large output and a declared size cap; not determinable black-box.", + ) + + +def check_mandatory_timeouts(client: MCPClient) -> CheckOutcome: + """Server-side timeout enforcement is not observable black-box without a hanging tool.""" + return CheckOutcome( + "mandatory-timeouts", + STATUS_NOT_EVALUATED, + "Server-side timeout enforcement is not observable black-box without a tool that intentionally hangs.", + ) + + +def check_openapi_conformance(client: MCPClient) -> CheckOutcome: + """No OpenAPI contract is supplied to the probe (it lives in the MCP server's repo).""" + return CheckOutcome( + "openapi-conformance", + STATUS_NOT_EVALUATED, + "No OpenAPI contract was provided to the probe; deferred until the MCP repo publishes one.", + ) + + +def probe_all(server_url: str, *, timeout: float = 30.0) -> list[CheckOutcome]: + """Run every live Phase 2 check against ``server_url``. + + If the server is unreachable, all checks are reported not_evaluated (ADR: + a readiness/reachability failure must never be scored as a fail). + """ + live_names = [ + "http-support", + "protocol-compliance", + "schema-conformance", + "tool-annotations", + "rate-limiting", + "response-size-limit", + "mandatory-timeouts", + "openapi-conformance", + ] + with MCPClient(server_url, timeout=timeout) as client: + try: + init = client.initialize() + client.send_initialized() + tools_resp = client.list_tools() + except MCPConnectionError as exc: + reason = f"Server unreachable at {server_url}: {exc}" + logger.warning(reason) + return [CheckOutcome(name, STATUS_NOT_EVALUATED, reason) for name in live_names] + + return [ + check_http_support(client, init), + check_protocol_compliance(client, init), + check_schema_conformance(client, tools_resp), + check_tool_annotations(client, tools_resp), + check_rate_limiting(client), + check_response_size_limit(client), + check_mandatory_timeouts(client), + check_openapi_conformance(client), + ] + + +def main() -> int: + parser = argparse.ArgumentParser(description="MCP Phase 2 live conformance probe") + parser.add_argument("--server-url", required=True, help="MCP endpoint URL, e.g. http://:8080/mcp") + parser.add_argument( + "--reports-dir", type=Path, required=True, help="Directory to write -check.json files into" + ) + parser.add_argument("--timeout", type=float, default=30.0, help="Per-request HTTP timeout in seconds") + args = parser.parse_args() + + start = time.monotonic() + outcomes = probe_all(args.server_url, timeout=args.timeout) + for outcome in outcomes: + write_check_result(args.reports_dir, outcome) + logger.info("Phase 2 probe complete: %d checks in %.1fs", len(outcomes), time.monotonic() - start) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/mcp/run_phase2_gates.py b/scripts/mcp/run_phase2_gates.py new file mode 100644 index 0000000..b4db026 --- /dev/null +++ b/scripts/mcp/run_phase2_gates.py @@ -0,0 +1,93 @@ +#!/usr/bin/env python3 +"""Run the MCP Phase 2 gates over the three-state check result files. + +Reads the ``-check.json`` files produced by the Phase 2 probe and the +Compass-consume step, evaluates each gate, and writes results into the reports +directory: + +- ``-result.json`` per gate (the full GateResult) +- ``phase2-summary.json`` (overall pass/fail + per-check summary with status) + +Three-state aware: a ``not_evaluated`` check never fails the phase. The step +fails (non-zero exit) only when a check genuinely FAILED in block mode. Reporting +to the Evaluation-results database is still handled by the pipeline layer later; +this writes local JSON only. + +Usage: + python -m scripts.mcp.run_phase2_gates --reports-dir [--mode block|warn] +""" + +from __future__ import annotations + +import argparse +import json +import logging +import sys +from pathlib import Path + +from abevalflow.mcp.phase2 import run_phase2 +from abevalflow.schemas import GateMode, GatePolicy + +logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") +logger = logging.getLogger(__name__) + +SUMMARY_FILENAME = "phase2-summary.json" + + +def main() -> int: + parser = argparse.ArgumentParser(description="Run MCP Phase 2 gates") + parser.add_argument( + "--reports-dir", type=Path, required=True, help="Directory holding the Phase 2 check result files" + ) + parser.add_argument( + "--mode", + choices=[m.value for m in GateMode], + default=GateMode.BLOCK.value, + help="Gate enforcement mode (default: block)", + ) + args = parser.parse_args() + + if not args.reports_dir.is_dir(): + logger.error("Not a directory: %s", args.reports_dir) + return 1 + + policy = GatePolicy(default_mode=GateMode(args.mode)) + results = run_phase2(args.reports_dir, policy) + + summary: list[dict] = [] + for result in results: + check_key = result.get_policy_key() + status = result.details.get("status", "unknown") + (args.reports_dir / f"{check_key}-result.json").write_text(result.model_dump_json(indent=2)) + summary.append( + { + "check": check_key, + "status": status, + "passed": result.passed, + "score": result.score, + "findings": len(result.findings), + "source": result.details.get("source"), + "message": result.message, + } + ) + logger.info("%s: status=%s passed=%s score=%.2f", check_key, status, result.passed, result.score) + + overall_passed = all(r.passed for r in results) + counts = { + "pass": sum(1 for s in summary if s["status"] == "pass"), + "fail": sum(1 for s in summary if s["status"] == "fail"), + "not_evaluated": sum(1 for s in summary if s["status"] == "not_evaluated"), + } + (args.reports_dir / SUMMARY_FILENAME).write_text( + json.dumps({"phase": "phase2", "passed": overall_passed, "counts": counts, "checks": summary}, indent=2) + ) + + logger.info("Phase 2 overall: %s (%s)", "PASS" if overall_passed else "FAIL", counts) + # A gate is only not-passed when a real check FAILED in block mode + # (not_evaluated and warn-mode failures keep passed=True), so this exits + # non-zero exactly when the step should fail. + return 0 if overall_passed else 1 + + +if __name__ == "__main__": + sys.exit(main()) From cf42d898bf6d1b7a4b5a22cbdbfbc5f6cd5118de Mon Sep 17 00:00:00 2001 From: operetz Date: Wed, 30 Sep 2026 19:53:22 +0300 Subject: [PATCH 3/9] harden Phase 2 three-state conformance evaluation --- abevalflow/mcp/phase2/base.py | 11 ++++- scripts/mcp/compass_fetch.py | 86 +++++++++++++++++++---------------- scripts/mcp/phase2_probe.py | 46 ++++++++++++++++--- 3 files changed, 95 insertions(+), 48 deletions(-) diff --git a/abevalflow/mcp/phase2/base.py b/abevalflow/mcp/phase2/base.py index c3d798b..fe9e73c 100644 --- a/abevalflow/mcp/phase2/base.py +++ b/abevalflow/mcp/phase2/base.py @@ -99,9 +99,12 @@ def evaluate(self, reports_dir: Path, policy: GatePolicy) -> GateResult: try: data = json.loads(path.read_text()) except (json.JSONDecodeError, OSError) as exc: + # A corrupt result file is a real pipeline defect, so log it loudly and + # score 0.0 - but per the three-state model a "couldn't read" must not + # block outside block mode (mirrors the fail branch below). logger.error("Failed to read %s: %s", path, exc) return self._result( - passed=False, + passed=mode != GateMode.BLOCK, score=0.0, mode=mode, status=STATUS_FAIL, @@ -134,9 +137,13 @@ def evaluate(self, reports_dir: Path, policy: GatePolicy) -> GateResult: ) # fail passed = mode != GateMode.BLOCK + # A fail with no findings must not report a perfect score: _weighted_score + # returns 1.0 for an empty list (correct only on the pass path), so a bare + # fail is floored to 0.0 here. + score = _weighted_score(findings) if findings else 0.0 return self._result( passed=passed, - score=_weighted_score(findings), + score=score, mode=mode, status=STATUS_FAIL, reason=reason, diff --git a/scripts/mcp/compass_fetch.py b/scripts/mcp/compass_fetch.py index 094c098..b0611ec 100644 --- a/scripts/mcp/compass_fetch.py +++ b/scripts/mcp/compass_fetch.py @@ -85,53 +85,59 @@ def map_facts(facts: dict[str, Any]) -> list[CheckOutcome]: ) # OAuth matches catalog = enforced AND auth-server matches AND scopes match. - enforced = _get(facts, FACT_SECURITY, "oauth", "enforced") - server_match = _get(facts, FACT_SECURITY, "entityAuth", "authServerMatch") - scope_match = _get(facts, FACT_SECURITY, "scopeMatch", "allScopesMatch") - if enforced is None and server_match is None and scope_match is None: + # Distinguish an explicit False (a real violation) from an absent fact (None): + # a missing subfield must NOT be read as a violation, or a partially-reported + # entity would falsely FAIL and block the phase. Only when every subfield is + # present and True can we assert the conjunction; any absent subfield with no + # explicit False leaves us unable to evaluate. + oauth_fields = [ + ("oauth-enforced", _get(facts, FACT_SECURITY, "oauth", "enforced"), "OAuth not enforced per Compass"), + ( + "oauth-server-match", + _get(facts, FACT_SECURITY, "entityAuth", "authServerMatch"), + "OAuth auth server does not match catalog per Compass", + ), + ( + "oauth-scope-match", + _get(facts, FACT_SECURITY, "scopeMatch", "allScopesMatch"), + "OAuth scopes do not match catalog per Compass", + ), + ] + violations = [ + make_finding(severity="high", message=msg, rule_id=rule) for rule, value, msg in oauth_fields if value is False + ] + missing = [rule for rule, value, _ in oauth_fields if value is None] + if violations: outcomes.append( CheckOutcome( - "oauth-catalog-match", STATUS_NOT_EVALUATED, "No Compass OAuth facts present.", source="compass" + "oauth-catalog-match", + STATUS_FAIL, + "OAuth configuration does not match the catalog.", + source="compass", + findings=violations, ) ) - else: - findings = [] - if enforced is not True: - findings.append( - make_finding(severity="high", message="OAuth not enforced per Compass", rule_id="oauth-enforced") - ) - if server_match is not True: - findings.append( - make_finding( - severity="high", - message="OAuth auth server does not match catalog per Compass", - rule_id="oauth-server-match", - ) - ) - if scope_match is not True: - findings.append( - make_finding( - severity="high", - message="OAuth scopes do not match catalog per Compass", - rule_id="oauth-scope-match", - ) + elif not missing: + outcomes.append( + CheckOutcome( + "oauth-catalog-match", STATUS_PASS, "OAuth enforced and matches the catalog.", source="compass" ) - if findings: - outcomes.append( - CheckOutcome( - "oauth-catalog-match", - STATUS_FAIL, - "OAuth configuration does not match the catalog.", - source="compass", - findings=findings, - ) + ) + elif len(missing) == len(oauth_fields): + outcomes.append( + CheckOutcome( + "oauth-catalog-match", STATUS_NOT_EVALUATED, "No Compass OAuth facts present.", source="compass" ) - else: - outcomes.append( - CheckOutcome( - "oauth-catalog-match", STATUS_PASS, "OAuth enforced and matches the catalog.", source="compass" - ) + ) + else: + outcomes.append( + CheckOutcome( + "oauth-catalog-match", + STATUS_NOT_EVALUATED, + f"Incomplete Compass OAuth facts (missing: {', '.join(missing)}); cannot assert OAuth-matches-catalog.", + source="compass", ) + ) return outcomes diff --git a/scripts/mcp/phase2_probe.py b/scripts/mcp/phase2_probe.py index 28d858d..97b59ae 100644 --- a/scripts/mcp/phase2_probe.py +++ b/scripts/mcp/phase2_probe.py @@ -156,8 +156,20 @@ def _tools_from(tools_resp: RpcResponse) -> list[dict]: return [] +def _schema_is_object(schema: object) -> bool: + """A JSON-Schema-ish object: declares ``type: object`` or carries ``properties``.""" + return isinstance(schema, dict) and (schema.get("type") == "object" or "properties" in schema) + + def check_schema_conformance(client: MCPClient, tools_resp: RpcResponse) -> CheckOutcome: - """Every tool declares a well-formed input schema (object with type/properties).""" + """Every tool declares a well-formed input schema, plus a well-formed output + schema when one is present. + + inputSchema is required by the MCP tool type, so a missing/non-object one is a + fail. outputSchema is optional in the MCP spec, so it is validated only when + present (its absence is not a violation). Whether actual tool RESPONSES conform + to the declared schema is behavioral and is left to Phase 3. + """ tools = _tools_from(tools_resp) if not tools: return CheckOutcome("schema-conformance", STATUS_NOT_EVALUATED, "tools/list returned no tools to validate.") @@ -171,8 +183,7 @@ def check_schema_conformance(client: MCPClient, tools_resp: RpcResponse) -> Chec severity="high", message=f"tool '{name}' has no object inputSchema", rule_id="schema-missing" ) ) - continue - if schema.get("type") != "object" and "properties" not in schema: + elif not _schema_is_object(schema): findings.append( make_finding( severity="medium", @@ -180,11 +191,21 @@ def check_schema_conformance(client: MCPClient, tools_resp: RpcResponse) -> Chec rule_id="schema-shape", ) ) + # outputSchema is optional; validate its shape only when the tool declares one. + out_schema = tool.get("outputSchema") + if out_schema is not None and not _schema_is_object(out_schema): + findings.append( + make_finding( + severity="medium", + message=f"tool '{name}' outputSchema is present but not a JSON-Schema object", + rule_id="output-schema-shape", + ) + ) if findings: return CheckOutcome( "schema-conformance", STATUS_FAIL, "One or more tool schemas are not well-formed.", findings=findings ) - return CheckOutcome("schema-conformance", STATUS_PASS, f"All {len(tools)} tool input schemas are well-formed.") + return CheckOutcome("schema-conformance", STATUS_PASS, f"All {len(tools)} tool schemas are well-formed.") _HINT_FIELDS = ("readOnlyHint", "destructiveHint", "idempotentHint", "openWorldHint") @@ -272,11 +293,24 @@ def check_mandatory_timeouts(client: MCPClient) -> CheckOutcome: def check_openapi_conformance(client: MCPClient) -> CheckOutcome: - """No OpenAPI contract is supplied to the probe (it lives in the MCP server's repo).""" + """OpenAPI conformance - deferred (not_evaluated). + + NOTE (decision-gated, deferred): the ADR wants requests/responses validated + against the server's OpenAPI contract (ADR lines 103, 209, 257). Unlike + mandatory-timeouts / response-size-limit (undecidable black-box), this is a + solvable check awaiting inputs, deferred on purpose because two prerequisites + are missing: (1) no MCP repo publishes an OpenAPI contract at the standardized + location yet, and (2) the OpenAPI-operation <-> MCP-tool/method mapping is not + defined in the ADR. Wiring a validator before both exist would mean inventing + that convention. When settled, add an optional --openapi-file to this probe, + thread it through mcp-phase2.yaml, and validate here; until then this stays + not_evaluated (never a fail). + """ return CheckOutcome( "openapi-conformance", STATUS_NOT_EVALUATED, - "No OpenAPI contract was provided to the probe; deferred until the MCP repo publishes one.", + "No OpenAPI contract provided and the contract<->tool mapping is unspecified; " + "deferred until an MCP repo publishes a contract at the standardized location.", ) From 69b6e9fd0aa9fcf8cb7609cc9ad54b7a429e4d04 Mon Sep 17 00:00:00 2001 From: operetz Date: Wed, 30 Sep 2026 19:53:33 +0300 Subject: [PATCH 4/9] surface no-user-code scan coverage --- scripts/mcp/_common.py | 23 ++++++++--- scripts/mcp/no_user_code_scan.py | 67 +++++++++++++++++++++++++++++++- 2 files changed, 82 insertions(+), 8 deletions(-) diff --git a/scripts/mcp/_common.py b/scripts/mcp/_common.py index 424380a..4adc3bb 100644 --- a/scripts/mcp/_common.py +++ b/scripts/mcp/_common.py @@ -61,14 +61,25 @@ def write_scan( scanner: str, target: Path, findings: list[dict[str, Any]], + *, + extra: dict[str, Any] | None = None, ) -> None: - """Write a normalized scan report to ``scan_path``.""" + """Write a normalized scan report to ``scan_path``. + + ``extra`` merges additional top-level metadata (e.g. a ``coverage`` block) + into the payload. The Phase 1 gates read only ``findings``, so extra keys are + informational and never change pass/fail. The canonical keys always win, so a + caller cannot accidentally clobber ``findings`` via ``extra``. + """ scan_path.parent.mkdir(parents=True, exist_ok=True) - payload = { - "scanner": scanner, - "target": str(target), - "findings": findings, - } + payload: dict[str, Any] = dict(extra or {}) + payload.update( + { + "scanner": scanner, + "target": str(target), + "findings": findings, + } + ) scan_path.write_text(json.dumps(payload, indent=2)) logger.info("Wrote %d findings to %s", len(findings), scan_path) diff --git a/scripts/mcp/no_user_code_scan.py b/scripts/mcp/no_user_code_scan.py index 1eb1afc..e50025c 100644 --- a/scripts/mcp/no_user_code_scan.py +++ b/scripts/mcp/no_user_code_scan.py @@ -5,6 +5,17 @@ MCP server repository and normalizes results into no-user-code-scan.json for NoUserCodeGate. +Language coverage: the bundled ruleset covers the languages MCP servers are +commonly written in (Python, JavaScript, TypeScript, Go, Java). semgrep only +inspects files in a covered language, so a server written in an uncovered +language (Ruby, Rust, C#, ...) is effectively not scanned - a zero-finding pass +then means "nothing to scan", not "clean". To make that visible, the scan records +a ``coverage`` block (files_scanned / languages / target_files_total) in +no-user-code-scan.json and logs a warning when zero files were scanned, so a pass +shows its own coverage. To cover another language, add rules to no_user_code.yml +(keep it offline; do not switch to `--config auto`, which pulls rules from the +semgrep registry over the network). + Usage: python -m scripts.mcp.no_user_code_scan --reports-dir """ @@ -24,11 +35,49 @@ SCAN_FILENAME = "no-user-code-scan.json" SCANNER = "semgrep" +# Offline multi-language ruleset (Python/JS/TS/Go/Java - see the module docstring +# "Language coverage" note). A file in a language the ruleset does not cover is not +# scanned, so a zero-finding pass on such an artifact is not meaningful; the +# coverage block in the scan JSON records what was actually scanned. RULES_PATH = Path(__file__).parent / "rules" / "no_user_code.yml" # semgrep severity -> our severity. _SEVERITY_MAP = {"ERROR": "high", "WARNING": "medium", "INFO": "low"} +# File extensions -> the ruleset's covered languages, for the coverage report. +_EXT_LANG = { + ".py": "python", + ".pyi": "python", + ".js": "javascript", + ".jsx": "javascript", + ".mjs": "javascript", + ".cjs": "javascript", + ".ts": "typescript", + ".tsx": "typescript", + ".mts": "typescript", + ".cts": "typescript", + ".go": "go", + ".java": "java", +} + + +def coverage_summary(target_dir: Path, scanned: list[str]) -> dict: + """Summarize what semgrep actually inspected. + + ``scanned`` is semgrep's ``paths.scanned`` - the files it ran rules against. + semgrep lists only files in a language the ruleset covers, so this already + excludes uncovered languages and non-code files. ``target_files_total`` counts + every file in the artifact, so a reader can see the scanned fraction and spot a + pass that inspected little or nothing. + """ + languages = sorted({_EXT_LANG.get(Path(p).suffix.lower(), "other") for p in scanned}) + total = sum(1 for p in target_dir.rglob("*") if p.is_file() and ".git" not in p.parts) + return { + "files_scanned": len(scanned), + "languages": languages, + "target_files_total": total, + } + def normalize(semgrep_results: list[dict]) -> list[dict]: """Map semgrep JSON results to normalized findings.""" @@ -94,8 +143,22 @@ def main() -> int: return 1 findings = normalize(data.get("results", [])) - write_scan(scan_path, SCANNER, args.target_dir, findings) - logger.info("No-user-code scan complete: %d finding(s)", len(findings)) + coverage = coverage_summary(args.target_dir, data.get("paths", {}).get("scanned", [])) + write_scan(scan_path, SCANNER, args.target_dir, findings, extra={"coverage": coverage}) + if coverage["files_scanned"] == 0: + logger.warning( + "no-user-code: semgrep scanned 0 files in a covered language (%d files in artifact); " + "this PASS does not show the server avoids user-code execution - add rules for its language", + coverage["target_files_total"], + ) + else: + logger.info( + "No-user-code scan complete: %d finding(s); scanned %d/%d files (%s)", + len(findings), + coverage["files_scanned"], + coverage["target_files_total"], + ", ".join(coverage["languages"]), + ) return 0 From 579e8454428e3a793c069fac2dbd8f03e8c4ea17 Mon Sep 17 00:00:00 2001 From: operetz Date: Thu, 1 Oct 2026 20:00:01 +0300 Subject: [PATCH 5/9] chore(mcp): align Phase 1/2 tasks with konflux/ layout and trim comments - Move mcp-phase{1,2} tasks to pipeline/tasks/konflux/, rename to mcp-phase1-static / mcp-phase2-conformance, and adopt the app.kubernetes.io labels used on the deploy-pipeline branch (name == filename, no namespace). - Trim redundant/duplicated comments in the Phase 1/2 scanners and gates; keep only the WHY notes. No logic change. Co-Authored-By: Claude Opus 4.8 --- abevalflow/mcp/phase2/__init__.py | 3 +- .../mcp-phase1-static.yaml} | 4 ++- .../mcp-phase2-conformance.yaml} | 4 ++- scripts/mcp/compass_fetch.py | 9 +++--- scripts/mcp/no_user_code_scan.py | 31 ++++++------------- scripts/mcp/phase2_probe.py | 16 ++++------ scripts/mcp/run_phase1_gates.py | 5 ++- scripts/mcp/secrets_scan.py | 6 ++-- 8 files changed, 30 insertions(+), 48 deletions(-) rename pipeline/tasks/{phases/mcp-phase1.yaml => konflux/mcp-phase1-static.yaml} (98%) rename pipeline/tasks/{phases/mcp-phase2.yaml => konflux/mcp-phase2-conformance.yaml} (98%) diff --git a/abevalflow/mcp/phase2/__init__.py b/abevalflow/mcp/phase2/__init__.py index e4c0dd0..ed958dc 100644 --- a/abevalflow/mcp/phase2/__init__.py +++ b/abevalflow/mcp/phase2/__init__.py @@ -11,8 +11,7 @@ conformance. * **Compass-consumed** (scripts/mcp/compass_fetch.py) - the Compass tier-1 automated facts consumed rather than re-probed: tool-name rules and - OAuth-matches-catalog. Metadata Compass also owns (schema-present, - documentation, data-source approval) is left to Compass to report directly. + OAuth-matches-catalog. Like Phase 1, this package is kept isolated from the skill pipeline: the gates are instantiated only here via ``run_phase2`` and are never registered in the diff --git a/pipeline/tasks/phases/mcp-phase1.yaml b/pipeline/tasks/konflux/mcp-phase1-static.yaml similarity index 98% rename from pipeline/tasks/phases/mcp-phase1.yaml rename to pipeline/tasks/konflux/mcp-phase1-static.yaml index cfc7192..193861b 100644 --- a/pipeline/tasks/phases/mcp-phase1.yaml +++ b/pipeline/tasks/konflux/mcp-phase1-static.yaml @@ -2,7 +2,9 @@ apiVersion: tekton.dev/v1 kind: Task metadata: name: mcp-phase1-static - namespace: ab-eval-flow + labels: + app.kubernetes.io/name: abevalflow + app.kubernetes.io/component: konflux spec: description: >- MCP evaluation Phase 1 - static / build-time analysis (ADR Approach 4). diff --git a/pipeline/tasks/phases/mcp-phase2.yaml b/pipeline/tasks/konflux/mcp-phase2-conformance.yaml similarity index 98% rename from pipeline/tasks/phases/mcp-phase2.yaml rename to pipeline/tasks/konflux/mcp-phase2-conformance.yaml index 13e07f9..47a134f 100644 --- a/pipeline/tasks/phases/mcp-phase2.yaml +++ b/pipeline/tasks/konflux/mcp-phase2-conformance.yaml @@ -2,7 +2,9 @@ apiVersion: tekton.dev/v1 kind: Task metadata: name: mcp-phase2-conformance - namespace: ab-eval-flow + labels: + app.kubernetes.io/name: abevalflow + app.kubernetes.io/component: konflux spec: description: >- MCP evaluation Phase 2 - deterministic contract/conformance (ADR Approach 4). diff --git a/scripts/mcp/compass_fetch.py b/scripts/mcp/compass_fetch.py index b0611ec..ecc373f 100644 --- a/scripts/mcp/compass_fetch.py +++ b/scripts/mcp/compass_fetch.py @@ -12,11 +12,10 @@ $.entityAuth.authServerMatch $.scopeMatch.allScopesMatch -Only these two are consumed: they are the Compass tier-1 *automated* facts worth -surfacing in our aggregate. Metadata reads Compass also owns (schema-present, -documentation, data-source approval) are left to Compass to report directly, per -the ADR ("consumed as existing Compass Check Results"); re-gating them here would -just duplicate Compass. +Only these two are consumed: the Compass tier-1 *automated* facts worth surfacing +in our aggregate. Metadata Compass owns directly (schema-present, documentation, +data-source approval) is not re-gated here, per the ADR ("consumed as existing +Compass Check Results"). Facts are supplied from a local JSON file. Live SoundCheck retrieval is not wired yet: the read endpoint and access permissions are still being settled with the diff --git a/scripts/mcp/no_user_code_scan.py b/scripts/mcp/no_user_code_scan.py index e50025c..4a5e4a3 100644 --- a/scripts/mcp/no_user_code_scan.py +++ b/scripts/mcp/no_user_code_scan.py @@ -2,19 +2,11 @@ """MCP Phase 1 "does not execute user-provided code" scan. Runs semgrep with the bundled offline ruleset (rules/no_user_code.yml) over an -MCP server repository and normalizes results into no-user-code-scan.json for -NoUserCodeGate. - -Language coverage: the bundled ruleset covers the languages MCP servers are -commonly written in (Python, JavaScript, TypeScript, Go, Java). semgrep only -inspects files in a covered language, so a server written in an uncovered -language (Ruby, Rust, C#, ...) is effectively not scanned - a zero-finding pass -then means "nothing to scan", not "clean". To make that visible, the scan records -a ``coverage`` block (files_scanned / languages / target_files_total) in -no-user-code-scan.json and logs a warning when zero files were scanned, so a pass -shows its own coverage. To cover another language, add rules to no_user_code.yml -(keep it offline; do not switch to `--config auto`, which pulls rules from the -semgrep registry over the network). +MCP server repo and normalizes results into no-user-code-scan.json for +NoUserCodeGate. The ruleset covers Python/JS/TS/Go/Java; a file in an uncovered +language is not scanned, so the scan records a ``coverage`` block and warns when +zero files were scanned (a zero-finding pass then means "nothing to scan", not +"clean"). Keep the ruleset offline - do not use ``--config auto`` (network). Usage: python -m scripts.mcp.no_user_code_scan --reports-dir @@ -35,10 +27,7 @@ SCAN_FILENAME = "no-user-code-scan.json" SCANNER = "semgrep" -# Offline multi-language ruleset (Python/JS/TS/Go/Java - see the module docstring -# "Language coverage" note). A file in a language the ruleset does not cover is not -# scanned, so a zero-finding pass on such an artifact is not meaningful; the -# coverage block in the scan JSON records what was actually scanned. +# Offline multi-language ruleset (Python/JS/TS/Go/Java). See module docstring. RULES_PATH = Path(__file__).parent / "rules" / "no_user_code.yml" # semgrep severity -> our severity. @@ -64,11 +53,9 @@ def coverage_summary(target_dir: Path, scanned: list[str]) -> dict: """Summarize what semgrep actually inspected. - ``scanned`` is semgrep's ``paths.scanned`` - the files it ran rules against. - semgrep lists only files in a language the ruleset covers, so this already - excludes uncovered languages and non-code files. ``target_files_total`` counts - every file in the artifact, so a reader can see the scanned fraction and spot a - pass that inspected little or nothing. + ``scanned`` is semgrep's ``paths.scanned`` (only files in a covered language, so + uncovered languages are already excluded). ``target_files_total`` counts every + file, so a reader can see the scanned fraction. """ languages = sorted({_EXT_LANG.get(Path(p).suffix.lower(), "other") for p in scanned}) total = sum(1 for p in target_dir.rglob("*") if p.is_file() and ".git" not in p.parts) diff --git a/scripts/mcp/phase2_probe.py b/scripts/mcp/phase2_probe.py index 97b59ae..411564b 100644 --- a/scripts/mcp/phase2_probe.py +++ b/scripts/mcp/phase2_probe.py @@ -295,16 +295,12 @@ def check_mandatory_timeouts(client: MCPClient) -> CheckOutcome: def check_openapi_conformance(client: MCPClient) -> CheckOutcome: """OpenAPI conformance - deferred (not_evaluated). - NOTE (decision-gated, deferred): the ADR wants requests/responses validated - against the server's OpenAPI contract (ADR lines 103, 209, 257). Unlike - mandatory-timeouts / response-size-limit (undecidable black-box), this is a - solvable check awaiting inputs, deferred on purpose because two prerequisites - are missing: (1) no MCP repo publishes an OpenAPI contract at the standardized - location yet, and (2) the OpenAPI-operation <-> MCP-tool/method mapping is not - defined in the ADR. Wiring a validator before both exist would mean inventing - that convention. When settled, add an optional --openapi-file to this probe, - thread it through mcp-phase2.yaml, and validate here; until then this stays - not_evaluated (never a fail). + The ADR wants requests/responses validated against the server's OpenAPI + contract (ADR lines 103, 209, 257), but two prerequisites are missing: no MCP + repo publishes a contract at the standardized location yet, and the + OpenAPI-operation <-> MCP-tool/method mapping is undefined. Wiring a validator + before both exist would invent that convention, so this stays not_evaluated + (never a fail) until they are settled. """ return CheckOutcome( "openapi-conformance", diff --git a/scripts/mcp/run_phase1_gates.py b/scripts/mcp/run_phase1_gates.py index 7803110..fb3b0ec 100644 --- a/scripts/mcp/run_phase1_gates.py +++ b/scripts/mcp/run_phase1_gates.py @@ -7,9 +7,8 @@ - ``-result.json`` per gate (the full GateResult) - ``phase1-summary.json`` (overall pass/fail + per-gate summary) -This is the Phase 1 reporting stub. The real Evaluation-results-database writer -(three-state pass/fail/not-evaluated) is added by the pipeline layer later; Phase -1 always executes, so its checks resolve to pass/fail only. +Reporting stub: the three-state Evaluation-results-database writer is added by the +pipeline layer later; Phase 1 checks are pass/fail only. Usage: python -m scripts.mcp.run_phase1_gates --reports-dir [--mode block|warn] diff --git a/scripts/mcp/secrets_scan.py b/scripts/mcp/secrets_scan.py index 0d1e728..499922d 100644 --- a/scripts/mcp/secrets_scan.py +++ b/scripts/mcp/secrets_scan.py @@ -25,10 +25,8 @@ SCAN_FILENAME = "secrets-scan.json" SCANNER = "gitleaks" -# gitleaks has no severity field. Every gitleaks hit is a candidate hardcoded -# credential, so all findings are HIGH: the gate blocks in block mode only on -# HIGH/CRITICAL, and a leaked secret must never pass Phase 1 just because it was -# caught by the broad generic rule rather than a vendor-specific one. +# gitleaks has no severity field; every hit is a candidate hardcoded credential, +# so all findings are HIGH (the gate blocks on HIGH/CRITICAL in block mode). def normalize(gitleaks_findings: list[dict]) -> list[dict]: From a148d54006d6f060fc2d2389788c915a36b6745f Mon Sep 17 00:00:00 2001 From: operetz Date: Thu, 1 Oct 2026 20:05:31 +0300 Subject: [PATCH 6/9] add unit tests for both phases --- tests/test_mcp_phase1.py | 236 ++++++++++++++++++++++++++++ tests/test_mcp_phase2.py | 326 +++++++++++++++++++++++++++++++++++++++ 2 files changed, 562 insertions(+) create mode 100644 tests/test_mcp_phase1.py create mode 100644 tests/test_mcp_phase2.py diff --git a/tests/test_mcp_phase1.py b/tests/test_mcp_phase1.py new file mode 100644 index 0000000..5364bd4 --- /dev/null +++ b/tests/test_mcp_phase1.py @@ -0,0 +1,236 @@ +"""Tests for MCP Phase 1 static / build-time checks. + +Two layers: + +- Unit tests exercise the normalization functions and the gate pass/fail logic + with synthetic scan JSON. No external tools required - safe in any CI. +- Integration tests build a fixture repo in a temp dir and run the real scanner + (gitleaks / semgrep / licensee) end-to-end. Skipped when the tool is absent. +""" + +from __future__ import annotations + +import json +import secrets +import shutil +import string +from pathlib import Path + +import pytest + +from abevalflow.mcp.phase1 import ( + DEFAULT_ALLOWED_LICENSES, + LicenseGate, + NoUserCodeGate, + SecretsGate, + run_phase1, +) +from abevalflow.schemas import GateMode, GatePolicy +from scripts.mcp import license_scan, no_user_code_scan, secrets_scan + +BLOCK = GatePolicy(default_mode=GateMode.BLOCK) + + +def _write(reports_dir: Path, filename: str, findings: list[dict]) -> None: + (reports_dir / filename).write_text(json.dumps({"findings": findings})) + + +# --------------------------------------------------------------------------- +# Normalization unit tests (no external tools) +# --------------------------------------------------------------------------- + + +def test_secrets_normalize_maps_fields_and_severity(): + raw = [ + {"RuleID": "aws-access-token", "Description": "AWS key", "File": "a.py", "StartLine": 3}, + {"RuleID": "generic-api-key", "Description": "generic", "File": "b.py", "StartLine": 7}, + ] + findings = secrets_scan.normalize(raw) + assert [f["rule_id"] for f in findings] == ["aws-access-token", "generic-api-key"] + assert findings[0]["severity"] == "high" + assert findings[1]["severity"] == "high" # every gitleaks hit is HIGH + assert findings[0]["file_path"] == "a.py" + + +def test_no_user_code_normalize_maps_semgrep_severity(): + raw = [ + { + "check_id": "python-dynamic-exec", + "path": "bad.py", + "start": {"line": 2}, + "extra": {"severity": "ERROR", "message": "eval used"}, + } + ] + findings = no_user_code_scan.normalize(raw) + assert findings[0]["severity"] == "high" + assert findings[0]["rule_id"] == "python-dynamic-exec" + assert findings[0]["file_path"] == "bad.py" + + +def test_license_allowed_produces_no_findings(): + detected = [{"spdx_id": "MIT"}] + findings = license_scan.evaluate_licenses(detected, {"mit", "apache-2.0"}, []) + assert findings == [] + + +def test_license_disallowed_flags_high(): + detected = [{"spdx_id": "GPL-3.0"}] + matched = [{"filename": "LICENSE"}] + findings = license_scan.evaluate_licenses(detected, {"mit"}, matched) + assert len(findings) == 1 + assert findings[0]["severity"] == "high" + assert findings[0]["rule_id"] == "license-not-allowed" + assert findings[0]["file_path"] == "LICENSE" + + +def test_license_missing_flags_high(): + findings = license_scan.evaluate_licenses([], {"mit"}, []) + assert findings[0]["rule_id"] == "license-missing" + + +# --------------------------------------------------------------------------- +# Gate logic tests (synthetic scan JSON) +# --------------------------------------------------------------------------- + + +def test_secrets_gate_fails_on_high_in_block_mode(tmp_path): + _write(tmp_path, "secrets-scan.json", [{"severity": "high", "message": "leak", "rule_id": "r"}]) + result = SecretsGate().evaluate(tmp_path, BLOCK) + assert result.passed is False + + +def test_secrets_gate_passes_when_clean(tmp_path): + _write(tmp_path, "secrets-scan.json", []) + result = SecretsGate().evaluate(tmp_path, BLOCK) + assert result.passed is True + assert result.score == 1.0 + + +def test_run_phase1_all_clean_passes(tmp_path): + _write(tmp_path, "secrets-scan.json", []) + _write(tmp_path, "no-user-code-scan.json", []) + _write(tmp_path, "license-scan.json", []) + results = run_phase1(tmp_path, BLOCK) + assert len(results) == 3 + assert all(r.passed for r in results) + assert {r.policy_key for r in results} == {"mcp-secrets", "mcp-no-user-code", "mcp-license"} + + +def test_run_phase1_one_dirty_fails_that_gate(tmp_path): + _write(tmp_path, "secrets-scan.json", []) + _write(tmp_path, "no-user-code-scan.json", [{"severity": "high", "message": "eval", "rule_id": "x"}]) + _write(tmp_path, "license-scan.json", []) + results = {r.policy_key: r for r in run_phase1(tmp_path, BLOCK)} + assert results["mcp-secrets"].passed is True + assert results["mcp-no-user-code"].passed is False + assert results["mcp-license"].passed is True + + +# --------------------------------------------------------------------------- +# Integration tests (real tools, skipped when absent) +# --------------------------------------------------------------------------- + +_MIT = """MIT License + +Copyright (c) 2026 Example + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. +""" + + +@pytest.mark.skipif(not shutil.which("gitleaks"), reason="gitleaks not installed") +def test_integration_secrets_detects_planted_key(tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + # Synthetic key generated at runtime - no hardcoded credential in this source + # (only the public "AKIA" prefix). gitleaks matches by pattern, so a generated + # AKIA+16 key still trips its AWS rule; the full value lives only in the temp file. + access_key = "AKIA" + "".join(secrets.choice(string.ascii_uppercase + string.digits) for _ in range(16)) + secret_key = "".join(secrets.choice(string.ascii_letters + string.digits) for _ in range(40)) + (repo / "config.py").write_text( + f'AWS_ACCESS_KEY_ID = "{access_key}"\naws_secret = "{secret_key}"\n' + ) + reports = tmp_path / "reports" + reports.mkdir() + rc = _run_scanner(secrets_scan, [str(repo), "--reports-dir", str(reports)]) + assert rc == 0 + result = SecretsGate().evaluate(reports, BLOCK) + assert result.passed is False + assert result.findings + + +@pytest.mark.skipif(not shutil.which("semgrep"), reason="semgrep not installed") +def test_integration_no_user_code_detects_eval(tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + (repo / "bad.py").write_text("def handler(x):\n return eval(x)\n") + reports = tmp_path / "reports" + reports.mkdir() + rc = _run_scanner(no_user_code_scan, [str(repo), "--reports-dir", str(reports)]) + assert rc == 0 + result = NoUserCodeGate().evaluate(reports, BLOCK) + assert result.passed is False + + +@pytest.mark.skipif(not shutil.which("semgrep"), reason="semgrep not installed") +def test_integration_no_user_code_clean_passes(tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + (repo / "good.py").write_text("def handler(x):\n return x.upper()\n") + reports = tmp_path / "reports" + reports.mkdir() + rc = _run_scanner(no_user_code_scan, [str(repo), "--reports-dir", str(reports)]) + assert rc == 0 + result = NoUserCodeGate().evaluate(reports, BLOCK) + assert result.passed is True + + +@pytest.mark.skipif(not shutil.which("licensee"), reason="licensee not installed") +def test_integration_license_mit_passes(tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + (repo / "LICENSE").write_text(_MIT) + reports = tmp_path / "reports" + reports.mkdir() + rc = _run_scanner( + license_scan, + [str(repo), "--reports-dir", str(reports), *_allow_args()], + ) + assert rc == 0 + result = LicenseGate().evaluate(reports, BLOCK) + assert result.passed is True + + +def _allow_args() -> list[str]: + args: list[str] = [] + for spdx in DEFAULT_ALLOWED_LICENSES: + args += ["--allow", spdx] + return args + + +def _run_scanner(module, argv: list[str]) -> int: + """Invoke a scanner module's main() with a patched argv.""" + import sys + + old = sys.argv + sys.argv = [module.__name__, *argv] + try: + return module.main() + finally: + sys.argv = old diff --git a/tests/test_mcp_phase2.py b/tests/test_mcp_phase2.py new file mode 100644 index 0000000..11af6d0 --- /dev/null +++ b/tests/test_mcp_phase2.py @@ -0,0 +1,326 @@ +"""Tests for MCP Phase 2 (deterministic contract/conformance). + +Unit tests use synthetic JSON-RPC responses and a fake client, so they need no +running server and are CI-safe. The integration test runs the real probe against +a live MCP server and is skipped unless MCP_TEST_SERVER_URL is set (e.g. a local +generic-mock-mcp-server). +""" + +from __future__ import annotations + +import json +import os + +import pytest + +from abevalflow.gates.base import GateMode +from abevalflow.mcp.phase2 import PHASE2_CHECKS, run_phase2 +from abevalflow.mcp.phase2.base import Phase2Gate +from abevalflow.schemas import GatePolicy +from scripts.mcp import compass_fetch, phase2_probe, run_phase2_gates +from scripts.mcp._phase2 import STATUS_FAIL, STATUS_NOT_EVALUATED, STATUS_PASS, CheckOutcome, write_check_result +from scripts.mcp.mcp_client import RpcResponse, _parse_sse + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + + +def _rpc(message, *, status=200, content_type="application/json", headers=None): + return RpcResponse( + status_code=status, + headers=headers or {}, + content_type=content_type, + message=message, + raw_text=json.dumps(message) if message is not None else "", + ) + + +class FakeClient: + """Routes .post()/.list_tools() to canned RpcResponses by JSON-RPC method.""" + + def __init__(self, by_method): + self._by_method = by_method + + def post(self, body, **_kw): + method = body.get("method") if isinstance(body, dict) else None + resp = self._by_method.get(method) + if resp is None: + return _rpc({"jsonrpc": "2.0", "id": body.get("id"), "error": {"code": -32601, "message": "not found"}}) + return resp + + def list_tools(self, request_id=2): + return self._by_method["tools/list"] + + +# --------------------------------------------------------------------------- +# mcp_client +# --------------------------------------------------------------------------- + + +def test_parse_sse_extracts_json_rpc(): + body = 'event: message\ndata: {"jsonrpc": "2.0", "id": 1, "result": {}}\n\n' + parsed = _parse_sse(body) + assert parsed == {"jsonrpc": "2.0", "id": 1, "result": {}} + + +def test_parse_sse_returns_none_without_data(): + assert _parse_sse("event: ping\n\n") is None + + +# --------------------------------------------------------------------------- +# Probe checks (synthetic responses) +# --------------------------------------------------------------------------- + + +def _init_ok(): + return _rpc( + { + "jsonrpc": "2.0", + "id": 1, + "result": {"protocolVersion": "2025-06-18", "capabilities": {}, "serverInfo": {"name": "x"}}, + } + ) + + +def test_http_support_pass_and_fail(): + assert phase2_probe.check_http_support(None, _init_ok()).status == STATUS_PASS + bad = _rpc(None, status=500, content_type="text/plain") + out = phase2_probe.check_http_support(None, bad) + assert out.status == STATUS_FAIL + assert len(out.findings) == 2 # bad status + unparseable body + + +def test_protocol_compliance_pass(): + client = FakeClient( + { + "no/such/method": _rpc({"jsonrpc": "2.0", "id": 90, "error": {"code": -32601, "message": "nope"}}), + "tools/list": _rpc({"jsonrpc": "2.0", "id": 91, "error": {"code": -32600, "message": "bad"}}), + } + ) + out = phase2_probe.check_protocol_compliance(client, _init_ok()) + assert out.status == STATUS_PASS, out.findings + + +def test_protocol_compliance_flags_missing_result_and_bad_id(): + init = _rpc({"jsonrpc": "2.0", "id": 999, "result": {"capabilities": {}}}) # wrong id, missing fields + client = FakeClient( + { + "no/such/method": _rpc({"jsonrpc": "2.0", "id": 90, "error": {"code": -32601}}), + "tools/list": _rpc({"jsonrpc": "2.0", "id": 91, "error": {"code": -32600}}), + } + ) + out = phase2_probe.check_protocol_compliance(client, init) + assert out.status == STATUS_FAIL + rule_ids = {f["rule_id"] for f in out.findings} + assert "id-echo" in rule_ids + assert "init-result" in rule_ids + + +def test_schema_conformance_pass_and_fail(): + good = _rpc({"result": {"tools": [{"name": "a", "inputSchema": {"type": "object", "properties": {}}}]}}) + assert phase2_probe.check_schema_conformance(None, good).status == STATUS_PASS + + bad = _rpc({"result": {"tools": [{"name": "b"}]}}) # no inputSchema + out = phase2_probe.check_schema_conformance(None, bad) + assert out.status == STATUS_FAIL + + empty = _rpc({"result": {"tools": []}}) + assert phase2_probe.check_schema_conformance(None, empty).status == STATUS_NOT_EVALUATED + + +def test_schema_conformance_output_schema_when_present(): + # Valid input + valid output -> pass. + both = _rpc( + {"result": {"tools": [{"name": "a", "inputSchema": {"type": "object"}, "outputSchema": {"type": "object"}}]}} + ) + assert phase2_probe.check_schema_conformance(None, both).status == STATUS_PASS + + # Present-but-malformed output schema -> fail (optional, but must be well-formed when declared). + bad_out = _rpc({"result": {"tools": [{"name": "b", "inputSchema": {"type": "object"}, "outputSchema": "nope"}]}}) + out = phase2_probe.check_schema_conformance(None, bad_out) + assert out.status == STATUS_FAIL + assert out.findings[0]["rule_id"] == "output-schema-shape" + + # No output schema at all -> not a violation (optional in the MCP spec). + no_out = _rpc({"result": {"tools": [{"name": "c", "inputSchema": {"type": "object"}}]}}) + assert phase2_probe.check_schema_conformance(None, no_out).status == STATUS_PASS + + +def test_tool_annotations_absent_is_not_evaluated(): + # Annotations are optional in the MCP spec: none declared -> not_evaluated. + resp = _rpc({"result": {"tools": [{"name": "a"}, {"name": "b"}]}}) + assert phase2_probe.check_tool_annotations(None, resp).status == STATUS_NOT_EVALUATED + + +def test_tool_annotations_wellformed_passes(): + resp = _rpc({"result": {"tools": [{"name": "a", "annotations": {"readOnlyHint": True}}, {"name": "b"}]}}) + assert phase2_probe.check_tool_annotations(None, resp).status == STATUS_PASS + + +def test_tool_annotations_malformed_fails(): + resp = _rpc({"result": {"tools": [{"name": "a", "annotations": {"readOnlyHint": "yes"}}]}}) + out = phase2_probe.check_tool_annotations(None, resp) + assert out.status == STATUS_FAIL + assert out.findings[0]["rule_id"] == "annotation-malformed" + + +def test_rate_limiting_pass_on_429(): + class Burst: + def list_tools(self, request_id=0): + return _rpc({}, status=429 if request_id == 1005 else 200) + + assert phase2_probe.check_rate_limiting(Burst(), burst=10).status == STATUS_PASS + + +def test_rate_limiting_not_evaluated_without_throttle(): + class Burst: + def list_tools(self, request_id=0): + return _rpc({}, status=200) + + assert phase2_probe.check_rate_limiting(Burst(), burst=5).status == STATUS_NOT_EVALUATED + + +def test_deferred_checks_are_not_evaluated(): + assert phase2_probe.check_response_size_limit(None).status == STATUS_NOT_EVALUATED + assert phase2_probe.check_mandatory_timeouts(None).status == STATUS_NOT_EVALUATED + assert phase2_probe.check_openapi_conformance(None).status == STATUS_NOT_EVALUATED + + +# --------------------------------------------------------------------------- +# Compass consume +# --------------------------------------------------------------------------- + + +def test_map_facts_pass(): + facts = { + "mcp:default/tools": {"allToolNamesValid": True}, + "mcp:default/security": { + "oauth": {"enforced": True}, + "entityAuth": {"authServerMatch": True}, + "scopeMatch": {"allScopesMatch": True}, + }, + } + by_check = {o.check: o for o in compass_fetch.map_facts(facts)} + assert set(by_check) == {"tool-name-rules", "oauth-catalog-match"} + assert by_check["tool-name-rules"].status == STATUS_PASS + assert by_check["oauth-catalog-match"].status == STATUS_PASS + assert all(o.source == "compass" for o in by_check.values()) + + +def test_map_facts_fail_and_missing(): + facts = {"mcp:default/tools": {"allToolNamesValid": False}} + by_check = {o.check: o for o in compass_fetch.map_facts(facts)} + assert by_check["tool-name-rules"].status == STATUS_FAIL + assert by_check["oauth-catalog-match"].status == STATUS_NOT_EVALUATED # no OAuth facts + + +def test_map_facts_oauth_partial_is_not_evaluated(): + # Only one OAuth subfield present -> cannot assert the conjunction -> not_evaluated, + # never a fail (a partially-reported entity must not be blocked). + facts = {"mcp:default/security": {"oauth": {"enforced": True}}} + out = {o.check: o for o in compass_fetch.map_facts(facts)}["oauth-catalog-match"] + assert out.status == STATUS_NOT_EVALUATED + assert out.findings == [] + + +def test_map_facts_oauth_explicit_false_fails_only_for_false(): + # An explicit False is a real violation; an absent sibling is not. + facts = { + "mcp:default/security": { + "oauth": {"enforced": True}, + "entityAuth": {"authServerMatch": False}, + # scopeMatch absent + } + } + out = {o.check: o for o in compass_fetch.map_facts(facts)}["oauth-catalog-match"] + assert out.status == STATUS_FAIL + assert {f["rule_id"] for f in out.findings} == {"oauth-server-match"} + + +def test_missing_facts_file_fails(monkeypatch, tmp_path): + monkeypatch.setattr( + "sys.argv", + ["compass_fetch", "--facts-file", str(tmp_path / "nope.json"), "--reports-dir", str(tmp_path)], + ) + assert compass_fetch.main() == 1 + + +# --------------------------------------------------------------------------- +# Gates + runner (three-state) +# --------------------------------------------------------------------------- + + +def _write(tmp_path, check, status, findings=None): + write_check_result(tmp_path, CheckOutcome(check, status, "reason", findings=findings or [])) + + +def test_gate_pass_fail_not_evaluated(tmp_path): + _write(tmp_path, "http-support", STATUS_PASS) + _write(tmp_path, "protocol-compliance", STATUS_FAIL, [{"severity": "high", "message": "m", "rule_id": "r"}]) + _write(tmp_path, "openapi-conformance", STATUS_NOT_EVALUATED) + # remaining checks have no file -> not_evaluated, must not fail + + policy = GatePolicy(default_mode=GateMode.BLOCK) + results = {r.get_policy_key(): r for r in run_phase2(tmp_path, policy)} + + assert len(results) == len(PHASE2_CHECKS) + assert results["http-support"].passed is True + assert results["protocol-compliance"].passed is False # fail in block mode + assert results["openapi-conformance"].passed is True # not_evaluated never penalized + assert results["mandatory-timeouts"].details["status"] == STATUS_NOT_EVALUATED # missing file + + +def test_corrupt_result_file_blocks_only_in_block_mode(tmp_path): + (tmp_path / "http-support-check.json").write_text("{ not valid json ") + gate = Phase2Gate("http-support") + warn = gate.evaluate(tmp_path, GatePolicy(default_mode=GateMode.WARN)) + assert warn.details["status"] == STATUS_FAIL + assert warn.passed is True # warn never blocks, even on a corrupt file + block = gate.evaluate(tmp_path, GatePolicy(default_mode=GateMode.BLOCK)) + assert block.passed is False + + +def test_fail_without_findings_scores_zero(tmp_path): + # A bare fail (no findings) must not report a perfect score. + write_check_result(tmp_path, CheckOutcome("http-support", STATUS_FAIL, "boom", findings=[])) + res = Phase2Gate("http-support").evaluate(tmp_path, GatePolicy(default_mode=GateMode.BLOCK)) + assert res.passed is False + assert res.score == 0.0 + + +def test_not_evaluated_does_not_fail_runner(tmp_path, monkeypatch): + _write(tmp_path, "http-support", STATUS_PASS) + for check in PHASE2_CHECKS: + if check != "http-support": + _write(tmp_path, check, STATUS_NOT_EVALUATED) + monkeypatch.setattr("sys.argv", ["run_phase2_gates", "--reports-dir", str(tmp_path), "--mode", "block"]) + assert run_phase2_gates.main() == 0 + summary = json.loads((tmp_path / "phase2-summary.json").read_text()) + assert summary["passed"] is True + assert summary["counts"]["not_evaluated"] == len(PHASE2_CHECKS) - 1 + + +def test_real_fail_exits_nonzero_in_block(tmp_path, monkeypatch): + for check in PHASE2_CHECKS: + _write(tmp_path, check, STATUS_PASS) + _write(tmp_path, "protocol-compliance", STATUS_FAIL, [{"severity": "high", "message": "m", "rule_id": "r"}]) + monkeypatch.setattr("sys.argv", ["run_phase2_gates", "--reports-dir", str(tmp_path), "--mode", "block"]) + assert run_phase2_gates.main() == 1 + + +# --------------------------------------------------------------------------- +# Integration (real server) +# --------------------------------------------------------------------------- + +_SERVER_URL = os.environ.get("MCP_TEST_SERVER_URL") + + +@pytest.mark.skipif(not _SERVER_URL, reason="set MCP_TEST_SERVER_URL to a running MCP server") +def test_probe_against_live_server(tmp_path): + # Smoke against a real server, e.g. generic-mock-mcp-server run with + # --transport streamable-http on :8080 (endpoint http://localhost:8080/mcp). + outcomes = phase2_probe.probe_all(_SERVER_URL, timeout=10.0) + by_check = {o.check: o for o in outcomes} + assert by_check["http-support"].status == STATUS_PASS + assert by_check["protocol-compliance"].status in (STATUS_PASS, STATUS_FAIL) From 9a525a7e9542245478836f2871c30cfc67ac8373 Mon Sep 17 00:00:00 2001 From: operetz Date: Sun, 4 Oct 2026 11:54:26 +0300 Subject: [PATCH 7/9] style: format test fixture (ruff) --- tests/test_mcp_phase1.py | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/tests/test_mcp_phase1.py b/tests/test_mcp_phase1.py index 5364bd4..91daab1 100644 --- a/tests/test_mcp_phase1.py +++ b/tests/test_mcp_phase1.py @@ -163,9 +163,7 @@ def test_integration_secrets_detects_planted_key(tmp_path): # AKIA+16 key still trips its AWS rule; the full value lives only in the temp file. access_key = "AKIA" + "".join(secrets.choice(string.ascii_uppercase + string.digits) for _ in range(16)) secret_key = "".join(secrets.choice(string.ascii_letters + string.digits) for _ in range(40)) - (repo / "config.py").write_text( - f'AWS_ACCESS_KEY_ID = "{access_key}"\naws_secret = "{secret_key}"\n' - ) + (repo / "config.py").write_text(f'AWS_ACCESS_KEY_ID = "{access_key}"\naws_secret = "{secret_key}"\n') reports = tmp_path / "reports" reports.mkdir() rc = _run_scanner(secrets_scan, [str(repo), "--reports-dir", str(reports)]) From 1dd0ce252af8aeebc52f14cc00baa11061d67c85 Mon Sep 17 00:00:00 2001 From: operetz Date: Tue, 6 Oct 2026 14:29:54 +0300 Subject: [PATCH 8/9] refactor(mcp): address PR review - pre-built image, trim outputs, clearer layout - Add containers/mcp-eval/Containerfile baking scanners (gitleaks, semgrep, licensee) + deps; both Phase 1/2 tasks use it and drop the runtime git clone, tool installs, and pipeline-repo-url/revision params - Trim unused Tekton results (per-gate passes, not-evaluated-count); keep phase1/phase2-passed + report-path - Rename mcp_client.py -> _mcp_client.py and document entry points vs helpers - compass_fetch: degrade missing/malformed facts to not_evaluated instead of aborting Phase 2; pin image Python deps Co-Authored-By: Claude Opus 4.8 --- containers/mcp-eval/Containerfile | 61 ++++++++++++++ pipeline/tasks/konflux/mcp-phase1-static.yaml | 79 ++----------------- .../tasks/konflux/mcp-phase2-conformance.yaml | 48 +++-------- scripts/mcp/__init__.py | 21 ++++- scripts/mcp/{mcp_client.py => _mcp_client.py} | 0 scripts/mcp/compass_fetch.py | 15 +++- scripts/mcp/phase2_probe.py | 2 +- tests/test_mcp_phase2.py | 21 ++++- 8 files changed, 126 insertions(+), 121 deletions(-) create mode 100644 containers/mcp-eval/Containerfile rename scripts/mcp/{mcp_client.py => _mcp_client.py} (100%) diff --git a/containers/mcp-eval/Containerfile b/containers/mcp-eval/Containerfile new file mode 100644 index 0000000..45471a5 --- /dev/null +++ b/containers/mcp-eval/Containerfile @@ -0,0 +1,61 @@ +# MCP server evaluation image for ABEvalFlow Phase 1 (static) + Phase 2 (conformance). +# +# Dedicated image so the two MCP Tekton tasks no longer clone the pipeline repo or +# install tools at runtime. Nothing else pulls this image, so changes here cannot +# affect the shared eval-base pipelines. +# +# Bakes: +# - the MCP eval code and its full import closure (abevalflow.mcp / gates / +# schemas / observability, and scripts.mcp) importable via PYTHONPATH=/opt +# - the scanners/probes it shells out to: gitleaks, semgrep, licensee +# - the minimal Python deps the code imports: pydantic, pyyaml, httpx +# +# Build from repo root (COPY needs repo-root context): +# podman build -t quay.io/ecosystem-appeng/abevalflow-mcp-eval:0.1 \ +# -f containers/mcp-eval/Containerfile . + +# Base tracks :latest (same as Dockerfile.base / agent-eval-harness) so UBI9 CVE +# patches keep flowing. The tools and Python deps that affect runtime behavior are +# pinned below, which is where version drift would actually break the eval code. +FROM registry.access.redhat.com/ubi9/python-312:latest + +# Pinned tool + Python dep versions (avoids the version drift called out in review; +# e.g. an unpinned pydantic v3 would break the schemas the eval code relies on). +ARG GITLEAKS_VERSION=8.30.0 +ARG SEMGREP_VERSION=1.179.0 +ARG PYDANTIC_VERSION=2.13.5 +ARG PYYAML_VERSION=6.0.3 +ARG HTTPX_VERSION=0.28.1 + +USER 0 + +# MCP eval code — only the verified import closure (abevalflow.mcp pulls in +# gates, schemas, observability; scripts.mcp are the Tekton entry points). +COPY abevalflow/__init__.py /opt/abevalflow/__init__.py +COPY abevalflow/schemas.py /opt/abevalflow/schemas.py +COPY abevalflow/mcp /opt/abevalflow/mcp +COPY abevalflow/gates /opt/abevalflow/gates +COPY abevalflow/observability /opt/abevalflow/observability +COPY scripts/__init__.py /opt/scripts/__init__.py +COPY scripts/mcp /opt/scripts/mcp + +ENV PYTHONPATH=/opt + +# Scanners/probes (shelled out to, not imported) + minimal Python deps (imported). +# licensee's native dep (rugged -> libgit2) needs a build toolchain to compile; +# install it, build the gem, then drop the toolchain to keep the image lean. +RUN curl -LsSf \ + "https://github.com/gitleaks/gitleaks/releases/download/v${GITLEAKS_VERSION}/gitleaks_${GITLEAKS_VERSION}_linux_x64.tar.gz" \ + | tar -xz -C /usr/local/bin gitleaks \ + && pip install --no-cache-dir \ + "semgrep==${SEMGREP_VERSION}" \ + "pydantic==${PYDANTIC_VERSION}" \ + "pyyaml==${PYYAML_VERSION}" \ + "httpx==${HTTPX_VERSION}" \ + && dnf install -y ruby ruby-devel gcc gcc-c++ make cmake pkgconf-pkg-config openssl-devel \ + && gem install licensee \ + && dnf remove -y ruby-devel gcc gcc-c++ make cmake pkgconf-pkg-config openssl-devel \ + && dnf clean all + +USER 1001 +WORKDIR /workspace diff --git a/pipeline/tasks/konflux/mcp-phase1-static.yaml b/pipeline/tasks/konflux/mcp-phase1-static.yaml index 193861b..fdaaa33 100644 --- a/pipeline/tasks/konflux/mcp-phase1-static.yaml +++ b/pipeline/tasks/konflux/mcp-phase1-static.yaml @@ -15,6 +15,12 @@ spec: resolves to pass/fail only. Normalized scan JSON and per-gate GateResults are written under reports//. params: + - name: mcp-eval-image + type: string + default: "quay.io/ecosystem-appeng/abevalflow-mcp-eval:0.1" + description: >- + MCP eval image with the scanners (gitleaks, semgrep, licensee) and the + pipeline code (scripts.mcp, abevalflow.mcp) baked in at PYTHONPATH=/opt. - name: source-subdir type: string default: "." @@ -34,79 +40,27 @@ spec: description: >- Space-separated approved SPDX ids overriding the built-in allow-list (e.g. "apache-2.0 mit"). Empty uses DEFAULT_ALLOWED_LICENSES. - - name: pipeline-repo-url - type: string - default: "https://github.com/RHEcosystemAppEng/agentic_eval_flow.git" - description: URL of the Agentic Eval Flow pipeline repository (scanner scripts). - - name: pipeline-repo-revision - type: string - default: "main" - description: Branch or SHA of the pipeline repo to use for scripts. workspaces: - name: source description: Workspace containing the MCP server source to evaluate. results: - name: phase1-passed description: Whether all Phase 1 gates passed (true/false). - - name: secrets-passed - description: Whether the secrets-management gate passed. - - name: no-user-code-passed - description: Whether the no-user-code-execution gate passed. - - name: license-passed - description: Whether the license-compliance gate passed. - name: report-path description: Workspace-relative path to the Phase 1 reports directory. steps: - - name: clone-pipeline-repo - image: registry.access.redhat.com/ubi9/python-311:9.6 - script: | - #!/usr/bin/env bash - set -euo pipefail - PIPELINE_DIR="$(workspaces.source.path)/_pipeline" - if [ -d "$PIPELINE_DIR/.git" ]; then - echo "Pipeline repo already cloned, skipping" - exit 0 - fi - git clone --depth 1 --branch "$(params.pipeline-repo-revision)" \ - "$(params.pipeline-repo-url)" "$PIPELINE_DIR" - echo "Cloned pipeline repo at $(params.pipeline-repo-revision)" - - name: scan - image: registry.access.redhat.com/ubi9/python-311:9.6 - # Tekton steps run as separate containers sharing only the workspaces, not - # the container root filesystem. Tools must be installed in the SAME step - # that runs the scanners, so install and scan live together here. + image: $(params.mcp-eval-image) env: - name: SEMGREP_ENABLE_VERSION_CHECK value: "0" # no phone-home version check; stay offline script: | #!/usr/bin/env bash set -euo pipefail - PIPELINE_DIR="$(workspaces.source.path)/_pipeline" SOURCE_DIR="$(workspaces.source.path)/$(params.source-subdir)" REPORTS_DIR="$(workspaces.source.path)/reports/$(params.submission-name)" mkdir -p "$REPORTS_DIR" - echo "=== Installing gitleaks (secrets) ===" - GITLEAKS_VERSION="8.30.0" - curl -sL \ - "https://github.com/gitleaks/gitleaks/releases/download/v${GITLEAKS_VERSION}/gitleaks_${GITLEAKS_VERSION}_linux_x64.tar.gz" \ - | tar -xz -C /usr/local/bin gitleaks - gitleaks version - - echo "=== Installing semgrep (no-user-code) ===" - pip install --quiet --no-cache-dir semgrep - semgrep --version - - echo "=== Installing licensee (license) ===" - dnf install -y --quiet ruby ruby-devel gcc make redhat-rpm-config 2>&1 | tail -3 - gem install --quiet licensee - licensee version - - cd "$PIPELINE_DIR" - export PYTHONPATH="$PIPELINE_DIR" - pip install --quiet --no-cache-dir pydantic pyyaml - LICENSE_ARGS="" for spdx in $(params.allowed-licenses); do LICENSE_ARGS="$LICENSE_ARGS --allow $spdx" @@ -119,18 +73,13 @@ spec: python -m scripts.mcp.license_scan "$SOURCE_DIR" --reports-dir "$REPORTS_DIR" $LICENSE_ARGS - name: gate - image: registry.access.redhat.com/ubi9/python-311:9.6 + image: $(params.mcp-eval-image) script: | #!/usr/bin/env bash set -euo pipefail - PIPELINE_DIR="$(workspaces.source.path)/_pipeline" REPORTS_DIR="$(workspaces.source.path)/reports/$(params.submission-name)" REPORT_REL="reports/$(params.submission-name)" - cd "$PIPELINE_DIR" - export PYTHONPATH="$PIPELINE_DIR" - pip install --quiet --no-cache-dir pydantic pyyaml - # Never abort before writing results: capture the exit code, emit # Tekton results from the summary, then fail the step if a gate failed. set +e @@ -140,20 +89,8 @@ spec: set -e SUMMARY="$REPORTS_DIR/phase1-summary.json" - emit() { - python3 -c " - import json, sys - data = json.load(open('$SUMMARY')) - gates = {g['gate']: g['passed'] for g in data['gates']} - print(str(gates.get(sys.argv[1], False)).lower(), end='') - " "$1" - } - python3 -c "import json;print(str(json.load(open('$SUMMARY'))['passed']).lower(),end='')" \ > "$(results.phase1-passed.path)" - emit mcp-secrets > "$(results.secrets-passed.path)" - emit mcp-no-user-code > "$(results.no-user-code-passed.path)" - emit mcp-license > "$(results.license-passed.path)" echo -n "$REPORT_REL" > "$(results.report-path.path)" exit $GATE_EXIT diff --git a/pipeline/tasks/konflux/mcp-phase2-conformance.yaml b/pipeline/tasks/konflux/mcp-phase2-conformance.yaml index 47a134f..8ce1855 100644 --- a/pipeline/tasks/konflux/mcp-phase2-conformance.yaml +++ b/pipeline/tasks/konflux/mcp-phase2-conformance.yaml @@ -17,6 +17,12 @@ spec: the server was already deployed and became ready in a prior pipeline step (reached at :8080 by convention). params: + - name: mcp-eval-image + type: string + default: "quay.io/ecosystem-appeng/abevalflow-mcp-eval:0.1" + description: >- + MCP eval image with the probe dependencies (httpx, pydantic, pyyaml) and + the pipeline code (scripts.mcp, abevalflow.mcp) baked in at PYTHONPATH=/opt. - name: server-url type: string description: MCP endpoint URL, e.g. http://:8080/mcp @@ -38,52 +44,23 @@ spec: Workspace-relative path to a Compass facts JSON for the consumed checks. Empty skips the Compass-consume step; those checks then report not_evaluated. (Facts file only; live SoundCheck retrieval is not wired yet.) - - name: pipeline-repo-url - type: string - default: "https://github.com/RHEcosystemAppEng/agentic_eval_flow.git" - description: URL of the Agentic Eval Flow pipeline repository (probe scripts). - - name: pipeline-repo-revision - type: string - default: "main" - description: Branch or SHA of the pipeline repo to use for scripts. workspaces: - name: source - description: Workspace for the pipeline checkout and the reports directory. + description: Workspace for the reports directory. results: - name: phase2-passed description: Whether Phase 2 passed (no check FAILED in block mode). - - name: not-evaluated-count - description: Number of Phase 2 checks reported not_evaluated. - name: report-path description: Workspace-relative path to the Phase 2 reports directory. steps: - - name: clone-pipeline-repo - image: registry.access.redhat.com/ubi9/python-311:9.6 - script: | - #!/usr/bin/env bash - set -euo pipefail - PIPELINE_DIR="$(workspaces.source.path)/_pipeline" - if [ -d "$PIPELINE_DIR/.git" ]; then - echo "Pipeline repo already cloned, skipping" - exit 0 - fi - git clone --depth 1 --branch "$(params.pipeline-repo-revision)" \ - "$(params.pipeline-repo-url)" "$PIPELINE_DIR" - echo "Cloned pipeline repo at $(params.pipeline-repo-revision)" - - name: probe - image: registry.access.redhat.com/ubi9/python-311:9.6 + image: $(params.mcp-eval-image) script: | #!/usr/bin/env bash set -euo pipefail - PIPELINE_DIR="$(workspaces.source.path)/_pipeline" REPORTS_DIR="$(workspaces.source.path)/reports/$(params.submission-name)" mkdir -p "$REPORTS_DIR" - cd "$PIPELINE_DIR" - export PYTHONPATH="$PIPELINE_DIR" - pip install --quiet --no-cache-dir httpx pydantic pyyaml - echo "=== Phase 2 live probe against $(params.server-url) ===" python -m scripts.mcp.phase2_probe \ --server-url "$(params.server-url)" \ @@ -101,18 +78,13 @@ spec: fi - name: gate - image: registry.access.redhat.com/ubi9/python-311:9.6 + image: $(params.mcp-eval-image) script: | #!/usr/bin/env bash set -euo pipefail - PIPELINE_DIR="$(workspaces.source.path)/_pipeline" REPORTS_DIR="$(workspaces.source.path)/reports/$(params.submission-name)" REPORT_REL="reports/$(params.submission-name)" - cd "$PIPELINE_DIR" - export PYTHONPATH="$PIPELINE_DIR" - pip install --quiet --no-cache-dir httpx pydantic pyyaml - # Never abort before writing results: capture the exit code, emit the # Tekton results from the summary, then fail the step if a check failed. set +e @@ -124,8 +96,6 @@ spec: SUMMARY="$REPORTS_DIR/phase2-summary.json" python3 -c "import json;print(str(json.load(open('$SUMMARY'))['passed']).lower(),end='')" \ > "$(results.phase2-passed.path)" - python3 -c "import json;print(json.load(open('$SUMMARY'))['counts']['not_evaluated'],end='')" \ - > "$(results.not-evaluated-count.path)" echo -n "$REPORT_REL" > "$(results.report-path.path)" exit $GATE_EXIT diff --git a/scripts/mcp/__init__.py b/scripts/mcp/__init__.py index 70540c6..a551b4a 100644 --- a/scripts/mcp/__init__.py +++ b/scripts/mcp/__init__.py @@ -1,6 +1,19 @@ -"""Scanner scripts for the MCP Phase 1 static / build-time checks. +"""Execution / I/O layer for the MCP evaluation pipeline. -Each module runs an external, language-agnostic static-analysis tool against an -MCP server repository and normalizes its output into the shared -``{"findings": [...]}`` schema that ``abevalflow.mcp.phase1`` gates read. +These modules shell out to external tools, probe the live server, and read/write +report JSON. The pure pass/fail evaluation logic they feed lives in +``abevalflow.mcp``. + +Tekton entry points (run as ``python -m scripts.mcp.``): + +- ``secrets_scan`` - Phase 1 secrets scan (gitleaks) +- ``no_user_code_scan`` - Phase 1 no-user-code scan (semgrep) +- ``license_scan`` - Phase 1 license scan (licensee) +- ``run_phase1_gates`` - Phase 1 gate evaluation + summary +- ``phase2_probe`` - Phase 2 live conformance probe +- ``compass_fetch`` - Phase 2 Compass-fact consume +- ``run_phase2_gates`` - Phase 2 gate evaluation + summary + +Internal helpers (imported by the entry points, never run directly) are +underscore-prefixed: ``_common``, ``_phase2``, ``_mcp_client``. """ diff --git a/scripts/mcp/mcp_client.py b/scripts/mcp/_mcp_client.py similarity index 100% rename from scripts/mcp/mcp_client.py rename to scripts/mcp/_mcp_client.py diff --git a/scripts/mcp/compass_fetch.py b/scripts/mcp/compass_fetch.py index ecc373f..f45fe88 100644 --- a/scripts/mcp/compass_fetch.py +++ b/scripts/mcp/compass_fetch.py @@ -149,10 +149,19 @@ def main() -> int: parser.add_argument("--facts-file", type=Path, required=True, help="Local JSON file of Compass facts") args = parser.parse_args() + facts: dict[str, Any] = {} if not args.facts_file.is_file(): - logger.error("Facts file not found: %s", args.facts_file) - return 1 - facts = json.loads(args.facts_file.read_text()) + logger.warning("Facts file not found: %s; consumed checks reported not_evaluated.", args.facts_file) + else: + try: + loaded = json.loads(args.facts_file.read_text()) + except (json.JSONDecodeError, OSError) as exc: + logger.warning("Could not read facts file %s (%s); consumed checks reported not_evaluated.", args.facts_file, exc) + else: + if isinstance(loaded, dict): + facts = loaded + else: + logger.warning("Facts file %s is not a JSON object; consumed checks reported not_evaluated.", args.facts_file) outcomes = map_facts(facts) for outcome in outcomes: diff --git a/scripts/mcp/phase2_probe.py b/scripts/mcp/phase2_probe.py index 411564b..4607640 100644 --- a/scripts/mcp/phase2_probe.py +++ b/scripts/mcp/phase2_probe.py @@ -30,7 +30,7 @@ CheckOutcome, write_check_result, ) -from scripts.mcp.mcp_client import MCPClient, MCPConnectionError, RpcResponse +from scripts.mcp._mcp_client import MCPClient, MCPConnectionError, RpcResponse logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") logger = logging.getLogger(__name__) diff --git a/tests/test_mcp_phase2.py b/tests/test_mcp_phase2.py index 11af6d0..204d88c 100644 --- a/tests/test_mcp_phase2.py +++ b/tests/test_mcp_phase2.py @@ -19,7 +19,7 @@ from abevalflow.schemas import GatePolicy from scripts.mcp import compass_fetch, phase2_probe, run_phase2_gates from scripts.mcp._phase2 import STATUS_FAIL, STATUS_NOT_EVALUATED, STATUS_PASS, CheckOutcome, write_check_result -from scripts.mcp.mcp_client import RpcResponse, _parse_sse +from scripts.mcp._mcp_client import RpcResponse, _parse_sse # --------------------------------------------------------------------------- # Helpers @@ -238,12 +238,27 @@ def test_map_facts_oauth_explicit_false_fails_only_for_false(): assert {f["rule_id"] for f in out.findings} == {"oauth-server-match"} -def test_missing_facts_file_fails(monkeypatch, tmp_path): +def test_missing_facts_file_degrades_to_not_evaluated(monkeypatch, tmp_path): monkeypatch.setattr( "sys.argv", ["compass_fetch", "--facts-file", str(tmp_path / "nope.json"), "--reports-dir", str(tmp_path)], ) - assert compass_fetch.main() == 1 + assert compass_fetch.main() == 0 + for check in ("tool-name-rules", "oauth-catalog-match"): + data = json.loads((tmp_path / f"{check}-check.json").read_text()) + assert data["status"] == STATUS_NOT_EVALUATED + + +def test_malformed_facts_file_degrades_to_not_evaluated(monkeypatch, tmp_path): + bad = tmp_path / "bad.json" + bad.write_text("{ not valid json ") + monkeypatch.setattr( + "sys.argv", + ["compass_fetch", "--facts-file", str(bad), "--reports-dir", str(tmp_path)], + ) + assert compass_fetch.main() == 0 + data = json.loads((tmp_path / "tool-name-rules-check.json").read_text()) + assert data["status"] == STATUS_NOT_EVALUATED # --------------------------------------------------------------------------- From 032f4fd3975fceb272bd3b04d55c1743240a28fb Mon Sep 17 00:00:00 2001 From: operetz Date: Tue, 6 Oct 2026 14:59:26 +0300 Subject: [PATCH 9/9] fix ruff lint (line length, import order) --- scripts/mcp/compass_fetch.py | 8 ++++++-- scripts/mcp/phase2_probe.py | 2 +- tests/test_mcp_phase2.py | 2 +- 3 files changed, 8 insertions(+), 4 deletions(-) diff --git a/scripts/mcp/compass_fetch.py b/scripts/mcp/compass_fetch.py index f45fe88..f8ed9ce 100644 --- a/scripts/mcp/compass_fetch.py +++ b/scripts/mcp/compass_fetch.py @@ -156,12 +156,16 @@ def main() -> int: try: loaded = json.loads(args.facts_file.read_text()) except (json.JSONDecodeError, OSError) as exc: - logger.warning("Could not read facts file %s (%s); consumed checks reported not_evaluated.", args.facts_file, exc) + logger.warning( + "Could not read facts file %s (%s); consumed checks reported not_evaluated.", args.facts_file, exc + ) else: if isinstance(loaded, dict): facts = loaded else: - logger.warning("Facts file %s is not a JSON object; consumed checks reported not_evaluated.", args.facts_file) + logger.warning( + "Facts file %s is not a JSON object; consumed checks reported not_evaluated.", args.facts_file + ) outcomes = map_facts(facts) for outcome in outcomes: diff --git a/scripts/mcp/phase2_probe.py b/scripts/mcp/phase2_probe.py index 4607640..3a36fd8 100644 --- a/scripts/mcp/phase2_probe.py +++ b/scripts/mcp/phase2_probe.py @@ -23,6 +23,7 @@ from pathlib import Path from scripts.mcp._common import make_finding +from scripts.mcp._mcp_client import MCPClient, MCPConnectionError, RpcResponse from scripts.mcp._phase2 import ( STATUS_FAIL, STATUS_NOT_EVALUATED, @@ -30,7 +31,6 @@ CheckOutcome, write_check_result, ) -from scripts.mcp._mcp_client import MCPClient, MCPConnectionError, RpcResponse logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") logger = logging.getLogger(__name__) diff --git a/tests/test_mcp_phase2.py b/tests/test_mcp_phase2.py index 204d88c..fe00377 100644 --- a/tests/test_mcp_phase2.py +++ b/tests/test_mcp_phase2.py @@ -18,8 +18,8 @@ from abevalflow.mcp.phase2.base import Phase2Gate from abevalflow.schemas import GatePolicy from scripts.mcp import compass_fetch, phase2_probe, run_phase2_gates -from scripts.mcp._phase2 import STATUS_FAIL, STATUS_NOT_EVALUATED, STATUS_PASS, CheckOutcome, write_check_result from scripts.mcp._mcp_client import RpcResponse, _parse_sse +from scripts.mcp._phase2 import STATUS_FAIL, STATUS_NOT_EVALUATED, STATUS_PASS, CheckOutcome, write_check_result # --------------------------------------------------------------------------- # Helpers