Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 12 additions & 0 deletions abevalflow/mcp/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,12 @@
"""MCP server evaluation pipeline.

Isolated from the skill evaluation pipeline (see ADR: Evaluation Strategy for
MCP Servers as Standalone Software Components, Approach 4). Evaluates an MCP
server as a black box across three sequenced phases:

- Phase 1 - static / build-time analysis (no running server).
- Phase 2 - deterministic contract/conformance testing (live, no LLM).
- Phase 3 - behavioral testing (mcpchecker, conditional on a task suite).

Only Phase 1 is implemented here so far.
"""
67 changes: 67 additions & 0 deletions abevalflow/mcp/phase1/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,67 @@
"""MCP Phase 1 - static / build-time analysis.

Runs three static checks against the MCP server's built artifact/source, before
any deployment and without running the server (ADR Approach 4, Phase 1):

- Secrets management (SecretsGate)
- No user-provided code execution (NoUserCodeGate)
- License compliance (LicenseGate)

Each gate subclasses SecurityGate and reads a normalized ``{"findings": [...]}``
scan JSON emitted by the corresponding scanner in ``scripts/mcp/``. Phase 1 is
kept isolated from the skill pipeline's global security-gate registry
(``abevalflow.gates.security``): these gates are instantiated only via this
module's :func:`run_phase1`, so onboarding an MCP server does not pull skill
scanners and vice versa.

Phase 1 always executes, independent of runtime outcomes, so its checks resolve
to pass/fail only (never the not-evaluated state used for Phase 2/3).
"""

from __future__ import annotations

from pathlib import Path

from abevalflow.gates.base import GateResult
from abevalflow.gates.security.base import SecurityGate
from abevalflow.mcp.phase1.license import DEFAULT_ALLOWED_LICENSES, LicenseGate
from abevalflow.mcp.phase1.no_user_code import NoUserCodeGate
from abevalflow.mcp.phase1.secrets import SecretsGate
from abevalflow.schemas import GatePolicy

# Ordered list of the Phase 1 gate classes, in execution order.
PHASE1_GATES: tuple[type[SecurityGate], ...] = (
SecretsGate,
NoUserCodeGate,
LicenseGate,
)


def get_phase1_gates() -> list[SecurityGate]:
"""Instantiate all Phase 1 gates, in order."""
return [gate_cls() for gate_cls in PHASE1_GATES]


def run_phase1(reports_dir: Path, policy: GatePolicy) -> list[GateResult]:
"""Evaluate all Phase 1 gates against the scan reports.

Args:
reports_dir: Path to ``reports/{submission-name}/`` holding the
normalized scan JSON files written by ``scripts/mcp/*``.
policy: Gate policy controlling mode (disabled/warn/block) per gate.

Returns:
One GateResult per Phase 1 gate, in :data:`PHASE1_GATES` order.
"""
return [gate.evaluate(reports_dir, policy) for gate in get_phase1_gates()]


__all__ = [
"DEFAULT_ALLOWED_LICENSES",
"PHASE1_GATES",
"LicenseGate",
"NoUserCodeGate",
"SecretsGate",
"get_phase1_gates",
"run_phase1",
]
46 changes: 46 additions & 0 deletions abevalflow/mcp/phase1/license.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,46 @@
"""Phase 1 license-compliance gate.

Reads license-scan.json produced by scripts/mcp/license_scan.py (a licensee run
compared against the approved-license allow-list) and converts it into a
standardized GateResult. Fails in block mode when the top-level declared license
is missing or not in the allow-list (emitted as a HIGH finding by the scanner).

ADR Phase 1: "License compliance - verify the MCP server's top-level declared
license against a list of approved/supported licenses ... Scoped to the
top-level declared license only; transitive dependency license scanning is out
of scope."
"""

from __future__ import annotations

from pathlib import Path

from abevalflow.gates.base import GateResult
from abevalflow.gates.security.base import SecurityGate
from abevalflow.observability.decorators import timed_gate
from abevalflow.schemas import GatePolicy

# Approved SPDX license identifiers for the top-level declared license.
# Compared case-insensitively (see license_scan.py::evaluate_licenses).
DEFAULT_ALLOWED_LICENSES: tuple[str, ...] = (
"apache-2.0",
"mit",
"bsd-2-clause",
"bsd-3-clause",
)


class LicenseGate(SecurityGate):
"""Top-level declared-license allow-list gate (licensee-backed)."""

name = "mcp-license"
scan_filename = "license-scan.json"

@timed_gate
def evaluate(
self,
reports_dir: Path,
policy: GatePolicy,
) -> GateResult:
"""Evaluate the license allow-list scan."""
return self.evaluate_scan_json(reports_dir, policy)
36 changes: 36 additions & 0 deletions abevalflow/mcp/phase1/no_user_code.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,36 @@
"""Phase 1 no-user-code-execution gate.

Reads no-user-code-scan.json produced by scripts/mcp/no_user_code_scan.py (a
normalized semgrep run) and converts it into a standardized GateResult. Fails
in block mode when any HIGH/CRITICAL dynamic/arbitrary-execution pattern is
present (e.g. eval/exec, os.system, subprocess shell=True, Function(),
Runtime.exec).

ADR Phase 1: "Does not execute user-provided code - static analysis of the
artifact for dynamic/arbitrary code execution."
"""

from __future__ import annotations

from pathlib import Path

from abevalflow.gates.base import GateResult
from abevalflow.gates.security.base import SecurityGate
from abevalflow.observability.decorators import timed_gate
from abevalflow.schemas import GatePolicy


class NoUserCodeGate(SecurityGate):
"""Static dynamic-execution pattern scan gate (semgrep-backed)."""

name = "mcp-no-user-code"
scan_filename = "no-user-code-scan.json"

@timed_gate
def evaluate(
self,
reports_dir: Path,
policy: GatePolicy,
) -> GateResult:
"""Evaluate the normalized semgrep dynamic-execution scan."""
return self.evaluate_scan_json(reports_dir, policy)
34 changes: 34 additions & 0 deletions abevalflow/mcp/phase1/secrets.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
"""Phase 1 secrets-management gate.

Reads secrets-scan.json produced by scripts/mcp/secrets_scan.py (a normalized
gitleaks run) and converts it into a standardized GateResult. Fails in block
mode when any HIGH/CRITICAL hardcoded-credential finding is present.

ADR Phase 1: "Secrets management - static scan of the artifact for hardcoded
credentials."
"""

from __future__ import annotations

from pathlib import Path

from abevalflow.gates.base import GateResult
from abevalflow.gates.security.base import SecurityGate
from abevalflow.observability.decorators import timed_gate
from abevalflow.schemas import GatePolicy


class SecretsGate(SecurityGate):
"""Static hardcoded-credential scan gate (gitleaks-backed)."""

name = "mcp-secrets"
scan_filename = "secrets-scan.json"

@timed_gate
def evaluate(
self,
reports_dir: Path,
policy: GatePolicy,
) -> GateResult:
"""Evaluate the normalized gitleaks secrets scan."""
return self.evaluate_scan_json(reports_dir, policy)
60 changes: 60 additions & 0 deletions abevalflow/mcp/phase2/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,60 @@
"""MCP evaluation Phase 2 - deterministic contract/conformance (live, no LLM).

Phase 2 is black-box tested against the deployed, running MCP server (ADR
Approach 4). It is deterministic and uses NO LLM. Results follow the three-state
model (pass / fail / not_evaluated).

Two sources feed the same set of gates:
* **Live probes** (scripts/mcp/phase2_probe.py) - checks that need a running
server: protocol compliance, schema conformance, tool annotations, HTTP
support, rate limiting, response-size limit, mandatory timeouts, OpenAPI
conformance.
* **Compass-consumed** (scripts/mcp/compass_fetch.py) - the Compass tier-1
automated facts consumed rather than re-probed: tool-name rules and
OAuth-matches-catalog.

Like Phase 1, this package is kept isolated from the skill pipeline: the gates
are instantiated only here via ``run_phase2`` and are never registered in the
global security-gate registry.
"""

from __future__ import annotations

from pathlib import Path

from abevalflow.gates.base import GateResult
from abevalflow.mcp.phase2.base import Phase2Gate
from abevalflow.schemas import GatePolicy

# Live probe checks (need a running server).
LIVE_CHECKS: tuple[str, ...] = (
"http-support",
"protocol-compliance",
"schema-conformance",
"tool-annotations",
"rate-limiting",
"response-size-limit",
"mandatory-timeouts",
"openapi-conformance",
)

# Checks consumed from existing Compass facts (not re-probed).
COMPASS_CHECKS: tuple[str, ...] = (
"tool-name-rules",
"oauth-catalog-match",
)

PHASE2_CHECKS: tuple[str, ...] = LIVE_CHECKS + COMPASS_CHECKS


def get_phase2_gates() -> list[Phase2Gate]:
"""Instantiate one gate per Phase 2 check, in order."""
return [Phase2Gate(check) for check in PHASE2_CHECKS]


def run_phase2(reports_dir: Path, policy: GatePolicy) -> list[GateResult]:
"""Evaluate every Phase 2 gate against the check result files in ``reports_dir``."""
return [gate.evaluate(reports_dir, policy) for gate in get_phase2_gates()]


__all__ = ["COMPASS_CHECKS", "LIVE_CHECKS", "PHASE2_CHECKS", "Phase2Gate", "get_phase2_gates", "run_phase2"]
Loading
Loading