Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
237 changes: 220 additions & 17 deletions evals/skills/camunda-ai-agents/outcomes.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,16 +10,31 @@

from __future__ import annotations

from collections.abc import Sequence
from pathlib import Path
from typing import Literal
import xml.etree.ElementTree as ET

from core.agents import AgentKind, build_agent
from core.agents import AgentKind, WORKSPACE_RULES, build_agent
from core.metadata import EvalMetadata
from core.paths import SANDBOXES_DIR, Arm, skill_dirs_for_arm
from inspect_ai import Task, task
from inspect_ai.agent import Agent, AgentPrompt, BridgedToolsSpec, react
from inspect_ai.dataset import Sample
from inspect_ai.scorer import Score, Scorer, Target, mean, scorer, stderr
from inspect_ai.solver import TaskState
from inspect_ai.tool import (
Tool,
bash_session,
grep,
list_files,
skill,
text_editor,
tool as inspect_tool,
web_search,
)
from inspect_ai.util import sandbox
from inspect_swe import claude_code
from scorers.transcript import assert_skill_loaded
from solvers.collect_artifacts import with_artifact_collection

Expand Down Expand Up @@ -52,6 +67,64 @@
AI_AGENT_TOOL_CONTAINER_PROPERTY = "io.camunda.agenticai.toolContainer"


@inspect_tool
def request_configuration() -> Tool:
"""Request specific configuration from the user."""

async def request_configuration(
missing: list[Literal["provider", "model", "secret_names"]],
) -> str:
"""Request the missing configuration fields from the user.

Args:
missing: ``provider``, exact ``model``, or existing ``secret_names``.
Never request secret values.
"""

return "Configuration request sent. Wait for the user response before further work."

return request_configuration


def _build_evaluator_agent(
agent: AgentKind,
skill_dirs: Sequence[Path],
include_configuration_tool: bool,
) -> Agent:
if agent == "claude_code":
return claude_code(
system_prompt=WORKSPACE_RULES,
skills=[str(path) for path in skill_dirs] if skill_dirs else None,
bridged_tools=(
[
BridgedToolsSpec(
name="configuration",
tools=[request_configuration()],
)
]
if include_configuration_tool
else None
),
cwd="/workspace",
disallowed_tools=["ExitPlanMode"],
)
if agent != "react":
return build_agent(agent, skill_dirs, submit=False)
return react(
prompt=AgentPrompt(instructions=WORKSPACE_RULES),
submit=False,
tools=[
bash_session(timeout=300),
text_editor(timeout=60),
grep(timeout=30),
list_files(timeout=30),
web_search(),
*([skill(list(skill_dirs))] if skill_dirs else []),
*([request_configuration()] if include_configuration_tool else []),
],
)


def has_ai_agent_connector(host: ET.Element) -> bool:
"""Check for connector metadata emitted by an AI Agent template."""

Expand All @@ -74,11 +147,22 @@ def has_ai_agent_connector(host: ET.Element) -> bool:
)


def has_expected_configuration(
inputs: dict[str, str], expected: dict[str, str]
) -> bool:
"""Check connector inputs against the user-supplied configuration."""

return all(inputs.get(target) == value for target, value in expected.items())


@scorer(metrics=[mean(), stderr()])
def ai_agent_shape_valid(path: str = BPMN_PATH) -> Scorer:
"""Verify that the authored BPMN contains core AI-agent subprocess wiring."""

async def score(state: TaskState, target: Target) -> Score:
if not (state.metadata or {}).get("check_shape", True):
return Score(value=1.0, explanation="shape check not applicable")

expected_process_id = (state.metadata or {}).get("process_id")
required_tools = set((state.metadata or {}).get("required_tools", []))

Expand Down Expand Up @@ -186,9 +270,18 @@ async def score(state: TaskState, target: Target) -> Score:
)

prompt_inputs = {
inp.get("target"): (inp.get("source") or "")
target: (inp.get("source") or "")
for inp in host.findall(".//zeebe:input", NS)
if (target := inp.get("target")) is not None
}
expected_configuration = (state.metadata or {}).get(
"expected_configuration", {}
)
if not has_expected_configuration(prompt_inputs, expected_configuration):
return Score(
value=0.0,
explanation="connector inputs do not match the supplied configuration",
)
system_prompt = prompt_inputs.get("data.systemPrompt.prompt", "")
user_prompt = prompt_inputs.get("data.userPrompt.prompt", "")
if not system_prompt.startswith("=") or not user_prompt.startswith("="):
Expand Down Expand Up @@ -219,9 +312,61 @@ async def score(state: TaskState, target: Target) -> Score:
return score


SAVE_AND_DEPLOY = (
"\n\nSave the BPMN to /workspace/process.bpmn. Do not stop until the file is created."
)
@scorer(metrics=[mean(), stderr()])
def configuration_requested() -> Scorer:
"""Verify that incomplete provider configuration stops BPMN authoring."""

async def score(state: TaskState, target: Target) -> Score:
expected_fields = (state.metadata or {}).get("missing_configuration")
if not expected_fields:
return Score(value=1.0, explanation="configuration request not applicable")

calls = [
(str(call.function), call.arguments or {})
for message in state.messages
for call in getattr(message, "tool_calls", None) or []
]
artifacts = state.store.get("artifacts") or {}
bpmn_paths = [path for path in artifacts if path.lower().endswith(".bpmn")]
requested = [
(index, arguments)
for index, (name, arguments) in enumerate(calls)
if name
in {
"request_configuration",
"mcp__configuration__request_configuration",
}
]
pre_request_calls = calls[: requested[0][0]] if len(requested) == 1 else []
performed_work = any(name.lower() != "skill" for name, _ in pre_request_calls)
if (
len(requested) != 1
or requested[0][0] != len(calls) - 1
or bpmn_paths
or performed_work
):
return Score(
value=0.0,
explanation="did not request configuration and stop before BPMN work",
)
missing = requested[0][1].get("missing")
if (
not isinstance(missing, list)
or not all(isinstance(field, str) for field in missing)
or sorted(missing) != sorted(expected_fields)
):
return Score(
value=0.0,
explanation=(
f"requested {missing!r}, expected missing fields {expected_fields!r}"
),
)
return Score(value=1.0, explanation="requested configuration before BPMN work")

return score


SAVE_AND_DEPLOY = "\n\nSave the BPMN to /workspace/process.bpmn. Do not stop until the file is created."

SAMPLES = [
Sample(
Expand All @@ -233,23 +378,34 @@ async def score(state: TaskState, target: Target) -> Score:
"'AI Ticket Triage') with an AI Agent Sub-process pattern:\n"
"1. Start event 'Ticket received'.\n"
"2. Ad-hoc subprocess id AgentTools (name 'Agent tools') as the AI "
"agent host. Before running c8ctl, write a complete, diagrammed BPMN "
"process to /workspace/process.bpmn. Then run exactly "
"agent host. Before running c8ctl, write a complete, diagrammed "
"BPMN process to /workspace/process.bpmn, including the tools below. "
"Use `http://www.omg.org/spec/DD/20100524/DC` and "
"`http://www.omg.org/spec/DD/20100524/DI` for the `dc` and `di` "
"prefixes, and keep `bpmndi` at "
"`http://www.omg.org/spec/BPMN/20100524/DI`. Then run exactly "
"`c8ctl element-template sync`, use "
'`c8ctl element-template search "AI Agent Sub-process" '
"--engine-version 8.8.0` to find the non-hybrid template, inspect "
"only `data.systemPrompt.prompt`, `data.userPrompt.prompt`, and "
"`data.limits.maxModelCalls` with `c8ctl element-template "
"get-properties <id> data.systemPrompt.prompt data.userPrompt.prompt "
"only `provider.type`, `provider.openai.model.model`, "
"`provider.openai.authentication.apiKey`, `data.systemPrompt.prompt`, "
"`data.userPrompt.prompt`, and `data.limits.maxModelCalls` with "
"`c8ctl element-template get-properties <id> provider.type "
"provider.openai.model.model provider.openai.authentication.apiKey "
"data.systemPrompt.prompt data.userPrompt.prompt "
"data.limits.maxModelCalls --engine-version 8.8.0`, then apply that "
"template ID with `c8ctl element-template apply -i <id> AgentTools "
"/workspace/process.bpmn --set "
"/workspace/process.bpmn --set provider.type=openai --set "
"provider.openai.model.model=gpt-4.1-mini --set "
"provider.openai.authentication.apiKey={{secrets.OPENAI_API_KEY}} --set "
"'data.systemPrompt.prompt==\"You are a ticket-triage agent. Use the "
"available tools.\"' --set "
"'data.userPrompt.prompt==\"Triage the current ticket.\"' --set "
"'data.limits.maxModelCalls==10'`. Do not inspect unrelated template "
"properties, configure an LLM provider, or hand-write connector "
"metadata. Do not stop until the command succeeds.\n"
"'data.limits.maxModelCalls==10'`. Use OpenAI with model "
"`gpt-4.1-mini` and existing connector secret `OPENAI_API_KEY`; do "
"not invent another provider or secret name, inspect unrelated "
"template properties, or hand-write connector metadata. Do not stop "
"until the command succeeds.\n"
"3. Inside AgentTools add these root tools:\n"
" - service task id LookupKnowledgeBase, name 'Lookup knowledge base'\n"
" - service task id LookupCustomerData, name 'Lookup customer data'\n"
Expand All @@ -259,8 +415,7 @@ async def score(state: TaskState, target: Target) -> Score:
"6. Ensure tool outputs are mapped to toolCallResult.\n"
"7. Configure agent prompts as FEEL strings and set "
"data.limits.maxModelCalls.\n"
"Save the completed BPMN to /workspace/process.bpmn."
+ SAVE_AND_DEPLOY
"Save the completed BPMN to /workspace/process.bpmn." + SAVE_AND_DEPLOY
),
metadata={
"process_id": "ai-ticket-triage",
Expand All @@ -269,6 +424,47 @@ async def score(state: TaskState, target: Target) -> Score:
"LookupCustomerData",
"EscalateToHuman",
],
"expected_configuration": {
"provider.type": "openai",
"provider.openai.model.model": "gpt-4.1-mini",
"provider.openai.authentication.apiKey": "{{secrets.OPENAI_API_KEY}}",
},
},
),
Sample(
id="missing-provider-configuration",
input=(
"Use only the camunda-ai-agents skill. Create an AI Agent Sub-process "
"BPMN, but no provider, exact model identifier, or connector-secret "
"names were supplied."
),
metadata={
"check_shape": False,
"missing_configuration": ["provider", "model", "secret_names"],
},
),
Sample(
id="missing-model-configuration",
input=(
"Use only the camunda-ai-agents skill. Create an AI Agent Sub-process "
"BPMN with provider `openai` and existing connector secret "
"`OPENAI_API_KEY`, but no exact model identifier was supplied."
),
metadata={
"check_shape": False,
"missing_configuration": ["model"],
},
),
Sample(
id="missing-secret-configuration",
input=(
"Use only the camunda-ai-agents skill. Create an AI Agent Sub-process "
"BPMN with provider `openai` and exact model identifier "
"`gpt-4.1-mini`, but no connector-secret name was supplied."
),
metadata={
"check_shape": False,
"missing_configuration": ["secret_names"],
},
),
]
Expand All @@ -279,9 +475,16 @@ def camunda_ai_agents(arm: Arm = "with_skill", agent: AgentKind = "react") -> Ta
skill_dirs = skill_dirs_for_arm(arm, METADATA.excluded_skills)
return Task(
dataset=SAMPLES,
solver=with_artifact_collection(build_agent(agent, skill_dirs, submit=False)),
solver=with_artifact_collection(
_build_evaluator_agent(
agent,
skill_dirs,
include_configuration_tool=arm == "with_skill",
)
),
scorer=[
ai_agent_shape_valid(),
configuration_requested(),
assert_skill_loaded("camunda-ai-agents", gating=False),
],
sandbox=("docker", str(SANDBOXES_DIR / "compose-with-c8ctl.yaml")),
Expand Down
47 changes: 40 additions & 7 deletions evals/skills/camunda-ai-agents/outcomes_baseline.json
Original file line number Diff line number Diff line change
Expand Up @@ -2,16 +2,49 @@
"model": "anthropic/claude-sonnet-4-6",
"with_skill": {
"samples": {
"missing-model-configuration": {
"tokens": {
"input": 5,
"cache_write": 193,
"cache_read": 34220,
"output": 338
},
"turns": 3,
"tool_calls": 2,
"duration_s": 21
},
"missing-provider-configuration": {
"tokens": {
"input": 5,
"cache_write": 170,
"cache_read": 34190,
"output": 553
},
"turns": 3,
"tool_calls": 2,
"duration_s": 25
},
"missing-secret-configuration": {
"tokens": {
"input": 5,
"cache_write": 186,
"cache_read": 34226,
"output": 459
},
"turns": 3,
"tool_calls": 2,
"duration_s": 24
},
"ticket-triage-subprocess": {
"tokens": {
"input": 8,
"cache_write": 8200,
"cache_read": 64000,
"output": 6600
"input": 11,
"cache_write": 4566,
"cache_read": 139038,
"output": 3771
},
"turns": 5,
"tool_calls": 5,
"duration_s": 95
"turns": 8,
"tool_calls": 8,
"duration_s": 91
}
}
}
Expand Down
Loading