diff --git a/evals/skills/camunda-ai-agents/outcomes.py b/evals/skills/camunda-ai-agents/outcomes.py index 56d237ce..ebb438d4 100644 --- a/evals/skills/camunda-ai-agents/outcomes.py +++ b/evals/skills/camunda-ai-agents/outcomes.py @@ -10,16 +10,31 @@ from __future__ import annotations +from collections.abc import Sequence +from pathlib import Path +from typing import Literal import xml.etree.ElementTree as ET -from core.agents import AgentKind, build_agent +from core.agents import AgentKind, WORKSPACE_RULES, build_agent from core.metadata import EvalMetadata from core.paths import SANDBOXES_DIR, Arm, skill_dirs_for_arm from inspect_ai import Task, task +from inspect_ai.agent import Agent, AgentPrompt, BridgedToolsSpec, react from inspect_ai.dataset import Sample from inspect_ai.scorer import Score, Scorer, Target, mean, scorer, stderr from inspect_ai.solver import TaskState +from inspect_ai.tool import ( + Tool, + bash_session, + grep, + list_files, + skill, + text_editor, + tool as inspect_tool, + web_search, +) from inspect_ai.util import sandbox +from inspect_swe import claude_code from scorers.transcript import assert_skill_loaded from solvers.collect_artifacts import with_artifact_collection @@ -52,6 +67,64 @@ AI_AGENT_TOOL_CONTAINER_PROPERTY = "io.camunda.agenticai.toolContainer" +@inspect_tool +def request_configuration() -> Tool: + """Request specific configuration from the user.""" + + async def request_configuration( + missing: list[Literal["provider", "model", "secret_names"]], + ) -> str: + """Request the missing configuration fields from the user. + + Args: + missing: ``provider``, exact ``model``, or existing ``secret_names``. + Never request secret values. + """ + + return "Configuration request sent. Wait for the user response before further work." + + return request_configuration + + +def _build_evaluator_agent( + agent: AgentKind, + skill_dirs: Sequence[Path], + include_configuration_tool: bool, +) -> Agent: + if agent == "claude_code": + return claude_code( + system_prompt=WORKSPACE_RULES, + skills=[str(path) for path in skill_dirs] if skill_dirs else None, + bridged_tools=( + [ + BridgedToolsSpec( + name="configuration", + tools=[request_configuration()], + ) + ] + if include_configuration_tool + else None + ), + cwd="/workspace", + disallowed_tools=["ExitPlanMode"], + ) + if agent != "react": + return build_agent(agent, skill_dirs, submit=False) + return react( + prompt=AgentPrompt(instructions=WORKSPACE_RULES), + submit=False, + tools=[ + bash_session(timeout=300), + text_editor(timeout=60), + grep(timeout=30), + list_files(timeout=30), + web_search(), + *([skill(list(skill_dirs))] if skill_dirs else []), + *([request_configuration()] if include_configuration_tool else []), + ], + ) + + def has_ai_agent_connector(host: ET.Element) -> bool: """Check for connector metadata emitted by an AI Agent template.""" @@ -74,11 +147,22 @@ def has_ai_agent_connector(host: ET.Element) -> bool: ) +def has_expected_configuration( + inputs: dict[str, str], expected: dict[str, str] +) -> bool: + """Check connector inputs against the user-supplied configuration.""" + + return all(inputs.get(target) == value for target, value in expected.items()) + + @scorer(metrics=[mean(), stderr()]) def ai_agent_shape_valid(path: str = BPMN_PATH) -> Scorer: """Verify that the authored BPMN contains core AI-agent subprocess wiring.""" async def score(state: TaskState, target: Target) -> Score: + if not (state.metadata or {}).get("check_shape", True): + return Score(value=1.0, explanation="shape check not applicable") + expected_process_id = (state.metadata or {}).get("process_id") required_tools = set((state.metadata or {}).get("required_tools", [])) @@ -186,9 +270,18 @@ async def score(state: TaskState, target: Target) -> Score: ) prompt_inputs = { - inp.get("target"): (inp.get("source") or "") + target: (inp.get("source") or "") for inp in host.findall(".//zeebe:input", NS) + if (target := inp.get("target")) is not None } + expected_configuration = (state.metadata or {}).get( + "expected_configuration", {} + ) + if not has_expected_configuration(prompt_inputs, expected_configuration): + return Score( + value=0.0, + explanation="connector inputs do not match the supplied configuration", + ) system_prompt = prompt_inputs.get("data.systemPrompt.prompt", "") user_prompt = prompt_inputs.get("data.userPrompt.prompt", "") if not system_prompt.startswith("=") or not user_prompt.startswith("="): @@ -219,9 +312,61 @@ async def score(state: TaskState, target: Target) -> Score: return score -SAVE_AND_DEPLOY = ( - "\n\nSave the BPMN to /workspace/process.bpmn. Do not stop until the file is created." -) +@scorer(metrics=[mean(), stderr()]) +def configuration_requested() -> Scorer: + """Verify that incomplete provider configuration stops BPMN authoring.""" + + async def score(state: TaskState, target: Target) -> Score: + expected_fields = (state.metadata or {}).get("missing_configuration") + if not expected_fields: + return Score(value=1.0, explanation="configuration request not applicable") + + calls = [ + (str(call.function), call.arguments or {}) + for message in state.messages + for call in getattr(message, "tool_calls", None) or [] + ] + artifacts = state.store.get("artifacts") or {} + bpmn_paths = [path for path in artifacts if path.lower().endswith(".bpmn")] + requested = [ + (index, arguments) + for index, (name, arguments) in enumerate(calls) + if name + in { + "request_configuration", + "mcp__configuration__request_configuration", + } + ] + pre_request_calls = calls[: requested[0][0]] if len(requested) == 1 else [] + performed_work = any(name.lower() != "skill" for name, _ in pre_request_calls) + if ( + len(requested) != 1 + or requested[0][0] != len(calls) - 1 + or bpmn_paths + or performed_work + ): + return Score( + value=0.0, + explanation="did not request configuration and stop before BPMN work", + ) + missing = requested[0][1].get("missing") + if ( + not isinstance(missing, list) + or not all(isinstance(field, str) for field in missing) + or sorted(missing) != sorted(expected_fields) + ): + return Score( + value=0.0, + explanation=( + f"requested {missing!r}, expected missing fields {expected_fields!r}" + ), + ) + return Score(value=1.0, explanation="requested configuration before BPMN work") + + return score + + +SAVE_AND_DEPLOY = "\n\nSave the BPMN to /workspace/process.bpmn. Do not stop until the file is created." SAMPLES = [ Sample( @@ -233,23 +378,34 @@ async def score(state: TaskState, target: Target) -> Score: "'AI Ticket Triage') with an AI Agent Sub-process pattern:\n" "1. Start event 'Ticket received'.\n" "2. Ad-hoc subprocess id AgentTools (name 'Agent tools') as the AI " - "agent host. Before running c8ctl, write a complete, diagrammed BPMN " - "process to /workspace/process.bpmn. Then run exactly " + "agent host. Before running c8ctl, write a complete, diagrammed " + "BPMN process to /workspace/process.bpmn, including the tools below. " + "Use `http://www.omg.org/spec/DD/20100524/DC` and " + "`http://www.omg.org/spec/DD/20100524/DI` for the `dc` and `di` " + "prefixes, and keep `bpmndi` at " + "`http://www.omg.org/spec/BPMN/20100524/DI`. Then run exactly " "`c8ctl element-template sync`, use " '`c8ctl element-template search "AI Agent Sub-process" ' "--engine-version 8.8.0` to find the non-hybrid template, inspect " - "only `data.systemPrompt.prompt`, `data.userPrompt.prompt`, and " - "`data.limits.maxModelCalls` with `c8ctl element-template " - "get-properties data.systemPrompt.prompt data.userPrompt.prompt " + "only `provider.type`, `provider.openai.model.model`, " + "`provider.openai.authentication.apiKey`, `data.systemPrompt.prompt`, " + "`data.userPrompt.prompt`, and `data.limits.maxModelCalls` with " + "`c8ctl element-template get-properties provider.type " + "provider.openai.model.model provider.openai.authentication.apiKey " + "data.systemPrompt.prompt data.userPrompt.prompt " "data.limits.maxModelCalls --engine-version 8.8.0`, then apply that " "template ID with `c8ctl element-template apply -i AgentTools " - "/workspace/process.bpmn --set " + "/workspace/process.bpmn --set provider.type=openai --set " + "provider.openai.model.model=gpt-4.1-mini --set " + "provider.openai.authentication.apiKey={{secrets.OPENAI_API_KEY}} --set " "'data.systemPrompt.prompt==\"You are a ticket-triage agent. Use the " "available tools.\"' --set " "'data.userPrompt.prompt==\"Triage the current ticket.\"' --set " - "'data.limits.maxModelCalls==10'`. Do not inspect unrelated template " - "properties, configure an LLM provider, or hand-write connector " - "metadata. Do not stop until the command succeeds.\n" + "'data.limits.maxModelCalls==10'`. Use OpenAI with model " + "`gpt-4.1-mini` and existing connector secret `OPENAI_API_KEY`; do " + "not invent another provider or secret name, inspect unrelated " + "template properties, or hand-write connector metadata. Do not stop " + "until the command succeeds.\n" "3. Inside AgentTools add these root tools:\n" " - service task id LookupKnowledgeBase, name 'Lookup knowledge base'\n" " - service task id LookupCustomerData, name 'Lookup customer data'\n" @@ -259,8 +415,7 @@ async def score(state: TaskState, target: Target) -> Score: "6. Ensure tool outputs are mapped to toolCallResult.\n" "7. Configure agent prompts as FEEL strings and set " "data.limits.maxModelCalls.\n" - "Save the completed BPMN to /workspace/process.bpmn." - + SAVE_AND_DEPLOY + "Save the completed BPMN to /workspace/process.bpmn." + SAVE_AND_DEPLOY ), metadata={ "process_id": "ai-ticket-triage", @@ -269,6 +424,47 @@ async def score(state: TaskState, target: Target) -> Score: "LookupCustomerData", "EscalateToHuman", ], + "expected_configuration": { + "provider.type": "openai", + "provider.openai.model.model": "gpt-4.1-mini", + "provider.openai.authentication.apiKey": "{{secrets.OPENAI_API_KEY}}", + }, + }, + ), + Sample( + id="missing-provider-configuration", + input=( + "Use only the camunda-ai-agents skill. Create an AI Agent Sub-process " + "BPMN, but no provider, exact model identifier, or connector-secret " + "names were supplied." + ), + metadata={ + "check_shape": False, + "missing_configuration": ["provider", "model", "secret_names"], + }, + ), + Sample( + id="missing-model-configuration", + input=( + "Use only the camunda-ai-agents skill. Create an AI Agent Sub-process " + "BPMN with provider `openai` and existing connector secret " + "`OPENAI_API_KEY`, but no exact model identifier was supplied." + ), + metadata={ + "check_shape": False, + "missing_configuration": ["model"], + }, + ), + Sample( + id="missing-secret-configuration", + input=( + "Use only the camunda-ai-agents skill. Create an AI Agent Sub-process " + "BPMN with provider `openai` and exact model identifier " + "`gpt-4.1-mini`, but no connector-secret name was supplied." + ), + metadata={ + "check_shape": False, + "missing_configuration": ["secret_names"], }, ), ] @@ -279,9 +475,16 @@ def camunda_ai_agents(arm: Arm = "with_skill", agent: AgentKind = "react") -> Ta skill_dirs = skill_dirs_for_arm(arm, METADATA.excluded_skills) return Task( dataset=SAMPLES, - solver=with_artifact_collection(build_agent(agent, skill_dirs, submit=False)), + solver=with_artifact_collection( + _build_evaluator_agent( + agent, + skill_dirs, + include_configuration_tool=arm == "with_skill", + ) + ), scorer=[ ai_agent_shape_valid(), + configuration_requested(), assert_skill_loaded("camunda-ai-agents", gating=False), ], sandbox=("docker", str(SANDBOXES_DIR / "compose-with-c8ctl.yaml")), diff --git a/evals/skills/camunda-ai-agents/outcomes_baseline.json b/evals/skills/camunda-ai-agents/outcomes_baseline.json index 5328ca27..97fd9d66 100644 --- a/evals/skills/camunda-ai-agents/outcomes_baseline.json +++ b/evals/skills/camunda-ai-agents/outcomes_baseline.json @@ -2,16 +2,49 @@ "model": "anthropic/claude-sonnet-4-6", "with_skill": { "samples": { + "missing-model-configuration": { + "tokens": { + "input": 5, + "cache_write": 193, + "cache_read": 34220, + "output": 338 + }, + "turns": 3, + "tool_calls": 2, + "duration_s": 21 + }, + "missing-provider-configuration": { + "tokens": { + "input": 5, + "cache_write": 170, + "cache_read": 34190, + "output": 553 + }, + "turns": 3, + "tool_calls": 2, + "duration_s": 25 + }, + "missing-secret-configuration": { + "tokens": { + "input": 5, + "cache_write": 186, + "cache_read": 34226, + "output": 459 + }, + "turns": 3, + "tool_calls": 2, + "duration_s": 24 + }, "ticket-triage-subprocess": { "tokens": { - "input": 8, - "cache_write": 8200, - "cache_read": 64000, - "output": 6600 + "input": 11, + "cache_write": 4566, + "cache_read": 139038, + "output": 3771 }, - "turns": 5, - "tool_calls": 5, - "duration_s": 95 + "turns": 8, + "tool_calls": 8, + "duration_s": 91 } } } diff --git a/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py b/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py index 699d4611..d1f506a6 100644 --- a/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py +++ b/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py @@ -1,11 +1,13 @@ from __future__ import annotations +import asyncio import importlib.util import xml.etree.ElementTree as ET from pathlib import Path -from types import ModuleType +from types import ModuleType, SimpleNamespace import pytest +from inspect_ai.tool import ToolDef def _load_outcomes() -> ModuleType: @@ -114,3 +116,100 @@ def test_requires_ai_agent_tool_container_property() -> None: ) assert not _outcomes.has_ai_agent_connector(host) + + +def _configuration_score( + *calls: tuple[str, dict[str, object]], + artifacts: tuple[str, ...] = (), + expected_fields: tuple[str, ...] = ("provider", "model", "secret_names"), +) -> float: + state = SimpleNamespace( + metadata={"missing_configuration": list(expected_fields)}, + messages=[ + SimpleNamespace( + tool_calls=[ + SimpleNamespace(function=function, arguments=arguments) + for function, arguments in calls + ] + ) + ], + store={"artifacts": {path: "" for path in artifacts}}, + ) + return asyncio.run(_outcomes.configuration_requested()(state, None)).value + + +def test_configuration_tool_requests_required_values() -> None: + tool = ToolDef(_outcomes.request_configuration()) + message = asyncio.run(tool.tool(missing=["provider"])) + + assert tool.name == "request_configuration" + assert "configuration fields" in tool.description + assert tool.parameters.required == ["missing"] + assert "Wait for the user response" in message + + +def test_configuration_request_stops_before_bpmn_work() -> None: + request = ( + "request_configuration", + {"missing": ["provider", "model", "secret_names"]}, + ) + + assert _configuration_score(request) == 1.0 + assert ( + _configuration_score(("skill", {"command": "camunda-ai-agents"}), request) + == 1.0 + ) + assert _configuration_score(request, ("list_files", {})) == 0.0 + assert ( + _configuration_score( + ("text_editor", {"path": "/workspace/process.bpmn"}), + request, + ) + == 0.0 + ) + assert _configuration_score(request, artifacts=("/workspace/process.BPMN",)) == 0.0 + assert ( + _configuration_score( + ("request_configuration", {"missing": ["model"]}), + expected_fields=("model",), + ) + == 1.0 + ) + assert ( + _configuration_score( + ("request_configuration", {"missing": ["secret"]}), + expected_fields=("secret_names",), + ) + == 0.0 + ) + assert ( + _configuration_score( + ( + "mcp__configuration__request_configuration", + {"missing": ["provider", "model", "secret_names"]}, + ) + ) + == 1.0 + ) + + +def test_requires_expected_configuration_inputs() -> None: + expected = { + "provider.type": "openai", + "provider.openai.model.model": "gpt-4.1-mini", + "provider.openai.authentication.apiKey": "{{secrets.OPENAI_API_KEY}}", + } + + assert _outcomes.has_expected_configuration(expected, expected) + assert not _outcomes.has_expected_configuration( + {**expected, "provider.type": "anthropic"}, + expected, + ) + + +def test_claude_code_evaluates_configuration_samples() -> None: + task = _outcomes.camunda_ai_agents(agent="claude_code") + + assert [sample.id for sample in task.dataset] == [ + sample.id for sample in _outcomes.SAMPLES + ] diff --git a/skills/camunda-ai-agents/SKILL.md b/skills/camunda-ai-agents/SKILL.md index 9a3eeb5e..ae6502b8 100644 --- a/skills/camunda-ai-agents/SKILL.md +++ b/skills/camunda-ai-agents/SKILL.md @@ -20,7 +20,22 @@ The older **Task variant** (AI Agent connector on a service task paired with an - Camunda 8.8+ cluster (the AI Agent connector ships in 8.8+) - c8ctl CLI installed and a profile configured — see **camunda-c8ctl** -- An API key for the model provider you'll use (Anthropic, Amazon Bedrock, Azure OpenAI, Google Vertex AI, OpenAI, or any OpenAI-compatible provider). Store it as a Camunda cluster secret, never in the BPMN file. For local c8run, see **camunda-c8ctl** for the secrets bootstrap flow. +- A provider, exact model identifier, and names of its existing connector secrets, + supplied by the user. Store secret values as Camunda cluster secrets, never in + the BPMN file. For local c8run, see **camunda-c8ctl** for the secrets bootstrap + flow. + +## Provider Configuration + +Before creating or editing BPMN, confirm the provider, exact model identifier, +and every required existing connector-secret name. If any are missing, ask the +user and stop until they are confirmed. Do not choose a default provider or +model, invent a secret name, or ask for secret values. Inspect the selected +template for provider-specific model and authentication fields. + +For Camunda SaaS with Camunda-hosted connectors, configure secret values in +Console. Use the confirmed secret names in BPMN as `{{secrets.NAME}}` +references. ## Cross-References @@ -53,15 +68,15 @@ c8ctl element-template sync # 1. Find the current template ID and version — they evolve c8ctl element-template search "ai agent" -# 2. Inspect the properties you care about +# 2. Inspect the selected provider's model and authentication fields. c8ctl element-template get-properties c8ctl element-template get-properties --detailed data.systemPrompt.prompt -# 3. Apply to your ad-hoc subprocess element +# 3. Replace placeholders with user-supplied values and selected-template fields. c8ctl element-template apply -i AgentTools process.bpmn \ - --set provider.type=anthropic \ - --set provider.anthropic.authentication.apiKey='{{secrets.ANTHROPIC_API_KEY}}' \ - --set provider.anthropic.model.model=claude-sonnet-4-5 \ + --set 'provider.type=' \ + --set 'provider..authentication.={{secrets.}}' \ + --set 'provider..model.=' \ --set data.systemPrompt.prompt='="You are a customer support agent. Use the available tools to look up customers and orders, and escalate to a human only when needed."' \ --set data.userPrompt.prompt='="Customer " + customerId + " reports: " + issue' \ --set data.limits.maxModelCalls='=10' @@ -225,7 +240,8 @@ Lint catches structural BPMN problems but does not validate connector-template i - Every tool's flow ends with `toolCallResult` set in scope. - Both prompts start with `=`. - `data.limits.maxModelCalls` is set. -- API keys are pulled from `{{secrets.*}}`, not literal values. +- Provider, model, and `{{secrets.NAME}}` references are user-confirmed; never + put secret values in BPMN. ## References