From 07e06f730a0c1dde52b852cccebddc60134ed2a0 Mon Sep 17 00:00:00 2001 From: Peter Bojtos Date: Thu, 17 Sep 2026 18:02:20 +0200 Subject: [PATCH 01/12] fix(camunda-ai-agents): require provider configuration Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- skills/camunda-ai-agents/SKILL.md | 30 +++++++++++++++++++++++------- 1 file changed, 23 insertions(+), 7 deletions(-) diff --git a/skills/camunda-ai-agents/SKILL.md b/skills/camunda-ai-agents/SKILL.md index 9a3eeb5e..07fe5e02 100644 --- a/skills/camunda-ai-agents/SKILL.md +++ b/skills/camunda-ai-agents/SKILL.md @@ -20,7 +20,22 @@ The older **Task variant** (AI Agent connector on a service task paired with an - Camunda 8.8+ cluster (the AI Agent connector ships in 8.8+) - c8ctl CLI installed and a profile configured — see **camunda-c8ctl** -- An API key for the model provider you'll use (Anthropic, Amazon Bedrock, Azure OpenAI, Google Vertex AI, OpenAI, or any OpenAI-compatible provider). Store it as a Camunda cluster secret, never in the BPMN file. For local c8run, see **camunda-c8ctl** for the secrets bootstrap flow. +- A provider, exact model identifier, and names of its existing connector secrets, + supplied by the user. Store secret values as Camunda cluster secrets, never in + the BPMN file. For local c8run, see **camunda-c8ctl** for the secrets bootstrap + flow. + +## Provider Configuration + +Before creating or editing BPMN, confirm the provider, exact model identifier, +and every required existing connector-secret name. If any are missing, ask the +user and stop until they are confirmed. Do not choose a default provider or +model, invent a secret name, or ask for secret values. Inspect the selected +template for provider-specific model and authentication fields. + +For Camunda SaaS with Camunda-hosted connectors, connector secrets are +configured and referenced through Console. Confirm the names for the target +cluster before deployment. ## Cross-References @@ -53,15 +68,15 @@ c8ctl element-template sync # 1. Find the current template ID and version — they evolve c8ctl element-template search "ai agent" -# 2. Inspect the properties you care about +# 2. Inspect the selected provider's model and authentication fields. c8ctl element-template get-properties c8ctl element-template get-properties --detailed data.systemPrompt.prompt -# 3. Apply to your ad-hoc subprocess element +# 3. Replace placeholders with user-supplied values and selected-template fields. c8ctl element-template apply -i AgentTools process.bpmn \ - --set provider.type=anthropic \ - --set provider.anthropic.authentication.apiKey='{{secrets.ANTHROPIC_API_KEY}}' \ - --set provider.anthropic.model.model=claude-sonnet-4-5 \ + --set 'provider.type=' \ + --set 'provider..authentication.={{secrets.}}' \ + --set 'provider..model.=' \ --set data.systemPrompt.prompt='="You are a customer support agent. Use the available tools to look up customers and orders, and escalate to a human only when needed."' \ --set data.userPrompt.prompt='="Customer " + customerId + " reports: " + issue' \ --set data.limits.maxModelCalls='=10' @@ -225,7 +240,8 @@ Lint catches structural BPMN problems but does not validate connector-template i - Every tool's flow ends with `toolCallResult` set in scope. - Both prompts start with `=`. - `data.limits.maxModelCalls` is set. -- API keys are pulled from `{{secrets.*}}`, not literal values. +- Provider, model, and `{{secrets.NAME}}` references are user-confirmed; never + put secret values in BPMN. ## References From 1b7bdba0a7f699ad8647d36adb24b2dc3abdb782 Mon Sep 17 00:00:00 2001 From: Peter Bojtos Date: Thu, 17 Sep 2026 18:25:54 +0200 Subject: [PATCH 02/12] test(camunda-ai-agents): cover provider clarification Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- evals/skills/camunda-ai-agents/outcomes.py | 125 +++++++++++++++--- .../test_ai_agent_outcomes.py | 41 +++++- 2 files changed, 150 insertions(+), 16 deletions(-) diff --git a/evals/skills/camunda-ai-agents/outcomes.py b/evals/skills/camunda-ai-agents/outcomes.py index 56d237ce..adcec577 100644 --- a/evals/skills/camunda-ai-agents/outcomes.py +++ b/evals/skills/camunda-ai-agents/outcomes.py @@ -10,15 +10,28 @@ from __future__ import annotations +from collections.abc import Sequence +from pathlib import Path import xml.etree.ElementTree as ET -from core.agents import AgentKind, build_agent +from core.agents import AgentKind, WORKSPACE_RULES, build_agent from core.metadata import EvalMetadata from core.paths import SANDBOXES_DIR, Arm, skill_dirs_for_arm from inspect_ai import Task, task +from inspect_ai.agent import Agent, AgentPrompt, react from inspect_ai.dataset import Sample from inspect_ai.scorer import Score, Scorer, Target, mean, scorer, stderr from inspect_ai.solver import TaskState +from inspect_ai.tool import ( + Tool, + bash_session, + grep, + list_files, + skill, + text_editor, + tool as inspect_tool, + web_search, +) from inspect_ai.util import sandbox from scorers.transcript import assert_skill_loaded from solvers.collect_artifacts import with_artifact_collection @@ -52,6 +65,37 @@ AI_AGENT_TOOL_CONTAINER_PROPERTY = "io.camunda.agenticai.toolContainer" +@inspect_tool +def request_configuration() -> Tool: + """Ask for missing provider configuration before BPMN work.""" + + async def execute() -> str: + return ( + "Ask the user for the provider, exact model identifier, and names of " + "existing connector secrets, then stop." + ) + + return execute + + +def _build_evaluator_agent(agent: AgentKind, skill_dirs: Sequence[Path]) -> Agent: + if agent != "react": + return build_agent(agent, skill_dirs, submit=False) + return react( + prompt=AgentPrompt(instructions=WORKSPACE_RULES), + submit=False, + tools=[ + bash_session(timeout=300), + text_editor(timeout=60), + grep(timeout=30), + list_files(timeout=30), + web_search(), + *([skill(list(skill_dirs))] if skill_dirs else []), + request_configuration(), + ], + ) + + def has_ai_agent_connector(host: ET.Element) -> bool: """Check for connector metadata emitted by an AI Agent template.""" @@ -79,6 +123,9 @@ def ai_agent_shape_valid(path: str = BPMN_PATH) -> Scorer: """Verify that the authored BPMN contains core AI-agent subprocess wiring.""" async def score(state: TaskState, target: Target) -> Score: + if not (state.metadata or {}).get("check_shape", True): + return Score(value=1.0, explanation="shape check not applicable") + expected_process_id = (state.metadata or {}).get("process_id") required_tools = set((state.metadata or {}).get("required_tools", [])) @@ -219,9 +266,35 @@ async def score(state: TaskState, target: Target) -> Score: return score -SAVE_AND_DEPLOY = ( - "\n\nSave the BPMN to /workspace/process.bpmn. Do not stop until the file is created." -) +@scorer(metrics=[mean(), stderr()]) +def configuration_requested() -> Scorer: + """Verify that incomplete provider configuration stops BPMN authoring.""" + + async def score(state: TaskState, target: Target) -> Score: + if not (state.metadata or {}).get("requires_configuration_request"): + return Score(value=1.0, explanation="configuration request not applicable") + + calls = [ + str(call.function) + for message in state.messages + for call in getattr(message, "tool_calls", None) or [] + ] + artifacts = state.store.get("artifacts") or {} + bpmn_paths = [path for path in artifacts if path.lower().endswith(".bpmn")] + requested = [ + index for index, name in enumerate(calls) if name == "request_configuration" + ] + if len(requested) != 1 or requested[0] != len(calls) - 1 or bpmn_paths: + return Score( + value=0.0, + explanation="did not request configuration and stop before BPMN work", + ) + return Score(value=1.0, explanation="requested configuration before BPMN work") + + return score + + +SAVE_AND_DEPLOY = "\n\nSave the BPMN to /workspace/process.bpmn. Do not stop until the file is created." SAMPLES = [ Sample( @@ -238,18 +311,25 @@ async def score(state: TaskState, target: Target) -> Score: "`c8ctl element-template sync`, use " '`c8ctl element-template search "AI Agent Sub-process" ' "--engine-version 8.8.0` to find the non-hybrid template, inspect " - "only `data.systemPrompt.prompt`, `data.userPrompt.prompt`, and " - "`data.limits.maxModelCalls` with `c8ctl element-template " - "get-properties data.systemPrompt.prompt data.userPrompt.prompt " + "only `provider.type`, `provider.openai.model.model`, " + "`provider.openai.authentication.apiKey`, `data.systemPrompt.prompt`, " + "`data.userPrompt.prompt`, and `data.limits.maxModelCalls` with " + "`c8ctl element-template get-properties provider.type " + "provider.openai.model.model provider.openai.authentication.apiKey " + "data.systemPrompt.prompt data.userPrompt.prompt " "data.limits.maxModelCalls --engine-version 8.8.0`, then apply that " "template ID with `c8ctl element-template apply -i AgentTools " - "/workspace/process.bpmn --set " + "/workspace/process.bpmn --set provider.type=openai --set " + "provider.openai.model.model=gpt-4.1-mini --set " + "provider.openai.authentication.apiKey={{secrets.OPENAI_API_KEY}} --set " "'data.systemPrompt.prompt==\"You are a ticket-triage agent. Use the " "available tools.\"' --set " "'data.userPrompt.prompt==\"Triage the current ticket.\"' --set " - "'data.limits.maxModelCalls==10'`. Do not inspect unrelated template " - "properties, configure an LLM provider, or hand-write connector " - "metadata. Do not stop until the command succeeds.\n" + "'data.limits.maxModelCalls==10'`. Use OpenAI with model " + "`gpt-4.1-mini` and existing connector secret `OPENAI_API_KEY`; do " + "not invent another provider or secret name, inspect unrelated " + "template properties, or hand-write connector metadata. Do not stop " + "until the command succeeds.\n" "3. Inside AgentTools add these root tools:\n" " - service task id LookupKnowledgeBase, name 'Lookup knowledge base'\n" " - service task id LookupCustomerData, name 'Lookup customer data'\n" @@ -259,8 +339,7 @@ async def score(state: TaskState, target: Target) -> Score: "6. Ensure tool outputs are mapped to toolCallResult.\n" "7. Configure agent prompts as FEEL strings and set " "data.limits.maxModelCalls.\n" - "Save the completed BPMN to /workspace/process.bpmn." - + SAVE_AND_DEPLOY + "Save the completed BPMN to /workspace/process.bpmn." + SAVE_AND_DEPLOY ), metadata={ "process_id": "ai-ticket-triage", @@ -271,17 +350,33 @@ async def score(state: TaskState, target: Target) -> Score: ], }, ), + Sample( + id="missing-provider-configuration", + input=( + "Use only the camunda-ai-agents skill. I want an AI Agent Sub-process " + "BPMN, but have not provided the provider, exact model identifier, or " + "connector-secret names. Ask for them and then stop. Do not create or " + "edit BPMN. To ask in this evaluation, call request_configuration." + ), + metadata={ + "check_shape": False, + "requires_configuration_request": True, + }, + ), ] @task def camunda_ai_agents(arm: Arm = "with_skill", agent: AgentKind = "react") -> Task: skill_dirs = skill_dirs_for_arm(arm, METADATA.excluded_skills) + # Claude Code does not receive the test-only request_configuration tool. + samples = SAMPLES if agent == "react" else SAMPLES[:1] return Task( - dataset=SAMPLES, - solver=with_artifact_collection(build_agent(agent, skill_dirs, submit=False)), + dataset=samples, + solver=with_artifact_collection(_build_evaluator_agent(agent, skill_dirs)), scorer=[ ai_agent_shape_valid(), + configuration_requested(), assert_skill_loaded("camunda-ai-agents", gating=False), ], sandbox=("docker", str(SANDBOXES_DIR / "compose-with-c8ctl.yaml")), diff --git a/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py b/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py index 699d4611..9dc8ab90 100644 --- a/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py +++ b/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py @@ -1,9 +1,10 @@ from __future__ import annotations +import asyncio import importlib.util import xml.etree.ElementTree as ET from pathlib import Path -from types import ModuleType +from types import ModuleType, SimpleNamespace import pytest @@ -114,3 +115,41 @@ def test_requires_ai_agent_tool_container_property() -> None: ) assert not _outcomes.has_ai_agent_connector(host) + + +def _configuration_score(*calls: str, artifacts: tuple[str, ...] = ()) -> float: + state = SimpleNamespace( + metadata={"requires_configuration_request": True}, + messages=[ + SimpleNamespace( + tool_calls=[SimpleNamespace(function=call) for call in calls] + ) + ], + store={"artifacts": {path: "" for path in artifacts}}, + ) + return asyncio.run(_outcomes.configuration_requested()(state, None)).value + + +def test_configuration_tool_requests_required_values() -> None: + message = asyncio.run(_outcomes.request_configuration()()) + + assert "provider" in message + assert "exact model identifier" in message + assert "connector secrets" in message + + +def test_configuration_request_stops_before_bpmn_work() -> None: + assert _configuration_score("request_configuration") == 1.0 + assert _configuration_score("request_configuration", "list_files") == 0.0 + assert ( + _configuration_score( + "request_configuration", artifacts=("/workspace/process.BPMN",) + ) + == 0.0 + ) + + +def test_claude_code_keeps_positive_sample() -> None: + task = _outcomes.camunda_ai_agents(agent="claude_code") + + assert [sample.id for sample in task.dataset] == ["ticket-triage-subprocess"] From 0f0280bfc053b5e78454a67eae9206132fe7a6ea Mon Sep 17 00:00:00 2001 From: Peter Bojtos Date: Thu, 17 Sep 2026 18:48:18 +0200 Subject: [PATCH 03/12] fix(camunda-ai-agents): describe configuration test tool Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- evals/skills/camunda-ai-agents/outcomes.py | 2 ++ evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py | 6 +++++- 2 files changed, 7 insertions(+), 1 deletion(-) diff --git a/evals/skills/camunda-ai-agents/outcomes.py b/evals/skills/camunda-ai-agents/outcomes.py index adcec577..e1d638b3 100644 --- a/evals/skills/camunda-ai-agents/outcomes.py +++ b/evals/skills/camunda-ai-agents/outcomes.py @@ -70,6 +70,8 @@ def request_configuration() -> Tool: """Ask for missing provider configuration before BPMN work.""" async def execute() -> str: + """Ask for provider configuration and stop BPMN work.""" + return ( "Ask the user for the provider, exact model identifier, and names of " "existing connector secrets, then stop." diff --git a/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py b/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py index 9dc8ab90..cb7f45fe 100644 --- a/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py +++ b/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py @@ -7,6 +7,7 @@ from types import ModuleType, SimpleNamespace import pytest +from inspect_ai.tool import ToolDef def _load_outcomes() -> ModuleType: @@ -131,8 +132,11 @@ def _configuration_score(*calls: str, artifacts: tuple[str, ...] = ()) -> float: def test_configuration_tool_requests_required_values() -> None: - message = asyncio.run(_outcomes.request_configuration()()) + tool = ToolDef(_outcomes.request_configuration()) + message = asyncio.run(tool.tool()) + assert tool.name == "request_configuration" + assert "provider configuration" in tool.description assert "provider" in message assert "exact model identifier" in message assert "connector secrets" in message From 68e6fdfcc9bd65c15b9d952c858212d9f170bc6a Mon Sep 17 00:00:00 2001 From: Peter Bojtos Date: Thu, 17 Sep 2026 19:46:08 +0200 Subject: [PATCH 04/12] test(evals): refine provider configuration coverage Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- evals/skills/camunda-ai-agents/outcomes.py | 113 ++++++++++-------- .../test_ai_agent_outcomes.py | 41 +++++-- evals/src/core/agents.py | 6 +- 3 files changed, 98 insertions(+), 62 deletions(-) diff --git a/evals/skills/camunda-ai-agents/outcomes.py b/evals/skills/camunda-ai-agents/outcomes.py index e1d638b3..8524a409 100644 --- a/evals/skills/camunda-ai-agents/outcomes.py +++ b/evals/skills/camunda-ai-agents/outcomes.py @@ -10,28 +10,16 @@ from __future__ import annotations -from collections.abc import Sequence -from pathlib import Path import xml.etree.ElementTree as ET -from core.agents import AgentKind, WORKSPACE_RULES, build_agent +from core.agents import AgentKind, build_agent from core.metadata import EvalMetadata from core.paths import SANDBOXES_DIR, Arm, skill_dirs_for_arm from inspect_ai import Task, task -from inspect_ai.agent import Agent, AgentPrompt, react from inspect_ai.dataset import Sample from inspect_ai.scorer import Score, Scorer, Target, mean, scorer, stderr from inspect_ai.solver import TaskState -from inspect_ai.tool import ( - Tool, - bash_session, - grep, - list_files, - skill, - text_editor, - tool as inspect_tool, - web_search, -) +from inspect_ai.tool import Tool, tool as inspect_tool from inspect_ai.util import sandbox from scorers.transcript import assert_skill_loaded from solvers.collect_artifacts import with_artifact_collection @@ -67,35 +55,18 @@ @inspect_tool def request_configuration() -> Tool: - """Ask for missing provider configuration before BPMN work.""" - - async def execute() -> str: - """Ask for provider configuration and stop BPMN work.""" + """Request user-provided configuration.""" - return ( - "Ask the user for the provider, exact model identifier, and names of " - "existing connector secrets, then stop." - ) + async def execute(message: str) -> str: + """Send a user-facing request for missing configuration. - return execute + Args: + message: Concise question asking the user for required configuration. + """ + return "Configuration request sent. Wait for the user response before further work." -def _build_evaluator_agent(agent: AgentKind, skill_dirs: Sequence[Path]) -> Agent: - if agent != "react": - return build_agent(agent, skill_dirs, submit=False) - return react( - prompt=AgentPrompt(instructions=WORKSPACE_RULES), - submit=False, - tools=[ - bash_session(timeout=300), - text_editor(timeout=60), - grep(timeout=30), - list_files(timeout=30), - web_search(), - *([skill(list(skill_dirs))] if skill_dirs else []), - request_configuration(), - ], - ) + return execute def has_ai_agent_connector(host: ET.Element) -> bool: @@ -273,24 +244,38 @@ def configuration_requested() -> Scorer: """Verify that incomplete provider configuration stops BPMN authoring.""" async def score(state: TaskState, target: Target) -> Score: - if not (state.metadata or {}).get("requires_configuration_request"): + expected_terms = (state.metadata or {}).get("missing_configuration") + if not expected_terms: return Score(value=1.0, explanation="configuration request not applicable") calls = [ - str(call.function) + (str(call.function), call.arguments or {}) for message in state.messages for call in getattr(message, "tool_calls", None) or [] ] artifacts = state.store.get("artifacts") or {} bpmn_paths = [path for path in artifacts if path.lower().endswith(".bpmn")] requested = [ - index for index, name in enumerate(calls) if name == "request_configuration" + (index, arguments) + for index, (name, arguments) in enumerate(calls) + if name == "request_configuration" ] - if len(requested) != 1 or requested[0] != len(calls) - 1 or bpmn_paths: + if len(requested) != 1 or requested[0][0] != len(calls) - 1 or bpmn_paths: return Score( value=0.0, explanation="did not request configuration and stop before BPMN work", ) + message = requested[0][1].get("message") + if not isinstance(message, str): + return Score(value=0.0, explanation="configuration request has no message") + missing_terms = [ + term for term in expected_terms if term.casefold() not in message.casefold() + ] + if missing_terms: + return Score( + value=0.0, + explanation=f"configuration request omits: {missing_terms}", + ) return Score(value=1.0, explanation="requested configuration before BPMN work") return score @@ -355,14 +340,37 @@ async def score(state: TaskState, target: Target) -> Score: Sample( id="missing-provider-configuration", input=( - "Use only the camunda-ai-agents skill. I want an AI Agent Sub-process " - "BPMN, but have not provided the provider, exact model identifier, or " - "connector-secret names. Ask for them and then stop. Do not create or " - "edit BPMN. To ask in this evaluation, call request_configuration." + "Use only the camunda-ai-agents skill. Create an AI Agent Sub-process " + "BPMN, but no provider, exact model identifier, or connector-secret " + "names were supplied." ), metadata={ "check_shape": False, - "requires_configuration_request": True, + "missing_configuration": ["provider", "model", "secret"], + }, + ), + Sample( + id="missing-model-configuration", + input=( + "Use only the camunda-ai-agents skill. Create an AI Agent Sub-process " + "BPMN with provider `openai` and existing connector secret " + "`OPENAI_API_KEY`, but no exact model identifier was supplied." + ), + metadata={ + "check_shape": False, + "missing_configuration": ["model"], + }, + ), + Sample( + id="missing-secret-configuration", + input=( + "Use only the camunda-ai-agents skill. Create an AI Agent Sub-process " + "BPMN with provider `openai` and exact model identifier " + "`gpt-4.1-mini`, but no connector-secret name was supplied." + ), + metadata={ + "check_shape": False, + "missing_configuration": ["secret"], }, ), ] @@ -375,7 +383,14 @@ def camunda_ai_agents(arm: Arm = "with_skill", agent: AgentKind = "react") -> Ta samples = SAMPLES if agent == "react" else SAMPLES[:1] return Task( dataset=samples, - solver=with_artifact_collection(_build_evaluator_agent(agent, skill_dirs)), + solver=with_artifact_collection( + build_agent( + agent, + skill_dirs, + submit=False, + extra_react_tools=[request_configuration()] if agent == "react" else (), + ) + ), scorer=[ ai_agent_shape_valid(), configuration_requested(), diff --git a/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py b/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py index cb7f45fe..bbda1b00 100644 --- a/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py +++ b/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py @@ -118,12 +118,19 @@ def test_requires_ai_agent_tool_container_property() -> None: assert not _outcomes.has_ai_agent_connector(host) -def _configuration_score(*calls: str, artifacts: tuple[str, ...] = ()) -> float: +def _configuration_score( + *calls: tuple[str, dict[str, object]], + artifacts: tuple[str, ...] = (), + expected_terms: tuple[str, ...] = ("provider", "model", "secret"), +) -> float: state = SimpleNamespace( - metadata={"requires_configuration_request": True}, + metadata={"missing_configuration": list(expected_terms)}, messages=[ SimpleNamespace( - tool_calls=[SimpleNamespace(function=call) for call in calls] + tool_calls=[ + SimpleNamespace(function=function, arguments=arguments) + for function, arguments in calls + ] ) ], store={"artifacts": {path: "" for path in artifacts}}, @@ -133,23 +140,33 @@ def _configuration_score(*calls: str, artifacts: tuple[str, ...] = ()) -> float: def test_configuration_tool_requests_required_values() -> None: tool = ToolDef(_outcomes.request_configuration()) - message = asyncio.run(tool.tool()) + message = asyncio.run(tool.tool(message="Please provide the missing values.")) assert tool.name == "request_configuration" - assert "provider configuration" in tool.description - assert "provider" in message - assert "exact model identifier" in message - assert "connector secrets" in message + assert "user-facing request" in tool.description + assert "Wait for the user response" in message def test_configuration_request_stops_before_bpmn_work() -> None: - assert _configuration_score("request_configuration") == 1.0 - assert _configuration_score("request_configuration", "list_files") == 0.0 + request = ( + "request_configuration", + { + "message": ( + "Please provide the provider, exact model identifier, and " + "connector-secret name." + ) + }, + ) + + assert _configuration_score(request) == 1.0 + assert _configuration_score(request, ("list_files", {})) == 0.0 + assert _configuration_score(request, artifacts=("/workspace/process.BPMN",)) == 0.0 assert ( _configuration_score( - "request_configuration", artifacts=("/workspace/process.BPMN",) + ("request_configuration", {"message": "Please provide the model."}), + expected_terms=("model",), ) - == 0.0 + == 1.0 ) diff --git a/evals/src/core/agents.py b/evals/src/core/agents.py index 35ed25aa..9b899d64 100644 --- a/evals/src/core/agents.py +++ b/evals/src/core/agents.py @@ -7,6 +7,7 @@ from inspect_ai.agent import Agent, AgentPrompt, react from inspect_ai.tool import ( + Tool, bash_session, grep, list_files, @@ -40,11 +41,13 @@ def build_agent( kind: AgentKind, skill_dirs: Sequence[Path], submit: bool = True, + extra_react_tools: Sequence[Tool] = (), ) -> Agent: """Construct the configured agent loop with the given skill set. ``submit=False`` removes react's submit() tool, so the agent halts when - it stops calling tools. + it stops calling tools. ``extra_react_tools`` adds evaluator-specific + tools without changing the default tool set. """ if kind == "react": instructions = INSTRUCTIONS_REACT if submit else WORKSPACE_RULES @@ -58,6 +61,7 @@ def build_agent( list_files(timeout=30), web_search(), *([skill(list(skill_dirs))] if skill_dirs else []), + *extra_react_tools, ], ) if kind == "claude_code": From da5ae5f52d400bc228ade516e002abfac63636c4 Mon Sep 17 00:00:00 2001 From: Peter Bojtos Date: Thu, 17 Sep 2026 20:01:01 +0200 Subject: [PATCH 05/12] test(camunda-ai-agents): validate configuration fields Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- evals/skills/camunda-ai-agents/outcomes.py | 37 +++++++++++-------- .../test_ai_agent_outcomes.py | 27 ++++++++------ 2 files changed, 36 insertions(+), 28 deletions(-) diff --git a/evals/skills/camunda-ai-agents/outcomes.py b/evals/skills/camunda-ai-agents/outcomes.py index 8524a409..42798ac7 100644 --- a/evals/skills/camunda-ai-agents/outcomes.py +++ b/evals/skills/camunda-ai-agents/outcomes.py @@ -10,6 +10,7 @@ from __future__ import annotations +from typing import Literal import xml.etree.ElementTree as ET from core.agents import AgentKind, build_agent @@ -55,13 +56,16 @@ @inspect_tool def request_configuration() -> Tool: - """Request user-provided configuration.""" + """Request specific configuration from the user.""" - async def execute(message: str) -> str: - """Send a user-facing request for missing configuration. + async def execute( + missing: list[Literal["provider", "model", "secret_names"]], + ) -> str: + """Request the missing configuration fields from the user. Args: - message: Concise question asking the user for required configuration. + missing: ``provider``, exact ``model``, or existing ``secret_names``. + Never request secret values. """ return "Configuration request sent. Wait for the user response before further work." @@ -244,8 +248,8 @@ def configuration_requested() -> Scorer: """Verify that incomplete provider configuration stops BPMN authoring.""" async def score(state: TaskState, target: Target) -> Score: - expected_terms = (state.metadata or {}).get("missing_configuration") - if not expected_terms: + expected_fields = (state.metadata or {}).get("missing_configuration") + if not expected_fields: return Score(value=1.0, explanation="configuration request not applicable") calls = [ @@ -265,16 +269,17 @@ async def score(state: TaskState, target: Target) -> Score: value=0.0, explanation="did not request configuration and stop before BPMN work", ) - message = requested[0][1].get("message") - if not isinstance(message, str): - return Score(value=0.0, explanation="configuration request has no message") - missing_terms = [ - term for term in expected_terms if term.casefold() not in message.casefold() - ] - if missing_terms: + missing = requested[0][1].get("missing") + if ( + not isinstance(missing, list) + or not all(isinstance(field, str) for field in missing) + or sorted(missing) != sorted(expected_fields) + ): return Score( value=0.0, - explanation=f"configuration request omits: {missing_terms}", + explanation=( + f"requested {missing!r}, expected missing fields {expected_fields!r}" + ), ) return Score(value=1.0, explanation="requested configuration before BPMN work") @@ -346,7 +351,7 @@ async def score(state: TaskState, target: Target) -> Score: ), metadata={ "check_shape": False, - "missing_configuration": ["provider", "model", "secret"], + "missing_configuration": ["provider", "model", "secret_names"], }, ), Sample( @@ -370,7 +375,7 @@ async def score(state: TaskState, target: Target) -> Score: ), metadata={ "check_shape": False, - "missing_configuration": ["secret"], + "missing_configuration": ["secret_names"], }, ), ] diff --git a/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py b/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py index bbda1b00..4a550530 100644 --- a/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py +++ b/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py @@ -121,10 +121,10 @@ def test_requires_ai_agent_tool_container_property() -> None: def _configuration_score( *calls: tuple[str, dict[str, object]], artifacts: tuple[str, ...] = (), - expected_terms: tuple[str, ...] = ("provider", "model", "secret"), + expected_fields: tuple[str, ...] = ("provider", "model", "secret_names"), ) -> float: state = SimpleNamespace( - metadata={"missing_configuration": list(expected_terms)}, + metadata={"missing_configuration": list(expected_fields)}, messages=[ SimpleNamespace( tool_calls=[ @@ -140,22 +140,18 @@ def _configuration_score( def test_configuration_tool_requests_required_values() -> None: tool = ToolDef(_outcomes.request_configuration()) - message = asyncio.run(tool.tool(message="Please provide the missing values.")) + message = asyncio.run(tool.tool(missing=["provider"])) assert tool.name == "request_configuration" - assert "user-facing request" in tool.description + assert "configuration fields" in tool.description + assert tool.parameters.required == ["missing"] assert "Wait for the user response" in message def test_configuration_request_stops_before_bpmn_work() -> None: request = ( "request_configuration", - { - "message": ( - "Please provide the provider, exact model identifier, and " - "connector-secret name." - ) - }, + {"missing": ["provider", "model", "secret_names"]}, ) assert _configuration_score(request) == 1.0 @@ -163,11 +159,18 @@ def test_configuration_request_stops_before_bpmn_work() -> None: assert _configuration_score(request, artifacts=("/workspace/process.BPMN",)) == 0.0 assert ( _configuration_score( - ("request_configuration", {"message": "Please provide the model."}), - expected_terms=("model",), + ("request_configuration", {"missing": ["model"]}), + expected_fields=("model",), ) == 1.0 ) + assert ( + _configuration_score( + ("request_configuration", {"missing": ["secret"]}), + expected_fields=("secret_names",), + ) + == 0.0 + ) def test_claude_code_keeps_positive_sample() -> None: From ffaf0e44568e5354a22527824769706db84409a0 Mon Sep 17 00:00:00 2001 From: Peter Bojtos Date: Thu, 17 Sep 2026 20:19:00 +0200 Subject: [PATCH 06/12] fix(evals): stabilize AI agent template fixture Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- evals/skills/camunda-ai-agents/outcomes.py | 50 ++++++++++++++++------ evals/src/core/agents.py | 6 +-- 2 files changed, 38 insertions(+), 18 deletions(-) diff --git a/evals/skills/camunda-ai-agents/outcomes.py b/evals/skills/camunda-ai-agents/outcomes.py index 42798ac7..2f576ea7 100644 --- a/evals/skills/camunda-ai-agents/outcomes.py +++ b/evals/skills/camunda-ai-agents/outcomes.py @@ -10,17 +10,29 @@ from __future__ import annotations +from collections.abc import Sequence +from pathlib import Path from typing import Literal import xml.etree.ElementTree as ET -from core.agents import AgentKind, build_agent +from core.agents import AgentKind, WORKSPACE_RULES, build_agent from core.metadata import EvalMetadata from core.paths import SANDBOXES_DIR, Arm, skill_dirs_for_arm from inspect_ai import Task, task +from inspect_ai.agent import Agent, AgentPrompt, react from inspect_ai.dataset import Sample from inspect_ai.scorer import Score, Scorer, Target, mean, scorer, stderr from inspect_ai.solver import TaskState -from inspect_ai.tool import Tool, tool as inspect_tool +from inspect_ai.tool import ( + Tool, + bash_session, + grep, + list_files, + skill, + text_editor, + tool as inspect_tool, + web_search, +) from inspect_ai.util import sandbox from scorers.transcript import assert_skill_loaded from solvers.collect_artifacts import with_artifact_collection @@ -73,6 +85,24 @@ async def execute( return execute +def _build_evaluator_agent(agent: AgentKind, skill_dirs: Sequence[Path]) -> Agent: + if agent != "react": + return build_agent(agent, skill_dirs, submit=False) + return react( + prompt=AgentPrompt(instructions=WORKSPACE_RULES), + submit=False, + tools=[ + bash_session(timeout=300), + text_editor(timeout=60), + grep(timeout=30), + list_files(timeout=30), + web_search(), + *([skill(list(skill_dirs))] if skill_dirs else []), + request_configuration(), + ], + ) + + def has_ai_agent_connector(host: ET.Element) -> bool: """Check for connector metadata emitted by an AI Agent template.""" @@ -298,8 +328,9 @@ async def score(state: TaskState, target: Target) -> Score: "'AI Ticket Triage') with an AI Agent Sub-process pattern:\n" "1. Start event 'Ticket received'.\n" "2. Ad-hoc subprocess id AgentTools (name 'Agent tools') as the AI " - "agent host. Before running c8ctl, write a complete, diagrammed BPMN " - "process to /workspace/process.bpmn. Then run exactly " + "agent host. Before running c8ctl, write a diagrammed BPMN process " + "with an empty AgentTools host to /workspace/process.bpmn. Then run " + "exactly " "`c8ctl element-template sync`, use " '`c8ctl element-template search "AI Agent Sub-process" ' "--engine-version 8.8.0` to find the non-hybrid template, inspect " @@ -321,7 +352,7 @@ async def score(state: TaskState, target: Target) -> Score: "`gpt-4.1-mini` and existing connector secret `OPENAI_API_KEY`; do " "not invent another provider or secret name, inspect unrelated " "template properties, or hand-write connector metadata. Do not stop " - "until the command succeeds.\n" + "until the command succeeds. Then add the tools below.\n" "3. Inside AgentTools add these root tools:\n" " - service task id LookupKnowledgeBase, name 'Lookup knowledge base'\n" " - service task id LookupCustomerData, name 'Lookup customer data'\n" @@ -388,14 +419,7 @@ def camunda_ai_agents(arm: Arm = "with_skill", agent: AgentKind = "react") -> Ta samples = SAMPLES if agent == "react" else SAMPLES[:1] return Task( dataset=samples, - solver=with_artifact_collection( - build_agent( - agent, - skill_dirs, - submit=False, - extra_react_tools=[request_configuration()] if agent == "react" else (), - ) - ), + solver=with_artifact_collection(_build_evaluator_agent(agent, skill_dirs)), scorer=[ ai_agent_shape_valid(), configuration_requested(), diff --git a/evals/src/core/agents.py b/evals/src/core/agents.py index 9b899d64..35ed25aa 100644 --- a/evals/src/core/agents.py +++ b/evals/src/core/agents.py @@ -7,7 +7,6 @@ from inspect_ai.agent import Agent, AgentPrompt, react from inspect_ai.tool import ( - Tool, bash_session, grep, list_files, @@ -41,13 +40,11 @@ def build_agent( kind: AgentKind, skill_dirs: Sequence[Path], submit: bool = True, - extra_react_tools: Sequence[Tool] = (), ) -> Agent: """Construct the configured agent loop with the given skill set. ``submit=False`` removes react's submit() tool, so the agent halts when - it stops calling tools. ``extra_react_tools`` adds evaluator-specific - tools without changing the default tool set. + it stops calling tools. """ if kind == "react": instructions = INSTRUCTIONS_REACT if submit else WORKSPACE_RULES @@ -61,7 +58,6 @@ def build_agent( list_files(timeout=30), web_search(), *([skill(list(skill_dirs))] if skill_dirs else []), - *extra_react_tools, ], ) if kind == "claude_code": From 0e17bda33d67060ef276617432fc2716903207d7 Mon Sep 17 00:00:00 2001 From: Peter Bojtos Date: Thu, 17 Sep 2026 20:39:22 +0200 Subject: [PATCH 07/12] test(camunda-ai-agents): cover configuration across agents Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- evals/skills/camunda-ai-agents/outcomes.py | 74 ++++++++++++++++--- .../test_ai_agent_outcomes.py | 29 +++++++- 2 files changed, 91 insertions(+), 12 deletions(-) diff --git a/evals/skills/camunda-ai-agents/outcomes.py b/evals/skills/camunda-ai-agents/outcomes.py index 2f576ea7..fa7e8edf 100644 --- a/evals/skills/camunda-ai-agents/outcomes.py +++ b/evals/skills/camunda-ai-agents/outcomes.py @@ -19,7 +19,7 @@ from core.metadata import EvalMetadata from core.paths import SANDBOXES_DIR, Arm, skill_dirs_for_arm from inspect_ai import Task, task -from inspect_ai.agent import Agent, AgentPrompt, react +from inspect_ai.agent import Agent, AgentPrompt, BridgedToolsSpec, react from inspect_ai.dataset import Sample from inspect_ai.scorer import Score, Scorer, Target, mean, scorer, stderr from inspect_ai.solver import TaskState @@ -34,6 +34,7 @@ web_search, ) from inspect_ai.util import sandbox +from inspect_swe import claude_code from scorers.transcript import assert_skill_loaded from solvers.collect_artifacts import with_artifact_collection @@ -85,7 +86,28 @@ async def execute( return execute -def _build_evaluator_agent(agent: AgentKind, skill_dirs: Sequence[Path]) -> Agent: +def _build_evaluator_agent( + agent: AgentKind, + skill_dirs: Sequence[Path], + include_configuration_tool: bool, +) -> Agent: + if agent == "claude_code": + return claude_code( + system_prompt=WORKSPACE_RULES, + skills=[str(path) for path in skill_dirs] if skill_dirs else None, + bridged_tools=( + [ + BridgedToolsSpec( + name="configuration", + tools=[request_configuration()], + ) + ] + if include_configuration_tool + else None + ), + cwd="/workspace", + disallowed_tools=["ExitPlanMode"], + ) if agent != "react": return build_agent(agent, skill_dirs, submit=False) return react( @@ -98,7 +120,7 @@ def _build_evaluator_agent(agent: AgentKind, skill_dirs: Sequence[Path]) -> Agen list_files(timeout=30), web_search(), *([skill(list(skill_dirs))] if skill_dirs else []), - request_configuration(), + *([request_configuration()] if include_configuration_tool else []), ], ) @@ -125,6 +147,14 @@ def has_ai_agent_connector(host: ET.Element) -> bool: ) +def has_expected_configuration( + inputs: dict[str, str], expected: dict[str, str] +) -> bool: + """Check connector inputs against the user-supplied configuration.""" + + return all(inputs.get(target) == value for target, value in expected.items()) + + @scorer(metrics=[mean(), stderr()]) def ai_agent_shape_valid(path: str = BPMN_PATH) -> Scorer: """Verify that the authored BPMN contains core AI-agent subprocess wiring.""" @@ -240,9 +270,18 @@ async def score(state: TaskState, target: Target) -> Score: ) prompt_inputs = { - inp.get("target"): (inp.get("source") or "") + target: (inp.get("source") or "") for inp in host.findall(".//zeebe:input", NS) + if (target := inp.get("target")) is not None } + expected_configuration = (state.metadata or {}).get( + "expected_configuration", {} + ) + if not has_expected_configuration(prompt_inputs, expected_configuration): + return Score( + value=0.0, + explanation="connector inputs do not match the supplied configuration", + ) system_prompt = prompt_inputs.get("data.systemPrompt.prompt", "") user_prompt = prompt_inputs.get("data.userPrompt.prompt", "") if not system_prompt.startswith("=") or not user_prompt.startswith("="): @@ -292,7 +331,11 @@ async def score(state: TaskState, target: Target) -> Score: requested = [ (index, arguments) for index, (name, arguments) in enumerate(calls) - if name == "request_configuration" + if name + in { + "request_configuration", + "mcp__configuration__request_configuration", + } ] if len(requested) != 1 or requested[0][0] != len(calls) - 1 or bpmn_paths: return Score( @@ -352,7 +395,9 @@ async def score(state: TaskState, target: Target) -> Score: "`gpt-4.1-mini` and existing connector secret `OPENAI_API_KEY`; do " "not invent another provider or secret name, inspect unrelated " "template properties, or hand-write connector metadata. Do not stop " - "until the command succeeds. Then add the tools below.\n" + "until the command succeeds. Your next action must use the text " + "editor to add all three tool activities and mappings inside " + "AgentTools; do not read the file or stop before doing so.\n" "3. Inside AgentTools add these root tools:\n" " - service task id LookupKnowledgeBase, name 'Lookup knowledge base'\n" " - service task id LookupCustomerData, name 'Lookup customer data'\n" @@ -371,6 +416,11 @@ async def score(state: TaskState, target: Target) -> Score: "LookupCustomerData", "EscalateToHuman", ], + "expected_configuration": { + "provider.type": "openai", + "provider.openai.model.model": "gpt-4.1-mini", + "provider.openai.authentication.apiKey": "{{secrets.OPENAI_API_KEY}}", + }, }, ), Sample( @@ -415,11 +465,15 @@ async def score(state: TaskState, target: Target) -> Score: @task def camunda_ai_agents(arm: Arm = "with_skill", agent: AgentKind = "react") -> Task: skill_dirs = skill_dirs_for_arm(arm, METADATA.excluded_skills) - # Claude Code does not receive the test-only request_configuration tool. - samples = SAMPLES if agent == "react" else SAMPLES[:1] return Task( - dataset=samples, - solver=with_artifact_collection(_build_evaluator_agent(agent, skill_dirs)), + dataset=SAMPLES, + solver=with_artifact_collection( + _build_evaluator_agent( + agent, + skill_dirs, + include_configuration_tool=arm == "with_skill", + ) + ), scorer=[ ai_agent_shape_valid(), configuration_requested(), diff --git a/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py b/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py index 4a550530..aa80defd 100644 --- a/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py +++ b/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py @@ -171,9 +171,34 @@ def test_configuration_request_stops_before_bpmn_work() -> None: ) == 0.0 ) + assert ( + _configuration_score( + ( + "mcp__configuration__request_configuration", + {"missing": ["provider", "model", "secret_names"]}, + ) + ) + == 1.0 + ) + + +def test_requires_expected_configuration_inputs() -> None: + expected = { + "provider.type": "openai", + "provider.openai.model.model": "gpt-4.1-mini", + "provider.openai.authentication.apiKey": "{{secrets.OPENAI_API_KEY}}", + } + + assert _outcomes.has_expected_configuration(expected, expected) + assert not _outcomes.has_expected_configuration( + {**expected, "provider.type": "anthropic"}, + expected, + ) -def test_claude_code_keeps_positive_sample() -> None: +def test_claude_code_evaluates_configuration_samples() -> None: task = _outcomes.camunda_ai_agents(agent="claude_code") - assert [sample.id for sample in task.dataset] == ["ticket-triage-subprocess"] + assert [sample.id for sample in task.dataset] == [ + sample.id for sample in _outcomes.SAMPLES + ] From 7cc5a6e0d62dfc727c1d149a944cdf00480c7717 Mon Sep 17 00:00:00 2001 From: Peter Bojtos Date: Thu, 17 Sep 2026 20:57:17 +0200 Subject: [PATCH 08/12] fix(evals): use valid BPMN DI fixture Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- evals/skills/camunda-ai-agents/outcomes.py | 13 +++++++------ 1 file changed, 7 insertions(+), 6 deletions(-) diff --git a/evals/skills/camunda-ai-agents/outcomes.py b/evals/skills/camunda-ai-agents/outcomes.py index fa7e8edf..9cdb29f7 100644 --- a/evals/skills/camunda-ai-agents/outcomes.py +++ b/evals/skills/camunda-ai-agents/outcomes.py @@ -371,9 +371,12 @@ async def score(state: TaskState, target: Target) -> Score: "'AI Ticket Triage') with an AI Agent Sub-process pattern:\n" "1. Start event 'Ticket received'.\n" "2. Ad-hoc subprocess id AgentTools (name 'Agent tools') as the AI " - "agent host. Before running c8ctl, write a diagrammed BPMN process " - "with an empty AgentTools host to /workspace/process.bpmn. Then run " - "exactly " + "agent host. Before running c8ctl, write a complete, diagrammed " + "BPMN process to /workspace/process.bpmn, including the tools below. " + "Use `http://www.omg.org/spec/DD/20100524/DC` and " + "`http://www.omg.org/spec/DD/20100524/DI` for the `dc` and `di` " + "prefixes, and keep `bpmndi` at " + "`http://www.omg.org/spec/BPMN/20100524/DI`. Then run exactly " "`c8ctl element-template sync`, use " '`c8ctl element-template search "AI Agent Sub-process" ' "--engine-version 8.8.0` to find the non-hybrid template, inspect " @@ -395,9 +398,7 @@ async def score(state: TaskState, target: Target) -> Score: "`gpt-4.1-mini` and existing connector secret `OPENAI_API_KEY`; do " "not invent another provider or secret name, inspect unrelated " "template properties, or hand-write connector metadata. Do not stop " - "until the command succeeds. Your next action must use the text " - "editor to add all three tool activities and mappings inside " - "AgentTools; do not read the file or stop before doing so.\n" + "until the command succeeds.\n" "3. Inside AgentTools add these root tools:\n" " - service task id LookupKnowledgeBase, name 'Lookup knowledge base'\n" " - service task id LookupCustomerData, name 'Lookup customer data'\n" From 868752106d1e85b40af11c0019553152933970e5 Mon Sep 17 00:00:00 2001 From: Peter Bojtos Date: Thu, 17 Sep 2026 21:17:12 +0200 Subject: [PATCH 09/12] test(camunda-ai-agents): reject work before clarification Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- evals/skills/camunda-ai-agents/outcomes.py | 9 ++++++++- .../camunda-ai-agents/test_ai_agent_outcomes.py | 11 +++++++++++ 2 files changed, 19 insertions(+), 1 deletion(-) diff --git a/evals/skills/camunda-ai-agents/outcomes.py b/evals/skills/camunda-ai-agents/outcomes.py index 9cdb29f7..48f4e0b5 100644 --- a/evals/skills/camunda-ai-agents/outcomes.py +++ b/evals/skills/camunda-ai-agents/outcomes.py @@ -337,7 +337,14 @@ async def score(state: TaskState, target: Target) -> Score: "mcp__configuration__request_configuration", } ] - if len(requested) != 1 or requested[0][0] != len(calls) - 1 or bpmn_paths: + pre_request_calls = calls[: requested[0][0]] if len(requested) == 1 else [] + performed_work = any(name.lower() != "skill" for name, _ in pre_request_calls) + if ( + len(requested) != 1 + or requested[0][0] != len(calls) - 1 + or bpmn_paths + or performed_work + ): return Score( value=0.0, explanation="did not request configuration and stop before BPMN work", diff --git a/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py b/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py index aa80defd..d1f506a6 100644 --- a/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py +++ b/evals/skills/camunda-ai-agents/test_ai_agent_outcomes.py @@ -155,7 +155,18 @@ def test_configuration_request_stops_before_bpmn_work() -> None: ) assert _configuration_score(request) == 1.0 + assert ( + _configuration_score(("skill", {"command": "camunda-ai-agents"}), request) + == 1.0 + ) assert _configuration_score(request, ("list_files", {})) == 0.0 + assert ( + _configuration_score( + ("text_editor", {"path": "/workspace/process.bpmn"}), + request, + ) + == 0.0 + ) assert _configuration_score(request, artifacts=("/workspace/process.BPMN",)) == 0.0 assert ( _configuration_score( From 7205beebe3cf72b945a0285a68d3409826dfed85 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Thu, 17 Sep 2026 19:33:13 +0000 Subject: [PATCH 10/12] chore(evals): regenerate baselines on CI (epochs=3) --- .../camunda-ai-agents/outcomes_baseline.json | 47 ++++++++++++++++--- 1 file changed, 40 insertions(+), 7 deletions(-) diff --git a/evals/skills/camunda-ai-agents/outcomes_baseline.json b/evals/skills/camunda-ai-agents/outcomes_baseline.json index 5328ca27..97fd9d66 100644 --- a/evals/skills/camunda-ai-agents/outcomes_baseline.json +++ b/evals/skills/camunda-ai-agents/outcomes_baseline.json @@ -2,16 +2,49 @@ "model": "anthropic/claude-sonnet-4-6", "with_skill": { "samples": { + "missing-model-configuration": { + "tokens": { + "input": 5, + "cache_write": 193, + "cache_read": 34220, + "output": 338 + }, + "turns": 3, + "tool_calls": 2, + "duration_s": 21 + }, + "missing-provider-configuration": { + "tokens": { + "input": 5, + "cache_write": 170, + "cache_read": 34190, + "output": 553 + }, + "turns": 3, + "tool_calls": 2, + "duration_s": 25 + }, + "missing-secret-configuration": { + "tokens": { + "input": 5, + "cache_write": 186, + "cache_read": 34226, + "output": 459 + }, + "turns": 3, + "tool_calls": 2, + "duration_s": 24 + }, "ticket-triage-subprocess": { "tokens": { - "input": 8, - "cache_write": 8200, - "cache_read": 64000, - "output": 6600 + "input": 11, + "cache_write": 4566, + "cache_read": 139038, + "output": 3771 }, - "turns": 5, - "tool_calls": 5, - "duration_s": 95 + "turns": 8, + "tool_calls": 8, + "duration_s": 91 } } } From d94d32ce369aa161d25150783981e97b482905df Mon Sep 17 00:00:00 2001 From: Peter Bojtos Date: Thu, 17 Sep 2026 21:50:13 +0200 Subject: [PATCH 11/12] test(camunda-ai-agents): name configuration request tool Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- evals/skills/camunda-ai-agents/outcomes.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/evals/skills/camunda-ai-agents/outcomes.py b/evals/skills/camunda-ai-agents/outcomes.py index 48f4e0b5..ebb438d4 100644 --- a/evals/skills/camunda-ai-agents/outcomes.py +++ b/evals/skills/camunda-ai-agents/outcomes.py @@ -71,7 +71,7 @@ def request_configuration() -> Tool: """Request specific configuration from the user.""" - async def execute( + async def request_configuration( missing: list[Literal["provider", "model", "secret_names"]], ) -> str: """Request the missing configuration fields from the user. @@ -83,7 +83,7 @@ async def execute( return "Configuration request sent. Wait for the user response before further work." - return execute + return request_configuration def _build_evaluator_agent( From 75ce01fbfe8003f2bb240a787e6ac70e26d522f7 Mon Sep 17 00:00:00 2001 From: Peter Bojtos Date: Thu, 17 Sep 2026 22:04:29 +0200 Subject: [PATCH 12/12] docs(camunda-ai-agents): clarify SaaS secret references Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- skills/camunda-ai-agents/SKILL.md | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/skills/camunda-ai-agents/SKILL.md b/skills/camunda-ai-agents/SKILL.md index 07fe5e02..ae6502b8 100644 --- a/skills/camunda-ai-agents/SKILL.md +++ b/skills/camunda-ai-agents/SKILL.md @@ -33,9 +33,9 @@ user and stop until they are confirmed. Do not choose a default provider or model, invent a secret name, or ask for secret values. Inspect the selected template for provider-specific model and authentication fields. -For Camunda SaaS with Camunda-hosted connectors, connector secrets are -configured and referenced through Console. Confirm the names for the target -cluster before deployment. +For Camunda SaaS with Camunda-hosted connectors, configure secret values in +Console. Use the confirmed secret names in BPMN as `{{secrets.NAME}}` +references. ## Cross-References