Skip to content

Commit dbf892e

Browse files
authored
Merge branch 'main' into bai/prune-preserved-workspaces
2 parents d31f46f + fb1ad4c commit dbf892e

5 files changed

Lines changed: 297 additions & 54 deletions

File tree

‎pyproject.toml‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -63,7 +63,7 @@ dev = [
6363
"pytest-mock>=3.15.1",
6464
"pytest-cov>=7.1.0",
6565
"pytest-xdist[psutil]>=3.0.0",
66-
"mcp>=1.26.0",
66+
"mcp>=1.28.1", # >=1.28.1 clears CVE-2026-52869/52870 (1.27.2) + CVE-2026-59950 (1.28.1)
6767
"ruff>=0.15.7",
6868
"pyright>=1.1.408",
6969
"pip-audit>=2.10.0",

‎src/coder_eval/cli/run_command.py‎

Lines changed: 7 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -583,9 +583,13 @@ async def _run_with_experiment(
583583
default_experiment = experiment # fall back to custom as its own baseline
584584

585585
# Resolve tasks through experiment layer (applies all 5 config layers).
586-
# Layer-5 override failures (invalid -D value/path, sdk_options on a
587-
# non-claude agent, the agent.type guard) and duplicate-task-id checks raise
588-
# ValueError here; surface them as a clean CLI error instead of a traceback.
586+
# Global failures raise ValueError here — duplicate task IDs, early-stop
587+
# arming, or an invocation error that trips every task identically (bad
588+
# --type / -D value, repeats over the cap) — and we surface them as a clean
589+
# CLI error instead of a traceback. Per-task config-resolution failures
590+
# (e.g. sdk_options on a non-claude agent) among otherwise-resolvable tasks
591+
# are NOT raised: resolve_all_tasks isolates them into `skipped` so one
592+
# incompatible task can't abort the whole suite.
589593
try:
590594
resolved, skipped = resolve_all_tasks(
591595
task_files=all_task_files,

‎src/coder_eval/orchestration/experiment.py‎

Lines changed: 109 additions & 46 deletions
Original file line numberDiff line numberDiff line change
@@ -552,10 +552,18 @@ def resolve_all_tasks(
552552
Also handles tag filtering and unique task ID validation.
553553
554554
Task YAMLs that fail to load (YAML parse error, Pydantic validation,
555-
dataset expansion error) are recorded in the returned ``skipped`` list
556-
and excluded from the resolved set rather than aborting the suite. The
557-
caller surfaces ``skipped`` in the run summary so the failure is loud
558-
but recoverable.
555+
dataset expansion error) are recorded in the returned ``skipped`` list and
556+
excluded from the resolved set rather than aborting the suite.
557+
558+
Per-task config-resolution failures (a task whose own YAML is incompatible
559+
with the resolved run — e.g. Claude-only ``sdk_options`` surviving a
560+
``--type codex`` override, which ``CodexAgentConfig`` forbids) are likewise
561+
demoted to ``skipped`` — but only when other tasks resolve. If EVERY task
562+
that reaches resolution fails, the cause is a global invocation error (a bad
563+
``--type`` / ``-D`` value, repeats over the cap) rather than a per-task
564+
incompatibility, so it is re-raised and aborts the run. Early-stop arming
565+
errors always propagate. The caller surfaces ``skipped`` in the run summary
566+
so per-task failures are loud but recoverable.
559567
560568
Args:
561569
task_files: Paths to task YAML files.
@@ -572,10 +580,18 @@ def resolve_all_tasks(
572580
Raises:
573581
ValueError: If duplicate task IDs are found after resolution.
574582
"""
575-
from .early_stop import validate_early_stop
583+
from .early_stop import EarlyStopConfigError, validate_early_stop
576584

577585
resolved: list[ResolvedTask] = []
578586
skipped: list[SkippedTask] = []
587+
# Per-task config-resolution failures are collected here rather than raised
588+
# inline. After the loop they are demoted to ``skipped`` — UNLESS every task
589+
# that reached resolution failed, which signals a global invocation error
590+
# (bad --type, an invalid -D value, repeats over the cap) that trips every
591+
# task identically rather than a per-task incompatibility; that case is
592+
# re-raised so the CLI aborts cleanly instead of producing an empty run.
593+
resolution_errors: list[tuple[Path, Exception]] = []
594+
attempted = 0
579595

580596
# Resolve variant-level initial_prompt_file paths before the main loop
581597
exp_dir = experiment_file.parent if experiment_file is not None else None
@@ -623,50 +639,97 @@ def resolve_all_tasks(
623639
skipped.append(SkippedTask(path=str(task_file), reason=reason))
624640
continue
625641

626-
for expanded_task in expanded_tasks:
627-
for variant in experiment.variants:
628-
# Apply layers 1-4 (default → experiment-defaults → task → variant) + resolve repeats
629-
resolved_task, lineage, effective_repeats = resolve_task_for_variant(
630-
default_experiment, expanded_task, experiment, variant, config
631-
)
642+
# Isolate layer 1-5 config resolution per task file. A task whose own
643+
# YAML config is incompatible with the resolved run — e.g. Claude-only
644+
# agent fields (`sdk_options`) surviving a `--type codex` override, which
645+
# `CodexAgentConfig` forbids (`extra="forbid"`) — raises here. Without
646+
# isolation that single task aborts the entire coder-eval run. Buffer the
647+
# file's resolved tasks and commit them only once the whole file
648+
# resolves, so a mid-file failure discards this file's fan-out as a unit
649+
# (mirroring the load/expand isolation above) rather than leaving a
650+
# partial, lopsided fan-out behind.
651+
attempted += 1
652+
file_resolved: list[ResolvedTask] = []
653+
try:
654+
for expanded_task in expanded_tasks:
655+
for variant in experiment.variants:
656+
# Apply layers 1-4 (default → experiment-defaults → task → variant) + resolve repeats
657+
resolved_task, lineage, effective_repeats = resolve_task_for_variant(
658+
default_experiment, expanded_task, experiment, variant, config
659+
)
632660

633-
# Resolve file paths injected by variant overrides
634-
resolve_task_files(resolved_task, task_file, experiment_file)
635-
636-
# Apply prompt mutations or overrides (between file resolution and CLI overrides)
637-
_apply_prompt_overrides(resolved_task, experiment, variant, lineage)
638-
639-
# Apply layer 5 (CLI overrides)
640-
_apply_cli_overrides(resolved_task, config, lineage)
641-
642-
# Early-stop guardrails: run once the task is fully resolved (all 5
643-
# layers merged, incl. -D run_limits.stop_early). No-op unless armed;
644-
# a bad arming raises EarlyStopConfigError (a ValueError) which the
645-
# run path converts to a clean CLI error.
646-
validate_early_stop(resolved_task)
647-
648-
# Fan-out: simulation n_trials takes precedence over experiment repeats
649-
# when simulation is active; otherwise use experiment-level repeats.
650-
sim = resolved_task.simulation
651-
n_trials = sim.n_trials if (sim is not None and sim.enabled) else 1
652-
fan_count = n_trials if n_trials > 1 else effective_repeats
653-
for rep in range(fan_count):
654-
resolved.append(
655-
ResolvedTask(
656-
task=resolved_task,
657-
task_file=task_file,
658-
run_dir=build_task_run_dir(
659-
config.run_dir,
660-
variant.variant_id,
661-
resolved_task.task_id,
661+
# Resolve file paths injected by variant overrides
662+
resolve_task_files(resolved_task, task_file, experiment_file)
663+
664+
# Apply prompt mutations or overrides (between file resolution and CLI overrides)
665+
_apply_prompt_overrides(resolved_task, experiment, variant, lineage)
666+
667+
# Apply layer 5 (CLI overrides)
668+
_apply_cli_overrides(resolved_task, config, lineage)
669+
670+
# Early-stop guardrails: run once the task is fully resolved (all 5
671+
# layers merged, incl. -D run_limits.stop_early). No-op unless armed;
672+
# a bad arming raises EarlyStopConfigError (a ValueError).
673+
validate_early_stop(resolved_task)
674+
675+
# Fan-out: simulation n_trials takes precedence over experiment repeats
676+
# when simulation is active; otherwise use experiment-level repeats.
677+
sim = resolved_task.simulation
678+
n_trials = sim.n_trials if (sim is not None and sim.enabled) else 1
679+
fan_count = n_trials if n_trials > 1 else effective_repeats
680+
for rep in range(fan_count):
681+
file_resolved.append(
682+
ResolvedTask(
683+
task=resolved_task,
684+
task_file=task_file,
685+
run_dir=build_task_run_dir(
686+
config.run_dir,
687+
variant.variant_id,
688+
resolved_task.task_id,
689+
replicate_index=rep,
690+
),
691+
variant_id=variant.variant_id,
662692
replicate_index=rep,
663-
),
664-
variant_id=variant.variant_id,
665-
replicate_index=rep,
666-
source_yaml=source_yaml,
667-
config_lineage=dict(lineage),
693+
source_yaml=source_yaml,
694+
config_lineage=dict(lineage),
695+
)
668696
)
669-
)
697+
# Early-stop arming errors are a deliberate hard stop: they always
698+
# propagate (never demoted to skipped) so a misarmed run fails loudly.
699+
except EarlyStopConfigError:
700+
raise
701+
# Narrow set, matching the load/expand block above: config-resolution
702+
# and IO failures are collected (decided after the loop, below);
703+
# AttributeError / TypeError / ImportError still crash loudly as
704+
# regressions. Pydantic ValidationError is a ValueError subclass in v2,
705+
# so it's covered.
706+
except (FileNotFoundError, OSError, ValueError, yaml.YAMLError) as exc:
707+
resolution_errors.append((task_file, exc))
708+
continue
709+
710+
resolved.extend(file_resolved)
711+
712+
# Decide the fate of collected per-task resolution failures. If EVERY task
713+
# that reached resolution failed, refusing to proceed (rather than producing
714+
# an empty run) is the right call. We surface the first task's own error —
715+
# not a synthesized "global misconfig" message — because we can't actually
716+
# tell a genuine global cause (a bad --type / -D value that trips every task
717+
# identically) from N tasks each independently incompatible for the same
718+
# per-task reason. A ValueError (incl. Pydantic ValidationError and the "no
719+
# agent registered" guard) is re-raised verbatim so its message stays clean;
720+
# only a non-ValueError (FileNotFoundError/OSError/yaml.YAMLError — e.g. a
721+
# missing system_prompt_file) is normalized to ValueError so it still lands
722+
# in the caller's `except ValueError` (clean typer.BadParameter) instead of
723+
# escaping as a raw traceback.
724+
if resolution_errors and len(resolution_errors) == attempted:
725+
first_exc = resolution_errors[0][1]
726+
if isinstance(first_exc, ValueError):
727+
raise first_exc
728+
raise ValueError(str(first_exc)) from first_exc
729+
for err_file, exc in resolution_errors:
730+
reason = f"{type(exc).__name__}: {exc}"[:500]
731+
logger.warning("Skipping task file %s — config resolution failed: %s", err_file, reason)
732+
skipped.append(SkippedTask(path=str(err_file), reason=reason))
670733

671734
# Filter by tags
672735
if config.include_tags or config.exclude_tags:

0 commit comments

Comments
 (0)