@@ -552,10 +552,18 @@ def resolve_all_tasks(
552552 Also handles tag filtering and unique task ID validation.
553553
554554 Task YAMLs that fail to load (YAML parse error, Pydantic validation,
555- dataset expansion error) are recorded in the returned ``skipped`` list
556- and excluded from the resolved set rather than aborting the suite. The
557- caller surfaces ``skipped`` in the run summary so the failure is loud
558- but recoverable.
555+ dataset expansion error) are recorded in the returned ``skipped`` list and
556+ excluded from the resolved set rather than aborting the suite.
557+
558+ Per-task config-resolution failures (a task whose own YAML is incompatible
559+ with the resolved run — e.g. Claude-only ``sdk_options`` surviving a
560+ ``--type codex`` override, which ``CodexAgentConfig`` forbids) are likewise
561+ demoted to ``skipped`` — but only when other tasks resolve. If EVERY task
562+ that reaches resolution fails, the cause is a global invocation error (a bad
563+ ``--type`` / ``-D`` value, repeats over the cap) rather than a per-task
564+ incompatibility, so it is re-raised and aborts the run. Early-stop arming
565+ errors always propagate. The caller surfaces ``skipped`` in the run summary
566+ so per-task failures are loud but recoverable.
559567
560568 Args:
561569 task_files: Paths to task YAML files.
@@ -572,10 +580,18 @@ def resolve_all_tasks(
572580 Raises:
573581 ValueError: If duplicate task IDs are found after resolution.
574582 """
575- from .early_stop import validate_early_stop
583+ from .early_stop import EarlyStopConfigError , validate_early_stop
576584
577585 resolved : list [ResolvedTask ] = []
578586 skipped : list [SkippedTask ] = []
587+ # Per-task config-resolution failures are collected here rather than raised
588+ # inline. After the loop they are demoted to ``skipped`` — UNLESS every task
589+ # that reached resolution failed, which signals a global invocation error
590+ # (bad --type, an invalid -D value, repeats over the cap) that trips every
591+ # task identically rather than a per-task incompatibility; that case is
592+ # re-raised so the CLI aborts cleanly instead of producing an empty run.
593+ resolution_errors : list [tuple [Path , Exception ]] = []
594+ attempted = 0
579595
580596 # Resolve variant-level initial_prompt_file paths before the main loop
581597 exp_dir = experiment_file .parent if experiment_file is not None else None
@@ -623,50 +639,97 @@ def resolve_all_tasks(
623639 skipped .append (SkippedTask (path = str (task_file ), reason = reason ))
624640 continue
625641
626- for expanded_task in expanded_tasks :
627- for variant in experiment .variants :
628- # Apply layers 1-4 (default → experiment-defaults → task → variant) + resolve repeats
629- resolved_task , lineage , effective_repeats = resolve_task_for_variant (
630- default_experiment , expanded_task , experiment , variant , config
631- )
642+ # Isolate layer 1-5 config resolution per task file. A task whose own
643+ # YAML config is incompatible with the resolved run — e.g. Claude-only
644+ # agent fields (`sdk_options`) surviving a `--type codex` override, which
645+ # `CodexAgentConfig` forbids (`extra="forbid"`) — raises here. Without
646+ # isolation that single task aborts the entire coder-eval run. Buffer the
647+ # file's resolved tasks and commit them only once the whole file
648+ # resolves, so a mid-file failure discards this file's fan-out as a unit
649+ # (mirroring the load/expand isolation above) rather than leaving a
650+ # partial, lopsided fan-out behind.
651+ attempted += 1
652+ file_resolved : list [ResolvedTask ] = []
653+ try :
654+ for expanded_task in expanded_tasks :
655+ for variant in experiment .variants :
656+ # Apply layers 1-4 (default → experiment-defaults → task → variant) + resolve repeats
657+ resolved_task , lineage , effective_repeats = resolve_task_for_variant (
658+ default_experiment , expanded_task , experiment , variant , config
659+ )
632660
633- # Resolve file paths injected by variant overrides
634- resolve_task_files (resolved_task , task_file , experiment_file )
635-
636- # Apply prompt mutations or overrides (between file resolution and CLI overrides)
637- _apply_prompt_overrides (resolved_task , experiment , variant , lineage )
638-
639- # Apply layer 5 (CLI overrides)
640- _apply_cli_overrides (resolved_task , config , lineage )
641-
642- # Early-stop guardrails: run once the task is fully resolved (all 5
643- # layers merged, incl. -D run_limits.stop_early). No-op unless armed;
644- # a bad arming raises EarlyStopConfigError (a ValueError) which the
645- # run path converts to a clean CLI error.
646- validate_early_stop (resolved_task )
647-
648- # Fan-out: simulation n_trials takes precedence over experiment repeats
649- # when simulation is active; otherwise use experiment-level repeats.
650- sim = resolved_task .simulation
651- n_trials = sim .n_trials if (sim is not None and sim .enabled ) else 1
652- fan_count = n_trials if n_trials > 1 else effective_repeats
653- for rep in range (fan_count ):
654- resolved .append (
655- ResolvedTask (
656- task = resolved_task ,
657- task_file = task_file ,
658- run_dir = build_task_run_dir (
659- config .run_dir ,
660- variant .variant_id ,
661- resolved_task .task_id ,
661+ # Resolve file paths injected by variant overrides
662+ resolve_task_files (resolved_task , task_file , experiment_file )
663+
664+ # Apply prompt mutations or overrides (between file resolution and CLI overrides)
665+ _apply_prompt_overrides (resolved_task , experiment , variant , lineage )
666+
667+ # Apply layer 5 (CLI overrides)
668+ _apply_cli_overrides (resolved_task , config , lineage )
669+
670+ # Early-stop guardrails: run once the task is fully resolved (all 5
671+ # layers merged, incl. -D run_limits.stop_early). No-op unless armed;
672+ # a bad arming raises EarlyStopConfigError (a ValueError).
673+ validate_early_stop (resolved_task )
674+
675+ # Fan-out: simulation n_trials takes precedence over experiment repeats
676+ # when simulation is active; otherwise use experiment-level repeats.
677+ sim = resolved_task .simulation
678+ n_trials = sim .n_trials if (sim is not None and sim .enabled ) else 1
679+ fan_count = n_trials if n_trials > 1 else effective_repeats
680+ for rep in range (fan_count ):
681+ file_resolved .append (
682+ ResolvedTask (
683+ task = resolved_task ,
684+ task_file = task_file ,
685+ run_dir = build_task_run_dir (
686+ config .run_dir ,
687+ variant .variant_id ,
688+ resolved_task .task_id ,
689+ replicate_index = rep ,
690+ ),
691+ variant_id = variant .variant_id ,
662692 replicate_index = rep ,
663- ),
664- variant_id = variant .variant_id ,
665- replicate_index = rep ,
666- source_yaml = source_yaml ,
667- config_lineage = dict (lineage ),
693+ source_yaml = source_yaml ,
694+ config_lineage = dict (lineage ),
695+ )
668696 )
669- )
697+ # Early-stop arming errors are a deliberate hard stop: they always
698+ # propagate (never demoted to skipped) so a misarmed run fails loudly.
699+ except EarlyStopConfigError :
700+ raise
701+ # Narrow set, matching the load/expand block above: config-resolution
702+ # and IO failures are collected (decided after the loop, below);
703+ # AttributeError / TypeError / ImportError still crash loudly as
704+ # regressions. Pydantic ValidationError is a ValueError subclass in v2,
705+ # so it's covered.
706+ except (FileNotFoundError , OSError , ValueError , yaml .YAMLError ) as exc :
707+ resolution_errors .append ((task_file , exc ))
708+ continue
709+
710+ resolved .extend (file_resolved )
711+
712+ # Decide the fate of collected per-task resolution failures. If EVERY task
713+ # that reached resolution failed, refusing to proceed (rather than producing
714+ # an empty run) is the right call. We surface the first task's own error —
715+ # not a synthesized "global misconfig" message — because we can't actually
716+ # tell a genuine global cause (a bad --type / -D value that trips every task
717+ # identically) from N tasks each independently incompatible for the same
718+ # per-task reason. A ValueError (incl. Pydantic ValidationError and the "no
719+ # agent registered" guard) is re-raised verbatim so its message stays clean;
720+ # only a non-ValueError (FileNotFoundError/OSError/yaml.YAMLError — e.g. a
721+ # missing system_prompt_file) is normalized to ValueError so it still lands
722+ # in the caller's `except ValueError` (clean typer.BadParameter) instead of
723+ # escaping as a raw traceback.
724+ if resolution_errors and len (resolution_errors ) == attempted :
725+ first_exc = resolution_errors [0 ][1 ]
726+ if isinstance (first_exc , ValueError ):
727+ raise first_exc
728+ raise ValueError (str (first_exc )) from first_exc
729+ for err_file , exc in resolution_errors :
730+ reason = f"{ type (exc ).__name__ } : { exc } " [:500 ]
731+ logger .warning ("Skipping task file %s — config resolution failed: %s" , err_file , reason )
732+ skipped .append (SkippedTask (path = str (err_file ), reason = reason ))
670733
671734 # Filter by tags
672735 if config .include_tags or config .exclude_tags :
0 commit comments