diff --git a/_4_OptSim/StandardSens/configs/trade_study.yaml b/_4_OptSim/StandardSens/configs/trade_study.yaml new file mode 100644 index 0000000..9d6218a --- /dev/null +++ b/_4_OptSim/StandardSens/configs/trade_study.yaml @@ -0,0 +1,49 @@ +# A trade study: named vehicles compared across the standard sims. +# +# make opt-trade +# make opt-trade STUDY=path/to/another_study.yaml +# +# See docs/doe-reverse-engineering.md. Copy this file for each study; results land +# in _4_OptSim/results/trade/.md and .csv. +name: rear_roll_stiffness + +# Each candidate is the baseline vehicle with these variables changed. Any +# variable declared in vehicle_architecture.yaml may be used, mass and CG +# included, because every candidate is compiled rather than overridden. The +# baseline is always included and needs no entry. +# +# `both` makes exactly the changes of the other two, so the report also says +# whether those two changes stack or interact. +candidates: + stiff_rear_bar: + rear.stabar.rate_n_m_per_rad: 961.495352 + soft_front_spring: + front.actuation.spring_rate_n_per_m: 21015.2202 + both: + rear.stabar.rate_n_m_per_rad: 961.495352 + front.actuation.spring_rate_n_per_m: 21015.2202 + +# What to compare, per standard: SteadyStateEval, RampSteerEval or TransientEval. +# All three run against the one executable compiled for each candidate. +# +# `resolution` is the smallest change worth acting on. A smaller delta is still +# reported, marked `~`. Leave it off and the delta is reported unjudged. There is +# no score and no ranking: what a tenth of a degree of understeer is worth in +# settling time is your call, not the tool's. +metrics: + SteadyStateEval: + understeer_gradient_deg_per_g: {resolution: 0.02} + roll_gradient_deg_per_g: {resolution: 0.02} + peak_handwheel_torque_Nm: {resolution: 0.5} + TransientEval: + yaw_rise_time_s: {resolution: 0.005} + yaw_overshoot_pct: {resolution: 1.0} + roll_overshoot_pct: {resolution: 1.0} + settling_time_s: {resolution: 0.05} + +# Also render each standard's PDF report per vehicle. Worth having when the study +# goes in front of a design review; costs roughly ten seconds per run. +render_reports: true + +# CPUs available to simulations (the Docker container's count, not the host's). +cpus: 12 diff --git a/_4_OptSim/StandardSens/pipeline/batch.py b/_4_OptSim/StandardSens/pipeline/batch.py index be45e00..162d462 100644 --- a/_4_OptSim/StandardSens/pipeline/batch.py +++ b/_4_OptSim/StandardSens/pipeline/batch.py @@ -1,4 +1,4 @@ -"""batch.py — Run the SteadyStateEval report wrapper for all variants. +"""batch.py — Run each configured standard for all variants. For each variant_XXXX/build// that has a compiled executable: 1. Skip if metrics.csv already exists and is valid (correct row count) @@ -20,7 +20,7 @@ import pandas as pd import yaml -from StandardSens.pipeline.steady_state_eval_report import run_report +from StandardSens.pipeline.standards import get_standard, run_standard # --------------------------------------------------------------------------- # Config @@ -83,7 +83,10 @@ def run_variant( results_dir.mkdir(parents=True, exist_ok=True) try: - metrics_csv = run_report( + # Dispatch on the standard's name: running one standard's report against + # another's executable fails late and unhelpfully. + metrics_csv = run_standard( + get_standard(standard), variant_dir=variant_dir, build_dir=build_dir, exec_name=standard_cfg["model"], diff --git a/_4_OptSim/StandardSens/pipeline/compiler.py b/_4_OptSim/StandardSens/pipeline/compiler.py index 8b58ebd..6d3fa06 100644 --- a/_4_OptSim/StandardSens/pipeline/compiler.py +++ b/_4_OptSim/StandardSens/pipeline/compiler.py @@ -17,6 +17,7 @@ from __future__ import annotations +from collections.abc import Collection import platform import subprocess import sys @@ -55,6 +56,7 @@ # list would keep caches its neighbour had already declared stale. PIPELINE_TOOLING_INPUTS = ( DEFAULT_REPORT_WRAPPER, + STANDARD_DIR / "pipeline/standards.py", DEFAULT_STEADY_STATE_SIM, DEFAULT_STEADY_STATE_CONFIG, DEFAULT_MODELICA_RUNNER, @@ -269,9 +271,13 @@ def compile_all( template_path: Path = DEFAULT_MOS_TEMPLATE, doe_config_path: Path = DEFAULT_DOE_CONFIG, architecture_config_path: Path = DEFAULT_ARCHITECTURE_CONFIG, + only_standards: Collection[str] | None = None, ) -> dict[str, list[Path]]: """Compile all variants in population_dir for all standards in config. + `only_standards` narrows that to the named ones, for callers that run several + standards against one executable and must not pay for a build per standard. + Skips variants that are already compiled and whose inputs haven't changed. Runs variants in parallel using ProcessPoolExecutor. Returns dict mapping standard -> list of successful exe paths. @@ -279,6 +285,8 @@ def compile_all( """ cfg = load_compiler_config(compiler_config_path) standards: dict[str, dict] = cfg["standards"] + if only_standards is not None: + standards = {name: standards[name] for name in only_standards} max_workers: int = cfg.get("max_workers", 2) # Resolve boblib_path relative to the config file diff --git a/_4_OptSim/StandardSens/pipeline/evaluator.py b/_4_OptSim/StandardSens/pipeline/evaluator.py index 9ba5437..e6177d3 100644 --- a/_4_OptSim/StandardSens/pipeline/evaluator.py +++ b/_4_OptSim/StandardSens/pipeline/evaluator.py @@ -7,10 +7,10 @@ value. That default is the slow one on purpose: a compile is always correct, and a wrong override is silent. See `overrides.py` for why the list is what it is. -Executables and evaluations are cached under the solve directory, and discarded -when BobLib, the vehicle or the SteadyStateEval tooling changes. The star design -does not depend on the targets, so a second solve against new targets reuses -every star evaluation and pays only for verification. +Executables live in a `VariantStore` and evaluations beside it; both are +discarded when BobLib, the vehicle or the SteadyStateEval tooling changes. The +star design does not depend on the targets, so a second solve against new targets +reuses every star evaluation and pays only for verification. """ from __future__ import annotations @@ -25,39 +25,19 @@ import time from typing import Any -import yaml - from StandardSens.pipeline import overrides as ov -from StandardSens.pipeline._pipeline_hash import check_pipeline_hash -from StandardSens.pipeline.compiler import ( - PIPELINE_TOOLING_INPUTS, - compile_all, - find_exe, - load_compiler_config, -) -from StandardSens.pipeline.generate_configs import SWEEP_SCOPE_ALL, refresh_doe_config -from StandardSens.pipeline.generator import build_context, generate_variants, read_metrics_csv -from StandardSens.pipeline.orchestration import ( - ARCHITECTURE_CONFIG, - COMPILER_CONFIG, - DOE_CONFIG, - STANDARD_BUILD_DIR, -) +from StandardSens.pipeline.generator import build_context, read_metrics_csv +from StandardSens.pipeline.orchestration import STANDARD_BUILD_DIR from StandardSens.pipeline.sampler import read_baseline +from StandardSens.pipeline.standards import case_loss from StandardSens.pipeline.steady_state_eval_report import Isoline, run_report +from StandardSens.pipeline.variants import BUILD_STANDARD, VariantStore, split_cpus, variant_key -STANDARD = "SteadyStateEval" SOLVE_DIR = STANDARD_BUILD_DIR / "solve" EVALUATION_TIMEOUT_S = 1800 -# Simulation throughput on a 12-CPU container was 1.9x higher with every CPU busy -# than with four, so concurrent evaluations are preferred until each would drop -# below this many case workers. -MIN_CASE_WORKERS = 4 - Variant = dict[str, float] Metrics = dict[str, float] -CompileKey = tuple[tuple[str, float], ...] class SteadyStateEvaluator: @@ -70,62 +50,28 @@ def __init__(self, *, isoline: Isoline, cpus: int) -> None: float(isoline.velocity_mps), tuple(sorted(map(float, isoline.target_ays), reverse=True)) ) self.cpus = max(1, int(cpus)) - self.exes_dir = SOLVE_DIR / "exes" + + self.store = VariantStore(SOLVE_DIR) self.evals_dir = SOLVE_DIR / "evals" + if self.store.was_reset: + shutil.rmtree(self.evals_dir, ignore_errors=True) - cfg = self._write_own_doe_config() + cfg = self.store.doe_config self.variables: dict[str, dict[str, Any]] = {v["path"]: v for v in cfg["variables"]} - self.context = build_context(cfg, self.doe_config_path) + self.context = build_context(cfg, self.store.doe_config_path) self.baseline: Variant = read_baseline(Path(cfg["baseline_mo"]), cfg["variables"]) - - compiler_cfg = load_compiler_config(COMPILER_CONFIG) - self.standard_cfg: dict[str, Any] = compiler_cfg["standards"][STANDARD] - boblib_path = (COMPILER_CONFIG.resolve().parent / compiler_cfg["boblib_path"]).resolve() - - self._drop_caches_if_inputs_changed(boblib_path) - self._index: dict[str, int] = ( - json.loads(self._index_path.read_text()) if self._index_path.exists() else {} - ) self._init_cache: dict[Path, dict[str, ov.InitParameter]] = {} - def _write_own_doe_config(self) -> dict[str, Any]: - """Generate the solver's DOE config beside its caches, not over the sweep's. - - The sweep's `_doe_config.yaml` is committed and records the scope its - population was built at, which `opt-search` relies on. The solver needs - every variable's spec whatever that scope was, so it keeps its own copy. - The generated file names the baseline record and vehicle template - relative to the sweep's config directory; here they are made absolute so - they resolve from anywhere. - """ - self.doe_config_path = (SOLVE_DIR / "_doe_config.yaml").resolve() - self.doe_config_path.parent.mkdir(parents=True, exist_ok=True) - cfg = refresh_doe_config( - architecture_config_path=ARCHITECTURE_CONFIG, - compiler_config_path=COMPILER_CONFIG, - doe_config_path=self.doe_config_path, - scope=SWEEP_SCOPE_ALL, - ) - config_dir = DOE_CONFIG.resolve().parent - cfg["baseline_mo"] = (config_dir / cfg["baseline_mo"]).resolve().as_posix() - cfg["architecture"]["template"] = ( - (config_dir.parent / cfg["architecture"]["template"]).resolve().as_posix() - ) - self.doe_config_path.write_text(yaml.safe_dump(cfg, sort_keys=False)) - return cfg - - # ---------------------------------------------------------------- evaluation - def __call__(self, variants: list[Variant]) -> list[Metrics]: pending = [v for v in variants if not self._metrics_path(v).exists()] if pending: - keys = {compile_key(v, self.baseline) for v in pending} - self._ensure_executables(keys) - for key in keys: # parse each init XML once, before the threads race to - self._init_parameters(self._build_dir(key)) + compiled = [compiled_part(v, self.baseline) for v in pending] + self.store.ensure_compiled(compiled) + for vehicle in compiled: # parse each init XML once, before the threads race to + self._init_parameters(self.store.build_dir(vehicle)) - concurrent = min(len(pending), max(1, self.cpus // MIN_CASE_WORKERS)) - workers = min(len(self.isoline.target_ays), max(1, self.cpus // concurrent)) + concurrent, workers = split_cpus(len(pending), self.cpus) + workers = min(workers, len(self.isoline.target_ays)) print( f"Simulating {len(pending)} setup(s), {concurrent} at a time with {workers} " f"case workers each ({len(variants) - len(pending)} cached)...", @@ -137,10 +83,13 @@ def __call__(self, variants: list[Variant]) -> list[Metrics]: for done, _ in enumerate(pool.map(simulate, pending), start=1): print(f" [{done}/{len(pending)}] {time.time() - started:.0f}s", flush=True) - return [read_metrics_csv(self._metrics_path(v)) for v in variants] + results = [read_metrics_csv(self._metrics_path(v)) for v in variants] + for variant, metrics in zip(variants, results, strict=True): + require_whole(variant, metrics) + return results def _simulate(self, variant: Variant, *, workers: int) -> None: - build_dir = self._build_dir(compile_key(variant, self.baseline)) + build_dir = self.store.build_dir(compiled_part(variant, self.baseline)) runtime = {p: v for p, v in variant.items() if p in ov.RUNTIME_SAFE_PATHS} eval_dir = self._eval_dir(variant) eval_dir.mkdir(parents=True, exist_ok=True) @@ -148,7 +97,7 @@ def _simulate(self, variant: Variant, *, workers: int) -> None: run_report( variant_dir=eval_dir, build_dir=build_dir, - exec_name=self.standard_cfg["model"], + exec_name=self.store.standard_cfg["model"], timeout=EVALUATION_TIMEOUT_S, init_parameters=ov.variant_overrides( runtime, self.variables, self.context, self._init_parameters(build_dir) @@ -158,100 +107,48 @@ def _simulate(self, variant: Variant, *, workers: int) -> None: render_report=False, # the solver reads the metrics CSV, never the PDF ) - # --------------------------------------------------------------- executables - - def _ensure_executables(self, keys: set[CompileKey]) -> None: - new = [k for k in sorted(keys) if _key_json(k) not in self._index] - if not new and all(find_exe(self._build_dir(k), self.standard_cfg) for k in keys): - return - - for key in new: - self._index[_key_json(key)] = len(self._index) - self.exes_dir.mkdir(parents=True, exist_ok=True) - ordered = sorted(self._index, key=self._index.__getitem__) - generate_variants( - self.doe_config_path, [json.loads(entry) for entry in ordered], self.exes_dir - ) - self._index_path.write_text(json.dumps(self._index, indent=1)) - compile_all(self.exes_dir, COMPILER_CONFIG, doe_config_path=self.doe_config_path) - - for key in keys: - build_dir = self._build_dir(key) - if find_exe(build_dir, self.standard_cfg) is None: - log = build_dir.parents[1] / f"compile_error_{STANDARD}.log" - detail = log.read_text()[-2000:] if log.exists() else "no compile log written" - raise RuntimeError(f"Compile failed for {dict(key) or 'baseline'}:\n{detail}") - - def _build_dir(self, key: CompileKey) -> Path: - return self.exes_dir / f"variant_{self._index[_key_json(key)]:04d}" / "build" / STANDARD - def _init_parameters(self, build_dir: Path) -> dict[str, ov.InitParameter]: if build_dir not in self._init_cache: - init_xml = ov.find_init_xml(build_dir, self.standard_cfg["model"]) + init_xml = ov.find_init_xml(build_dir, self.store.standard_cfg["model"]) self._init_cache[build_dir] = ov.load_init_parameters(init_xml) return self._init_cache[build_dir] - @property - def _index_path(self) -> Path: - return self.exes_dir / "index.json" - - # --------------------------------------------------------------------- cache - def _eval_dir(self, variant: Variant) -> Path: - payload = json.dumps( - {"variant": {p: _round(v) for p, v in sorted(variant.items())}, "isoline": self.isoline}, - sort_keys=True, - ) + payload = json.dumps({"variant": variant_key(variant), "isoline": self.isoline}) return self.evals_dir / hashlib.sha1(payload.encode()).hexdigest()[:16] def _metrics_path(self, variant: Variant) -> Path: - return self._eval_dir(variant) / "results" / STANDARD / "metrics.csv" - - def _drop_caches_if_inputs_changed(self, boblib_path: Path) -> None: - """A cache must never outlive the inputs it was built from.""" - if not self.exes_dir.exists(): - return - try: - # The same call compile_all makes, so the two cannot disagree about - # what counts as stale. - check_pipeline_hash( - self.exes_dir, - self.doe_config_path, - COMPILER_CONFIG, - boblib_path, - ARCHITECTURE_CONFIG, - PIPELINE_TOOLING_INPUTS, - ) - except RuntimeError: - print( - "Solver cache predates a change to BobLib, the vehicle, or the " - "SteadyStateEval tooling; discarding it." - ) - shutil.rmtree(self.exes_dir) - shutil.rmtree(self.evals_dir, ignore_errors=True) + return self._eval_dir(variant) / "results" / BUILD_STANDARD / "metrics.csv" -def compile_key(variant: Variant, baseline: Variant) -> CompileKey: - """The compile-only values that decide which executable a variant needs. +def require_whole(variant: Variant, metrics: Metrics) -> None: + """Refuse an evaluation that lost simulation cases. - Runtime-safe knobs never appear, so every setting of them shares one - executable. A compile-only knob left at its baseline does not appear either, - so it shares the baseline executable instead of forcing an identical rebuild. + Its gradients are fitted through fewer points than its neighbours', so it + would bend the surrogate without any number looking wrong, and because + evaluations are cached it would keep bending every later solve too. """ - return tuple( - sorted( - (path, _round(value)) - for path, value in variant.items() - if path not in ov.RUNTIME_SAFE_PATHS - and not math.isclose(value, baseline[path], rel_tol=1e-12, abs_tol=1e-12) + why = case_loss(metrics) + if why: + raise RuntimeError( + f"SteadyStateEval was not whole for {variant}: {why}. Its metrics are not " + "comparable with the other evaluations, so the solve stops here rather than " + "fit through them. The usual cause is a target a_y this setup cannot settle " + "at: lower the top of `test_matrix.target_ays` in solve_config.yaml (which " + "re-simulates the star), or inspect the run under Build/StandardSens/solve/." ) - ) - -def _round(value: float) -> float: - """Collapse float noise so equal settings share a cache entry.""" - return float(f"{float(value):.12g}") +def compiled_part(variant: Variant, baseline: Variant) -> Variant: + """The changes that decide which executable a variant needs. -def _key_json(key: CompileKey) -> str: - return json.dumps(dict(key), sort_keys=True) + Runtime-safe knobs never appear, so every setting of them shares one + executable. A compile-only knob left at its baseline does not appear either, + so it shares the baseline executable instead of forcing an identical rebuild. + """ + return { + path: value + for path, value in variant.items() + if path not in ov.RUNTIME_SAFE_PATHS + and not math.isclose(value, baseline[path], rel_tol=1e-12, abs_tol=1e-12) + } diff --git a/_4_OptSim/StandardSens/pipeline/standards.py b/_4_OptSim/StandardSens/pipeline/standards.py new file mode 100644 index 0000000..d542a85 --- /dev/null +++ b/_4_OptSim/StandardSens/pipeline/standards.py @@ -0,0 +1,227 @@ +"""standards.py — Run any VehicleSim standard against a compiled variant. + +SteadyStateEval, RampSteerEval and TransientEval are three questions asked of the +same compiled model, `BobLib.Experiments.Standards.VehicleSim`. `_3_StandardSim` +already builds that model once and points all three configs at it. They also +share one interface: `python -m `, a `simulation` block +naming the executable, and a `report` block whose `output_path` decides where the +PDF and its `_metrics.csv` land. So one compile per vehicle serves every +standard here, and adding a standard is one entry in `STANDARDS`. + +FourPostEval is not listed: it runs a different model (`FourPostSim`), which +means a second compile per vehicle and its own config conventions. +""" + +from __future__ import annotations + +from collections.abc import Callable, Mapping +import csv +from dataclasses import dataclass +import hashlib +import math +from pathlib import Path +import shutil +import subprocess +import sys +from typing import Any + +import yaml + +REPO_ROOT = Path(__file__).resolve().parents[3] +VEHICLE_SIM_MODEL = "BobLib.Experiments.Standards.VehicleSim" + +ConfigEdit = Callable[[dict[str, Any]], None] + + +@dataclass(frozen=True) +class Standard: + name: str + package: str # directory under _3_StandardSim, also the python package + stem: str # "_sim.py", "_config.yml", "_report.pdf" + + @property + def module(self) -> str: + return f"_3_StandardSim.{self.package}.{self.stem}_sim" + + @property + def sim_path(self) -> Path: + return REPO_ROOT / "_3_StandardSim" / self.package / f"{self.stem}_sim.py" + + @property + def config_path(self) -> Path: + return REPO_ROOT / "_3_StandardSim" / self.package / f"{self.stem}_config.yml" + + +STANDARDS: dict[str, Standard] = { + s.name: s + for s in ( + Standard("SteadyStateEval", "SteadyStateEval", "steady_state_eval"), + Standard("RampSteerEval", "RampSteerEval", "ramp_steer_eval"), + Standard("TransientEval", "TransientEval", "transient_eval"), + ) +} + + +def input_digest(standard: Standard) -> str: + """Fingerprint of what decides a standard's numbers besides the vehicle. + + The pipeline hash covers SteadyStateEval's tooling only, so results cached + for another standard are stamped with this and rerun when it changes. + """ + digest = hashlib.sha256() + for path in (standard.sim_path, standard.config_path): + digest.update(path.read_bytes()) + return digest.hexdigest() + + +def get_standard(name: str) -> Standard: + try: + return STANDARDS[name] + except KeyError: + raise KeyError( + f"{name!r} is not a VehicleSim standard OptSim can run. Known: {sorted(STANDARDS)}. " + "A standard on a different model (FourPostEval runs FourPostSim) needs its own " + "compile and is not wired in." + ) from None + + +def write_config( + standard: Standard, + *, + variant_dir: Path, + build_dir: Path, + exec_name: str = VEHICLE_SIM_MODEL, + render_report: bool = True, + max_workers: int | None = None, + edit: ConfigEdit | None = None, +) -> tuple[Path, Path]: + """Write the standard's config for one variant; return (config, metrics CSV). + + The shared standard's own config file is read, never written. `edit` is the + hook for the few callers that need more than the executable swapped. + """ + with standard.config_path.open(encoding="utf-8") as handle: + config = yaml.safe_load(handle) + if not isinstance(config, dict): + raise TypeError(f"Expected a mapping at top level: {standard.config_path}") + + simulation = config.setdefault("simulation", {}) + simulation["build_dir"] = str(build_dir) + simulation["exec_name"] = exec_name + + # Otherwise execution settings stay exactly as the standard's config has them. + if max_workers is not None: + execution = config.setdefault("execution", {}) + execution["parallel"] = True + execution["max_workers"] = int(max_workers) + + # The metrics CSV path is derived from output_path whether or not the PDF is + # rendered, so anchoring it to the variant keeps both beside its results. + results_dir = variant_dir / "results" / standard.name + report = config.setdefault("report", {}) + report["enabled"] = render_report + report["output_path"] = str(results_dir / f"{standard.stem}_report.pdf") + + if edit is not None: + edit(config) + + config_path = results_dir / f"{standard.stem}_config.generated.yml" + config_path.parent.mkdir(parents=True, exist_ok=True) + with config_path.open("w", encoding="utf-8") as handle: + yaml.safe_dump(config, handle, sort_keys=False) + return config_path, results_dir / f"{standard.stem}_report_metrics.csv" + + +def run_standard( + standard: Standard, + *, + variant_dir: Path, + build_dir: Path, + exec_name: str = VEHICLE_SIM_MODEL, + timeout: int | None = None, + render_report: bool = True, + max_workers: int | None = None, + edit: ConfigEdit | None = None, +) -> Path: + """Run one standard for one variant and return its metrics CSV.""" + config_path, metrics_csv = write_config( + standard, + variant_dir=variant_dir, + build_dir=build_dir, + exec_name=exec_name, + render_report=render_report, + max_workers=max_workers, + edit=edit, + ) + completed = subprocess.run( + [sys.executable, "-m", standard.module, str(config_path)], + cwd=str(REPO_ROOT), + capture_output=True, + text=True, + timeout=timeout, + ) + if completed.returncode != 0: + raise RuntimeError( + f"{standard.name} failed.\nConfig: {config_path}\n" + f"Stdout:\n{completed.stdout}\nStderr:\n{completed.stderr}" + ) + if not metrics_csv.exists(): + raise FileNotFoundError(f"{standard.name} produced no metrics CSV: {metrics_csv}") + + # A stable name beside the report-style one, which is what the sweep's + # batch runner and aggregator look for. + canonical = metrics_csv.with_name("metrics.csv") + shutil.copyfile(metrics_csv, canonical) + return canonical + + +def case_loss(metrics: Mapping[str, float]) -> str | None: + """Say why a run is not whole, or return None when every case settled. + + Every VehicleSim standard exports `n_cases` and `n_successful_cases`. A run + that lost cases fits its gradients through fewer points, so its metrics are + not comparable with a whole run's, and nothing downstream can tell from the + numbers alone. + """ + total = metrics.get("n_cases", math.nan) + good = metrics.get("n_successful_cases", math.nan) + if not (math.isfinite(total) and math.isfinite(good)): + return "reported no case counts" + if good < total: + return f"{int(total - good)} of {int(total)} cases failed" + return None + + +def read_metrics(path: Path) -> dict[str, tuple[float, str]]: + """Read a standard's metrics CSV as {name: (value, units)}. + + TransientEval reports some metrics once per group under the same name + (`ay_gain_dc` for the step and again for the frequency sweep). Keyed by bare + name the later row would silently replace the earlier, so a name that spans + groups is exposed only in its qualified form, `group.metric`. A row repeated + with the same value is one metric; repeated with a different value it is an + ambiguity nothing here can resolve, and an error. + """ + with path.open(newline="", encoding="utf-8") as handle: + rows = [row for row in csv.DictReader(handle) if row.get("metric")] + groups: dict[str, set[str]] = {} + for row in rows: + groups.setdefault(row["metric"], set()).add(row.get("group") or "") + + metrics: dict[str, tuple[float, str]] = {} + for row in rows: + name = row["metric"] + if len(groups[name]) > 1: + name = f"{row.get('group') or '?'}.{name}" + try: + value = float(row.get("value") or "nan") + except ValueError: + value = float("nan") + if name in metrics and not _same(metrics[name][0], value): + raise ValueError(f"{path} reports {name!r} twice with different values") + metrics[name] = (value, row.get("units") or "") + return metrics + + +def _same(a: float, b: float) -> bool: + return a == b or (a != a and b != b) # NaN equals NaN here diff --git a/_4_OptSim/StandardSens/pipeline/steady_state_eval_report.py b/_4_OptSim/StandardSens/pipeline/steady_state_eval_report.py index a55c608..bef3089 100644 --- a/_4_OptSim/StandardSens/pipeline/steady_state_eval_report.py +++ b/_4_OptSim/StandardSens/pipeline/steady_state_eval_report.py @@ -1,22 +1,16 @@ -"""steady_state_eval_report.py — Run the SteadyStateEval postprocessor for one DOE variant. +"""steady_state_eval_report.py — SteadyStateEval's own options on the standards runner. -This reuses the existing SteadyStateEval Python wrapper to generate the report-style -metrics CSV from the variant-specific compiled executable. +`standards.run_standard` runs any VehicleSim standard for a variant. This adds the +two things only SteadyStateEval callers ask for: Modelica parameter overrides +applied to every case, and a single isoline in place of the standard's four. """ from __future__ import annotations -import shutil -import subprocess -import sys from pathlib import Path from typing import Any, NamedTuple -import yaml - -STANDARD_DIR = Path(__file__).resolve().parents[1] -REPO_ROOT = STANDARD_DIR.parent.parent -BASE_CONFIG = REPO_ROOT / "_3_StandardSim/SteadyStateEval/steady_state_eval_config.yml" +from StandardSens.pipeline.standards import VEHICLE_SIM_MODEL, get_standard, run_standard class Isoline(NamedTuple): @@ -26,149 +20,46 @@ class Isoline(NamedTuple): target_ays: tuple[float, ...] -def _load_yaml(path: Path) -> dict[str, Any]: - with path.open("r", encoding="utf-8") as f: - data = yaml.safe_load(f) - if not isinstance(data, dict): - raise TypeError(f"Expected mapping at top level: {path}") - return data - - -def _dump_yaml(path: Path, data: dict[str, Any]) -> None: - path.parent.mkdir(parents=True, exist_ok=True) - with path.open("w", encoding="utf-8") as f: - yaml.safe_dump(data, f, sort_keys=False) - - -def build_report_config( - *, - variant_dir: Path, - build_dir: Path, - exec_name: str, - base_config_path: Path = BASE_CONFIG, - init_parameters: dict[str, float] | None = None, - isoline: Isoline | None = None, - max_workers: int | None = None, - render_report: bool = True, -) -> tuple[Path, Path]: - """Write a temporary SteadyStateEval config for one DOE variant. - - `init_parameters` are Modelica parameter overrides applied to every case, so - one compiled executable can stand in for a vehicle it was not compiled as. - `isoline` swaps the standard's four-isoline matrix for a single one, without - touching the shared standard's own config or its regression baselines. - `render_report=False` skips the PDF; the metrics CSV is written either way. - - Returns: - (config_path, canonical_metrics_csv_path) - """ - config = _load_yaml(base_config_path) - - simulation = config.setdefault("simulation", {}) - if not isinstance(simulation, dict): - raise TypeError("SteadyStateEval config simulation block must be a mapping") - - report = config.setdefault("report", {}) - if not isinstance(report, dict): - raise TypeError("SteadyStateEval config report block must be a mapping") - - execution = config.setdefault("execution", {}) - if not isinstance(execution, dict): - raise TypeError("SteadyStateEval config execution block must be a mapping") - - simulation["build_dir"] = str(build_dir) - simulation["exec_name"] = exec_name - - if init_parameters: - merged = dict(simulation.get("init_parameters") or {}) - merged.update({name: float(value) for name, value in init_parameters.items()}) - simulation["init_parameters"] = merged - - if isoline is not None: - # These three move together: the cap and the exported-metric velocity - # must name the one isoline being run, or the report selects nothing. - sweep = config.setdefault("sweep", {}) - sweep["testVels"] = [isoline.velocity_mps] - sweep["targetAys"] = list(isoline.target_ays) - sweep["maxAyByVelocity"] = {isoline.velocity_mps: max(isoline.target_ays)} - report["metric_target_velocity_mps"] = isoline.velocity_mps - - # Otherwise execution settings stay exactly as the standard-sim config has - # them; that config controls whether cases run serially or in parallel. - if max_workers is not None: - execution["parallel"] = True - execution["max_workers"] = int(max_workers) - - # Keep the report output location anchored to the variant so the generated - # PDF and metrics CSV live beside the DOE result artifacts. The CSV path is - # derived from output_path whether or not the PDF is rendered. - report["enabled"] = render_report - report["output_path"] = str( - variant_dir / "results" / "SteadyStateEval" / "steady_state_eval_report.pdf" - ) - - results_dir = variant_dir / "results" / "SteadyStateEval" - config_path = results_dir / "steady_state_eval_config.generated.yml" - metrics_csv = results_dir / "steady_state_eval_report_metrics.csv" - - _dump_yaml(config_path, config) - return config_path, metrics_csv - - def run_report( *, variant_dir: Path, build_dir: Path, - exec_name: str, + exec_name: str = VEHICLE_SIM_MODEL, timeout: int | None = None, - base_config_path: Path = BASE_CONFIG, init_parameters: dict[str, float] | None = None, isoline: Isoline | None = None, max_workers: int | None = None, render_report: bool = True, ) -> Path: - """Run the SteadyStateEval report wrapper and return the metrics CSV path.""" - config_path, metrics_csv = build_report_config( + """Run SteadyStateEval for one variant and return its metrics CSV. + + `init_parameters` lets one compiled executable stand in for a vehicle it was + not compiled as. `isoline` swaps the standard's four-isoline matrix for one, + without touching the shared standard's own config or regression baselines. + """ + + def edit(config: dict[str, Any]) -> None: + if init_parameters: + simulation = config["simulation"] + merged = dict(simulation.get("init_parameters") or {}) + merged.update({name: float(value) for name, value in init_parameters.items()}) + simulation["init_parameters"] = merged + if isoline is not None: + # These move together: the cap and the exported-metric velocity must + # name the one isoline being run, or the report selects nothing. + sweep = config.setdefault("sweep", {}) + sweep["testVels"] = [isoline.velocity_mps] + sweep["targetAys"] = list(isoline.target_ays) + sweep["maxAyByVelocity"] = {isoline.velocity_mps: max(isoline.target_ays)} + config["report"]["metric_target_velocity_mps"] = isoline.velocity_mps + + return run_standard( + get_standard("SteadyStateEval"), variant_dir=variant_dir, build_dir=build_dir, exec_name=exec_name, - base_config_path=base_config_path, - init_parameters=init_parameters, - isoline=isoline, - max_workers=max_workers, - render_report=render_report, - ) - - cmd = [ - sys.executable, - "-m", - "_3_StandardSim.SteadyStateEval.steady_state_eval_sim", - str(config_path), - ] - - completed = subprocess.run( - cmd, - cwd=str(REPO_ROOT), - capture_output=True, - text=True, timeout=timeout, + render_report=render_report, + max_workers=max_workers, + edit=edit, ) - - if completed.returncode != 0: - raise RuntimeError( - "SteadyStateEval report generation failed.\n" - f"Config: {config_path}\n" - f"Stdout:\n{completed.stdout}\n" - f"Stderr:\n{completed.stderr}" - ) - - if not metrics_csv.exists(): - raise FileNotFoundError( - f"Metrics CSV not produced by SteadyStateEval report: {metrics_csv}" - ) - - # Keep a stable, pipeline-friendly name alongside the report-style CSV. - canonical_metrics_csv = metrics_csv.with_name("metrics.csv") - shutil.copyfile(metrics_csv, canonical_metrics_csv) - - return canonical_metrics_csv diff --git a/_4_OptSim/StandardSens/pipeline/trade.py b/_4_OptSim/StandardSens/pipeline/trade.py new file mode 100644 index 0000000..dcbdc34 --- /dev/null +++ b/_4_OptSim/StandardSens/pipeline/trade.py @@ -0,0 +1,217 @@ +"""trade.py — Compare named vehicles across standard-sim metrics. + +A trade study asks "what does this change buy, and what does it cost", across +several metrics at once. This module turns simulated metrics into that +comparison and nothing more. There is deliberately no score and no ranking: how +much understeer is worth how much roll is the engineer's call, and a weighted sum +would hide that call inside a number. + +What it does insist on is that a difference be worth reading: + +- Every metric may carry a `resolution`, the smallest change worth acting on. A + smaller delta is still shown, marked as not resolved. +- A vehicle whose simulation lost cases is not comparable. Its fits run through + fewer points than the baseline's, so a delta would mix the design change with + the missing data. +- Where one candidate is exactly two others combined, the interaction is + reported: whether the two changes stack, or the combination does something + neither predicts. + +Nothing here touches the filesystem or a simulator. +""" + +from __future__ import annotations + +from dataclasses import dataclass +import difflib +from itertools import combinations +import math + +from StandardSens.pipeline.standards import case_loss + +BASELINE = "baseline" + +Variant = dict[str, float] +# candidate -> standard -> metric -> (value, units) +Results = dict[str, dict[str, dict[str, tuple[float, str]]]] + +RESOLVED = "resolved" +BELOW_RESOLUTION = "below resolution" +NOT_COMPARABLE = "not comparable" + + +@dataclass(frozen=True) +class MetricSpec: + standard: str + name: str + resolution: float | None = None + + +@dataclass(frozen=True) +class Comparison: + spec: MetricSpec + units: str + candidate: str + baseline: float + value: float + verdict: str # RESOLVED, BELOW_RESOLUTION, NOT_COMPARABLE, or "" with no resolution + + @property + def delta(self) -> float: + return self.value - self.baseline + + +@dataclass(frozen=True) +class Interaction: + """How far a combined change is from the sum of its two parts.""" + + spec: MetricSpec + combined: str + parts: tuple[str, str] + value: float # delta(combined) - delta(part A) - delta(part B) + + @property + def stacks(self) -> bool | None: + if self.spec.resolution is None: + return None + return abs(self.value) < self.spec.resolution + + +def lost_cases(results: Results) -> dict[tuple[str, str], str]: + """Return {(candidate, standard): why} for every run that is not whole.""" + problems: dict[tuple[str, str], str] = {} + for candidate, by_standard in results.items(): + for standard, metrics in by_standard.items(): + why = case_loss({name: value for name, (value, _units) in metrics.items()}) + if why: + problems[candidate, standard] = why + return problems + + +def compare(results: Results, specs: list[MetricSpec]) -> list[Comparison]: + """Compare every candidate with the baseline on every requested metric.""" + if BASELINE not in results: + raise KeyError(f"results must include {BASELINE!r}") + problems = lost_cases(results) + + comparisons = [] + for spec in specs: + reference, units = _lookup(results, BASELINE, spec) + for candidate in results: + if candidate == BASELINE: + continue + value, _ = _lookup(results, candidate, spec) + whole = (candidate, spec.standard) not in problems and ( + BASELINE, + spec.standard, + ) not in problems + if not (whole and math.isfinite(value) and math.isfinite(reference)): + verdict = NOT_COMPARABLE + elif spec.resolution is None: + verdict = "" + else: + verdict = RESOLVED if abs(value - reference) >= spec.resolution else BELOW_RESOLUTION + comparisons.append(Comparison(spec, units, candidate, reference, value, verdict)) + return comparisons + + +def _lookup(results: Results, candidate: str, spec: MetricSpec) -> tuple[float, str]: + metrics = results[candidate].get(spec.standard) + if metrics is None: + raise KeyError(f"{candidate!r} has no {spec.standard} results") + if spec.name not in metrics: + close = difflib.get_close_matches(spec.name, metrics, n=4, cutoff=0.5) + qualified = sorted(m for m in metrics if m.endswith("." + spec.name)) + hint = ( + f" It is reported once per group, so name one of: {qualified}." + if qualified + else f" Closest: {close}." if close else "" + ) + raise KeyError(f"{spec.standard} has no metric {spec.name!r}.{hint}") + return metrics[spec.name] + + +def interactions(candidates: dict[str, Variant], comparisons: list[Comparison]) -> list[Interaction]: + """Find candidates that are two others combined and measure how they stack. + + `both = a + b` qualifies when `a` and `b` change disjoint variables and `both` + makes exactly their changes. The interaction is then the part of `both`'s + effect that neither `a` nor `b` accounts for. + """ + deltas = { + (c.candidate, c.spec): c.delta for c in comparisons if c.verdict != NOT_COMPARABLE + } + specs = list(dict.fromkeys(c.spec for c in comparisons)) + + found = [] + for combined, changes in candidates.items(): + for a, b in combinations((n for n in candidates if n != combined), 2): + disjoint = not set(candidates[a]) & set(candidates[b]) + if not (disjoint and {**candidates[a], **candidates[b]} == changes): + continue + for spec in specs: + parts = [deltas.get((name, spec)) for name in (combined, a, b)] + if None not in parts: + found.append(Interaction(spec, combined, (a, b), parts[0] - parts[1] - parts[2])) + return found + + +def to_markdown( + name: str, + candidates: dict[str, Variant], + comparisons: list[Comparison], + found: list[Interaction], + problems: dict[tuple[str, str], str], +) -> str: + lines = [f"# Trade study: {name}", "", "| Candidate | Changes from baseline |", "| --- | --- |"] + for candidate, changes in candidates.items(): + listed = ", ".join(f"`{path}` = {value:g}" for path, value in changes.items()) + lines.append(f"| {candidate} | {listed} |") + + if problems: + lines += ["", "**Not comparable** — these runs lost cases, so their fits rest on fewer points:", ""] + lines += [f"- {c} / {s}: {why}" for (c, s), why in sorted(problems.items())] + + names = list(candidates) + for standard in dict.fromkeys(c.spec.standard for c in comparisons): + rows = [c for c in comparisons if c.spec.standard == standard] + lines += ["", f"## {standard}", ""] + lines.append("| Metric | Units | Resolution | baseline | " + " | ".join(names) + " |") + lines.append("| --- | --- | --- | --- | " + " | ".join("---" for _ in names) + " |") + for spec in dict.fromkeys(c.spec for c in rows): + cells = {c.candidate: _cell(c) for c in rows if c.spec == spec} + first = next(c for c in rows if c.spec == spec) + resolution = "" if spec.resolution is None else f"{spec.resolution:g}" + lines.append( + f"| {spec.name} | {first.units} | {resolution} | {first.baseline:.4g} | " + + " | ".join(cells[n] for n in names) + + " |" + ) + + if found: + lines += ["", "## Do the changes stack?", ""] + lines.append("| Combined | Parts | Standard | Metric | Interaction | |") + lines.append("| --- | --- | --- | --- | --- | --- |") + for i in found: + says = {None: "", True: "additive within resolution", False: "**interacts**"}[i.stacks] + lines.append( + f"| {i.combined} | {' + '.join(i.parts)} | {i.spec.standard} | {i.spec.name} | " + f"{i.value:+.4g} | {says} |" + ) + + lines += [ + "", + "Each cell is the simulated value with its change from baseline. `~` marks a change " + "smaller than the metric's resolution: reported, but too small to act on. `n/c` marks " + "a run that is not comparable. The interaction is the combined change minus the sum of " + "its parts. Below the resolution it cannot be told apart from zero, which is a weaker " + "claim than saying the changes add: read the number, not just the label.", + ] + return "\n".join(lines) + "\n" + + +def _cell(comparison: Comparison) -> str: + if comparison.verdict == NOT_COMPARABLE: + return "n/c" + mark = " ~" if comparison.verdict == BELOW_RESOLUTION else "" + return f"{comparison.value:.4g} ({comparison.delta:+.3g}){mark}" diff --git a/_4_OptSim/StandardSens/pipeline/variants.py b/_4_OptSim/StandardSens/pipeline/variants.py new file mode 100644 index 0000000..3601246 --- /dev/null +++ b/_4_OptSim/StandardSens/pipeline/variants.py @@ -0,0 +1,161 @@ +"""variants.py — A cache of compiled vehicles, addressed by what was changed. + +The sweep numbers its variants in sampling order, which is right for a population +that is generated once. The solver and the trade study instead ask for vehicles +by content — "the baseline with this toe", "the car with the stiff rear bar" — +repeatedly and in no fixed order, and a compile is the most expensive thing either +does. This store maps each distinct set of changes to one compiled variant +directory, compiles whatever is missing in a single parallel batch, and throws +everything away when BobLib, the vehicle or the simulation tooling changes. + +It reuses the sweep's generator and compiler unchanged, so a variant here is +compiled exactly as the same variant in a sweep would be. +""" + +from __future__ import annotations + +import json +from pathlib import Path +import shutil +from typing import Any + +import yaml + +from StandardSens.pipeline._pipeline_hash import check_pipeline_hash +from StandardSens.pipeline.compiler import ( + PIPELINE_TOOLING_INPUTS, + compile_all, + find_exe, + load_compiler_config, +) +from StandardSens.pipeline.generate_configs import SWEEP_SCOPE_ALL, refresh_doe_config +from StandardSens.pipeline.generator import generate_variants +from StandardSens.pipeline.orchestration import ARCHITECTURE_CONFIG, COMPILER_CONFIG, DOE_CONFIG + +# Every VehicleSim standard runs the executable the sweep builds for this one. +BUILD_STANDARD = "SteadyStateEval" + +Variant = dict[str, float] + + +def split_cpus(n_jobs: int, cpus: int, min_workers: int = 4) -> tuple[int, int]: + """Return (jobs to run at once, case workers for each). + + Simulation throughput on a 12-CPU container was 1.9x higher with every CPU + busy than with four, so concurrent jobs are preferred until each would drop + below `min_workers`; a lone job gets every CPU. + """ + concurrent = max(1, min(n_jobs, cpus // min_workers)) + return concurrent, max(1, cpus // concurrent) + + +def variant_key(variant: Variant) -> str: + """Canonical text for a set of changes; float noise must not split an entry.""" + return json.dumps({p: float(f"{float(v):.12g}") for p, v in sorted(variant.items())}) + + +class VariantStore: + def __init__(self, root: Path) -> None: + self.root = root.resolve() + self.root.mkdir(parents=True, exist_ok=True) + self.variants_dir = self.root / "variants" + self.doe_config_path = self.root / "_doe_config.yaml" + self.doe_config = self._write_own_doe_config() + + compiler_cfg = load_compiler_config(COMPILER_CONFIG) + self.standard_cfg: dict[str, Any] = compiler_cfg["standards"][BUILD_STANDARD] + self._boblib_path = ( + COMPILER_CONFIG.resolve().parent / compiler_cfg["boblib_path"] + ).resolve() + + # A cache must never outlive the inputs it was built from. + self.was_reset = self.variants_dir.exists() and self._inputs_changed() + if self.was_reset: + print( + f"Cache under {self.root.name}/ predates a change to BobLib, the vehicle, or " + "the simulation tooling; discarding it." + ) + shutil.rmtree(self.variants_dir) + index_path = self.variants_dir / "index.json" + self._index: dict[str, int] = json.loads(index_path.read_text()) if index_path.exists() else {} + + def _write_own_doe_config(self) -> dict[str, Any]: + """Generate a DOE config here rather than over the sweep's. + + The sweep's `_doe_config.yaml` is committed and records the scope its + population was built at, which `opt-search` relies on. Consumers of this + store need every variable's spec whatever that scope was, so they keep + their own copy. The generated file names the baseline record and vehicle + template relative to the sweep's config directory; here they are made + absolute so they resolve from anywhere. + """ + cfg = refresh_doe_config( + architecture_config_path=ARCHITECTURE_CONFIG, + compiler_config_path=COMPILER_CONFIG, + doe_config_path=self.doe_config_path, + scope=SWEEP_SCOPE_ALL, + ) + config_dir = DOE_CONFIG.resolve().parent + cfg["baseline_mo"] = (config_dir / cfg["baseline_mo"]).resolve().as_posix() + cfg["architecture"]["template"] = ( + (config_dir.parent / cfg["architecture"]["template"]).resolve().as_posix() + ) + self.doe_config_path.write_text(yaml.safe_dump(cfg, sort_keys=False)) + return cfg + + def _inputs_changed(self) -> bool: + """Whether the inputs differ from the ones the compiled variants were built from.""" + try: + # The same call compile_all makes, so the two cannot disagree about + # what counts as stale. + check_pipeline_hash( + self.variants_dir, + self.doe_config_path, + COMPILER_CONFIG, + self._boblib_path, + ARCHITECTURE_CONFIG, + PIPELINE_TOOLING_INPUTS, + ) + except RuntimeError: + return True + return False + + def variant_dir(self, variant: Variant) -> Path: + return self.variants_dir / f"variant_{self._index[variant_key(variant)]:04d}" + + def build_dir(self, variant: Variant) -> Path: + return self.variant_dir(variant) / "build" / BUILD_STANDARD + + def ensure_compiled(self, variants: list[Variant]) -> None: + """Compile whichever of these vehicles has no executable yet, in one batch.""" + new = sorted({variant_key(v) for v in variants} - set(self._index)) + if not new and all(find_exe(self.build_dir(v), self.standard_cfg) for v in variants): + return + + if self.variants_dir.exists() and self._inputs_changed(): + # Only reachable mid-run: a stale cache is discarded at construction. + raise RuntimeError( + "BobLib, the vehicle or the simulation tooling changed while this run was in " + "progress, so the vehicles already simulated and the ones still to compile " + "would not be comparable. Nothing on disk is damaged: run it again once the " + "edits have settled, and the cache will be rebuilt against the current inputs." + ) + + for key in new: + self._index[key] = len(self._index) + self.variants_dir.mkdir(parents=True, exist_ok=True) + ordered = sorted(self._index, key=self._index.__getitem__) + generate_variants(self.doe_config_path, [json.loads(key) for key in ordered], self.variants_dir) + (self.variants_dir / "index.json").write_text(json.dumps(self._index, indent=1)) + compile_all( + self.variants_dir, + COMPILER_CONFIG, + doe_config_path=self.doe_config_path, + only_standards=(BUILD_STANDARD,), + ) + + for variant in variants: + if find_exe(self.build_dir(variant), self.standard_cfg) is None: + log = self.variant_dir(variant) / f"compile_error_{BUILD_STANDARD}.log" + detail = log.read_text()[-2000:] if log.exists() else "no compile log written" + raise RuntimeError(f"Compile failed for {variant or 'the baseline'}:\n{detail}") diff --git a/_4_OptSim/StandardSens/trade_study.py b/_4_OptSim/StandardSens/trade_study.py new file mode 100644 index 0000000..306f7c7 --- /dev/null +++ b/_4_OptSim/StandardSens/trade_study.py @@ -0,0 +1,198 @@ +"""Compare named vehicles across the standard sims. + + make opt-trade # configs/trade_study.yaml + make opt-trade STUDY=path/to/another_study.yaml + +OptSim asks three different questions of the same compiled vehicles: + +- `opt-standard` samples many vehicles to learn which parameters matter. +- `opt-solve` inverts for the setup that hits target metrics. +- `opt-trade`, this one, compares vehicles you name, on metrics you choose. + +Every candidate is compiled rather than overridden, so any swept variable is fair +game — mass, CG and toe included — and SteadyStateEval, RampSteerEval and +TransientEval all run against that one executable. See `pipeline/trade.py` for +what the comparison insists on and why it offers no score. +""" + +from __future__ import annotations + +import argparse +from concurrent.futures import ThreadPoolExecutor +import csv +from pathlib import Path +import sys +import time +from typing import Any + +import yaml + +from _shared.console import elapsed +from StandardSens.pipeline import trade +from StandardSens.pipeline.orchestration import RESULTS_DIR, STANDARD_BUILD_DIR +from StandardSens.pipeline.standards import ( + Standard, + get_standard, + input_digest, + read_metrics, + run_standard, +) +from StandardSens.pipeline.variants import VariantStore, split_cpus + +DEFAULT_STUDY = Path(__file__).resolve().parent / "configs/trade_study.yaml" +# One store for every study: the baseline, and any candidate two studies share, +# is compiled and simulated once. +TRADE_DIR = STANDARD_BUILD_DIR / "trade" +RUN_TIMEOUT_S = 3600 + +Variant = dict[str, float] + + +def load_candidates(study: dict[str, Any], variables: dict[str, dict[str, Any]]) -> dict[str, Variant]: + candidates: dict[str, Variant] = {} + for name, changes in (study.get("candidates") or {}).items(): + if name == trade.BASELINE: + raise ValueError(f"{trade.BASELINE!r} is always included; name your candidate something else") + if not changes: + raise ValueError(f"Candidate {name!r} changes nothing, so it is the baseline") + unknown = sorted(set(changes) - set(variables)) + if unknown: + raise KeyError( + f"Candidate {name!r} changes {unknown}, which are not variables in " + "vehicle_architecture.yaml. A parameter has to be declared there, with the " + "Modelica record block it maps to, before a study can change it." + ) + candidates[name] = {path: float(value) for path, value in changes.items()} + for path, value in candidates[name].items(): + low, high = variables[path]["range"] + if not low <= value <= high: + print( + f"note: {name}.{path} = {value:g} is outside the sweep range " + f"[{low:g}, {high:g}]. Allowed, but nothing has been simulated out there." + ) + if not candidates: + raise ValueError("The study names no candidates") + return candidates + + +def load_specs(study: dict[str, Any]) -> list[trade.MetricSpec]: + specs = [] + for standard, metrics in (study.get("metrics") or {}).items(): + get_standard(standard) + for name, options in (metrics or {}).items(): + resolution = (options or {}).get("resolution") + specs.append( + trade.MetricSpec(standard, name, None if resolution is None else float(resolution)) + ) + if not specs: + raise ValueError("The study names no metrics") + return specs + + +def _stamp(variant_dir: Path, standard: Standard) -> Path: + return variant_dir / "results" / standard.name / ".input_digest" + + +def is_cached(variant_dir: Path, standard: Standard, render_reports: bool) -> bool: + results_dir = variant_dir / "results" / standard.name + stamp = _stamp(variant_dir, standard) + return ( + (results_dir / "metrics.csv").exists() + and stamp.exists() + and stamp.read_text() == input_digest(standard) + and (not render_reports or (results_dir / f"{standard.stem}_report.pdf").exists()) + ) + + +def run(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--study", type=Path, default=DEFAULT_STUDY) + args = parser.parse_args(argv) + + study = yaml.safe_load(args.study.read_text()) + name = str(study.get("name") or args.study.stem) + specs = load_specs(study) + standards = [get_standard(s) for s in dict.fromkeys(spec.standard for spec in specs)] + render_reports = bool(study.get("render_reports", True)) + + store = VariantStore(TRADE_DIR) + variables = {v["path"]: v for v in store.doe_config["variables"]} + candidates = load_candidates(study, variables) + vehicles: dict[str, Variant] = {trade.BASELINE: {}, **candidates} + + started = time.time() + store.ensure_compiled(list(vehicles.values())) + + jobs = [ + (vehicle, standard) + for vehicle in vehicles.values() + for standard in standards + if not is_cached(store.variant_dir(vehicle), standard, render_reports) + ] + if jobs: + concurrent, workers = split_cpus(len(jobs), int(study.get("cpus", 8))) + total = len(vehicles) * len(standards) + print( + f"Simulating {len(jobs)} of {total} vehicle x standard run(s), {concurrent} at a " + f"time with {workers} case workers each ({total - len(jobs)} cached)...", + flush=True, + ) + + def simulate(job: tuple[Variant, Standard]) -> str: + vehicle, standard = job + variant_dir = store.variant_dir(vehicle) + run_standard( + standard, + variant_dir=variant_dir, + build_dir=store.build_dir(vehicle), + exec_name=store.standard_cfg["model"], + timeout=RUN_TIMEOUT_S, + render_report=render_reports, + max_workers=workers, + ) + _stamp(variant_dir, standard).write_text(input_digest(standard)) + return standard.name + + with ThreadPoolExecutor(max_workers=concurrent) as pool: + for done, finished in enumerate(pool.map(simulate, jobs), start=1): + print(f" [{done}/{len(jobs)}] {finished} {time.time() - started:.0f}s", flush=True) + + results: trade.Results = { + label: { + standard.name: read_metrics( + store.variant_dir(vehicle) / "results" / standard.name / "metrics.csv" + ) + for standard in standards + } + for label, vehicle in vehicles.items() + } + comparisons = trade.compare(results, specs) + problems = trade.lost_cases(results) + report = trade.to_markdown( + name, candidates, comparisons, trade.interactions(candidates, comparisons), problems + ) + + output_dir = RESULTS_DIR / "trade" + output_dir.mkdir(parents=True, exist_ok=True) + (output_dir / f"{name}.md").write_text(report, encoding="utf-8") + with (output_dir / f"{name}.csv").open("w", newline="", encoding="utf-8") as handle: + writer = csv.writer(handle) + writer.writerow( + ["standard", "metric", "units", "resolution", "candidate", "baseline", "value", "delta", "verdict"] + ) + for c in comparisons: + writer.writerow( + [c.spec.standard, c.spec.name, c.units, c.spec.resolution, c.candidate, + c.baseline, c.value, c.delta, c.verdict] + ) + + print("\n" + report) + print(f"Total {elapsed(started)}") + print(f"Report: {output_dir / f'{name}.md'}") + if render_reports: + print(f"Per-vehicle PDF reports: {store.variants_dir}/variant_*/results/") + return 2 if problems else 0 + + +if __name__ == "__main__": + sys.exit(run()) diff --git a/docs/doe-reverse-engineering.md b/docs/doe-reverse-engineering.md index 5305936..2296664 100644 --- a/docs/doe-reverse-engineering.md +++ b/docs/doe-reverse-engineering.md @@ -2,6 +2,22 @@ **TL;DR:** Sweep a population of vehicle variants → simulate them all → aggregate metrics → search backwards. Given target performance numbers, find the car that hits them. Everything here lives under `_4_OptSim/StandardSens/`. +## Three questions, one set of compiled vehicles + +OptSim asks three different things, and they are separate tools rather than +stages of one: + +| Target | Question | Vehicles | Answer | +| --- | --- | --- | --- | +| `make opt-standard` (+ `opt-refined`, `opt-search`) | Which parameters matter, and roughly where is a car with these numbers? | many, sampled | sensitivities, response surfaces, nearest sampled car | +| `make opt-solve` | What do I set on *this* car to hit these numbers? | a star of `2n + 1` around the current car | one setup, simulated | +| `make opt-trade` | What does each of these specific changes buy, and cost? | the ones you name | a comparison table | + +None replaces another. The sweep is still how you learn a design space; the +solver and the trade study are what you reach for once you have a specific +question. All three write their vehicles with the same generator and compile them +with the same compiler, so a variant means the same thing in each. + ## Quick start: a small sweep From a clean checkout, this is the whole thing: @@ -425,6 +441,84 @@ verification), and each later solve against different targets took about 70 seconds. A four-variant sweep of the same car took 26 minutes and cannot interpolate between its samples. +## Trade studies + +A trade study compares vehicles you name, on metrics you choose, across more than +one standard sim: + +```bash +make opt-trade # configs/trade_study.yaml +make opt-trade STUDY=path/to/another_study.yaml +``` + +The study file names candidates as changes from the baseline, and the metrics to +compare per standard: + +```yaml +name: rear_roll_stiffness +candidates: + stiff_rear_bar: {rear.stabar.rate_n_m_per_rad: 961.495352} + soft_front_spring: {front.actuation.spring_rate_n_per_m: 21015.2202} + both: {rear.stabar.rate_n_m_per_rad: 961.495352, + front.actuation.spring_rate_n_per_m: 21015.2202} +metrics: + SteadyStateEval: + understeer_gradient_deg_per_g: {resolution: 0.02} + TransientEval: + yaw_rise_time_s: {resolution: 0.005} +``` + +The report lands in `_4_OptSim/results/trade/.md` and `.csv`: one table per +standard, each cell the simulated value with its change from baseline. + +**Every candidate is compiled, never overridden.** That is the opposite choice +from `opt-solve`, for a reason: a trade study wants to change mass, CG, toe and +camber, which are exactly the parameters an override silently fails to move (see +above). A compile is always correct, costs about 36 s per vehicle when several +build at once, and is cached by content, so the baseline and any candidate two +studies share are built and simulated once across all of them. + +**One compile serves every standard.** SteadyStateEval, RampSteerEval and +TransientEval all run the same model, `BobLib.Experiments.Standards.VehicleSim`; +`_3_StandardSim` itself builds it once and points all three configs at it. They +share one interface too, so a standard is one entry in `STANDARDS` in +`pipeline/standards.py`. FourPostEval is not there: it runs `FourPostSim`, a +different model, so it would need a second compile per vehicle. + +**There is no score and no ranking, on purpose.** How much understeer is worth how +much settling time is an engineering judgement, and a weighted sum would bury it +inside a number. What the report does insist on is that a difference be worth +reading: + +- **Resolution.** Each metric may state the smallest change worth acting on. A + smaller delta is still shown, marked `~`, so a 0.001 s change in rise time is + not read as a finding. Leave it off and the delta is reported unjudged. +- **Lost cases.** A run whose simulation lost cases is marked `n/c` and not + compared. Its fits run through fewer points than the baseline's, so the delta + would mix the design change with the missing data. A baseline that lost cases + makes every comparison on that standard `n/c`. The command exits 2. +- **Stacking.** When one candidate makes exactly the changes of two others, the + report gives the interaction: the combined effect minus the sum of the parts. + Above the metric's resolution the combination does something neither part + predicts, and "we'll do both" is not the sum you budgeted for. Below it the + interaction cannot be told apart from zero, which is weaker than saying the + changes add, so the number is printed beside the label. +- **Ambiguous names.** TransientEval reports some metrics once per group under + one name (`yaw_gain_dc` for the step and again for the frequency sweep). Those + are only available qualified, `step.yaw_gain_dc`, and asking for the bare name + is an error that lists the options. + +**A candidate can only change a declared variable.** The variables are the +`sweep.variables` entries in `configs/vehicle_architecture.yaml`, because each +needs the Modelica record block it maps to. To trade on something new, such as +wheelbase, declare it there first (see *Adding a swept parameter*). A value outside +a variable's sweep range is allowed and noted: the range bounds what the sweep +samples, not what is physically valid. + +Compare like with like: a metric here comes from each standard's own test matrix, +so a gradient from `opt-trade` matches `make standard-eval-steady-state`, not +`opt-solve`, which fits through its own denser isoline. + ## Envelope sensitivities `_4_OptSim/EnvelopeSens/` is the same idea against GGV/YMD envelope outputs diff --git a/makefile b/makefile index 34493d0..0639637 100644 --- a/makefile +++ b/makefile @@ -20,6 +20,7 @@ BUILD_FOUR_POST_MOS := _3_StandardSim/build_four_post_sim.mos SEARCH_TOP ?= 1 TARGETS ?= KNOBS ?= +STUDY ?= REDUCED_DOF ?= 6 REDUCED_KINEMATICS ?= lookup REDUCED_BOBLIB_CSV ?= @@ -137,7 +138,7 @@ CLEAN_DOCKER_IMAGE ?= bobdyn/bobsim:latest lap-validation-visuals \ envelope-ggv envelope-ymd envelope-all \ opt-standard opt-standard-setup opt-standard-architecture \ - opt-envelope opt-refined opt-search opt-solve opt-doe-smoke \ + opt-envelope opt-refined opt-search opt-solve opt-trade opt-doe-smoke \ clean clean-app clean-visual clean-standard clean-envelope clean-opt clean-owned clean-all ifeq ($(OS),Windows_NT) @@ -226,12 +227,14 @@ help: ' opt-refined Run StandardSens refined response surfaces' \ ' opt-search Reverse lookup: target metrics -> vehicle parameters' \ ' opt-solve Solve for the setup that hits target metrics, then simulate it' \ + ' opt-trade Compare named vehicles across the standard sims' \ '' \ ' Search variables:' \ ' METRICS="NAME=VALUE ..." Required target metrics' \ ' SEARCH_TOP= Nearest variants to return. Default: 1' \ ' TARGETS="NAME=VALUE ..." opt-solve targets. Default: configs/solve_config.yaml' \ ' KNOBS="path ..." opt-solve knobs. Default: configs/solve_config.yaml' \ + ' STUDY= opt-trade study. Default: configs/trade_study.yaml' \ '' \ ' DOE sweep variables (default: configs/vehicle_architecture.yaml):' \ ' DOE_METHOD=lhs|interval_splice Sampling method' \ @@ -506,6 +509,12 @@ opt-search: opt-solve: $(FOUR_POST_METRICS) $(RUN) env PYTHONPATH=$(WORKSPACE)/_4_OptSim:$(WORKSPACE) $(PYTHON) -m StandardSens.solve_setup $(if $(TARGETS),--targets $(TARGETS),) $(if $(KNOBS),--knobs $(KNOBS),) +# Compiles each named candidate once and runs every requested standard against +# that one executable. Needs the FourPostEval motion ratios whenever a candidate +# changes a spring rate, for the same reason opt-standard does. +opt-trade: $(FOUR_POST_METRICS) + $(RUN) env PYTHONPATH=$(WORKSPACE)/_4_OptSim:$(WORKSPACE) $(PYTHON) -m StandardSens.trade_study $(if $(STUDY),--study $(STUDY),) + clean: bash -lc 'find $(CLEAN_WORKSPACE) -type d -name "__pycache__" -exec rm -rf {} + 2>/dev/null; \ find $(CLEAN_WORKSPACE) -type f \( -name "*.pyc" -o -name "*.pyo" \) -delete 2>/dev/null; \ diff --git a/tests/test_optsim_overrides.py b/tests/test_optsim_overrides.py index 3a5c633..035609d 100644 --- a/tests/test_optsim_overrides.py +++ b/tests/test_optsim_overrides.py @@ -130,11 +130,11 @@ def test_runtime_safe_knobs_share_one_executable_and_the_rest_do_not() -> None: assert bar in overrides.RUNTIME_SAFE_PATHS assert toe not in overrides.RUNTIME_SAFE_PATHS, "toe is baked into the wheel rotation matrix" baseline = {bar: 535.0, toe: 0.0} - soft = evaluator.compile_key({bar: 400.0, toe: 0.0}, baseline) - stiff = evaluator.compile_key({bar: 900.0, toe: 0.0}, baseline) - toed = evaluator.compile_key({bar: 900.0, toe: 0.1}, baseline) - assert soft == stiff == (), "bar changes must reuse the baseline executable" - assert toed == ((toe, 0.1),), "a toe change must get its own executable" + soft = evaluator.compiled_part({bar: 400.0, toe: 0.0}, baseline) + stiff = evaluator.compiled_part({bar: 900.0, toe: 0.0}, baseline) + toed = evaluator.compiled_part({bar: 900.0, toe: 0.1}, baseline) + assert soft == stiff == {}, "bar changes must reuse the baseline executable" + assert toed == {toe: 0.1}, "a toe change must get its own executable" def test_command_line_targets_replace_the_configured_ones() -> None: diff --git a/tests/test_optsim_trade.py b/tests/test_optsim_trade.py new file mode 100644 index 0000000..f85abdb --- /dev/null +++ b/tests/test_optsim_trade.py @@ -0,0 +1,207 @@ +"""The trade study's comparison logic, and the multi-standard plumbing under it. + +A trade study is only worth running if its table can be trusted, so these pin the +ways a comparison goes quietly wrong: a delta too small to mean anything read as +a finding, a run that lost cases compared as if it were whole, two changes +assumed to add when they do not, and a metric name that means two things. +""" + +from __future__ import annotations + +from pathlib import Path +import sys + +import pytest + +ROOT = Path(__file__).resolve().parents[1] +OPTSIM_DIR = ROOT / "_4_OptSim" +if str(OPTSIM_DIR) not in sys.path: + sys.path.insert(0, str(OPTSIM_DIR)) + +pytest.importorskip("scipy", reason="the OptSim pipeline package imports scipy") + +from StandardSens import trade_study # noqa: E402 +from StandardSens.pipeline import standards, trade # noqa: E402 +from StandardSens.pipeline.variants import split_cpus, variant_key # noqa: E402 + +UNDERSTEER = trade.MetricSpec("SteadyStateEval", "understeer", resolution=0.02) +RISE = trade.MetricSpec("TransientEval", "yaw_rise_time_s", resolution=0.005) +WHOLE = {"n_cases": (8.0, "count"), "n_successful_cases": (8.0, "count")} + + +def run(understeer: float, rise: float, **overrides) -> dict[str, dict[str, tuple[float, str]]]: + return { + "SteadyStateEval": {**WHOLE, "understeer": (understeer, "deg/g"), **overrides}, + "TransientEval": {**WHOLE, "yaw_rise_time_s": (rise, "s")}, + } + + +def test_a_delta_below_the_resolution_is_shown_but_not_called_a_finding() -> None: + results = {"baseline": run(0.300, 0.055), "bar": run(0.345, 0.056)} + by_metric = {c.spec.name: c for c in trade.compare(results, [UNDERSTEER, RISE])} + assert by_metric["understeer"].verdict == trade.RESOLVED + assert by_metric["understeer"].delta == pytest.approx(0.045) + assert by_metric["yaw_rise_time_s"].verdict == trade.BELOW_RESOLUTION + assert by_metric["yaw_rise_time_s"].delta == pytest.approx(0.001), "still reported" + + +def test_a_run_that_lost_cases_is_not_compared() -> None: + """Its fits run through fewer points, so the delta would mix two causes.""" + lossy = run(0.40, 0.055, n_successful_cases=(6.0, "count")) + results = {"baseline": run(0.30, 0.055), "bar": lossy} + assert trade.lost_cases(results) == {("bar", "SteadyStateEval"): "2 of 8 cases failed"} + by_metric = {c.spec.name: c for c in trade.compare(results, [UNDERSTEER, RISE])} + assert by_metric["understeer"].verdict == trade.NOT_COMPARABLE + assert by_metric["yaw_rise_time_s"].verdict != trade.NOT_COMPARABLE, "other standard is fine" + + +def test_a_lossy_baseline_poisons_every_comparison_on_that_standard() -> None: + results = {"baseline": run(0.30, 0.055, n_successful_cases=(7.0, "count")), "bar": run(0.35, 0.040)} + verdicts = {c.spec.name: c.verdict for c in trade.compare(results, [UNDERSTEER, RISE])} + assert verdicts == {"understeer": trade.NOT_COMPARABLE, "yaw_rise_time_s": trade.RESOLVED} + + +def test_interaction_is_found_for_a_candidate_that_is_two_others_combined() -> None: + candidates = {"bar": {"rear.bar": 900.0}, "spring": {"front.spring": 21000.0}, + "both": {"rear.bar": 900.0, "front.spring": 21000.0}, + "unrelated": {"rear.bar": 700.0}} + results = { + "baseline": run(0.300, 0.055), + "bar": run(0.260, 0.055), # -0.040 + "spring": run(0.290, 0.055), # -0.010 + "both": run(0.200, 0.055), # -0.100, not the -0.050 the parts predict + "unrelated": run(0.280, 0.055), + } + found = trade.interactions(candidates, trade.compare(results, [UNDERSTEER])) + assert [(i.combined, i.parts) for i in found] == [("both", ("bar", "spring"))] + assert found[0].value == pytest.approx(-0.050) + assert found[0].stacks is False, "0.05 of unexplained understeer is five resolutions" + + +def test_changes_that_simply_add_are_reported_as_stacking() -> None: + candidates = {"a": {"x": 1.0}, "b": {"y": 1.0}, "ab": {"x": 1.0, "y": 1.0}} + results = {"baseline": run(0.30, 0.055), "a": run(0.26, 0.055), "b": run(0.29, 0.055), + "ab": run(0.251, 0.055)} + (found,) = trade.interactions(candidates, trade.compare(results, [UNDERSTEER])) + assert found.value == pytest.approx(0.001) and found.stacks is True + + +def test_overlapping_changes_are_not_mistaken_for_a_combination() -> None: + """`ab` sets x to a different value than `a` does, so it is not a + b.""" + candidates = {"a": {"x": 1.0}, "b": {"y": 1.0}, "ab": {"x": 2.0, "y": 1.0}} + results = {n: run(0.3, 0.055) for n in ("baseline", "a", "b", "ab")} + assert trade.interactions(candidates, trade.compare(results, [UNDERSTEER])) == [] + + +def test_an_unknown_metric_names_what_was_probably_meant() -> None: + results = {"baseline": run(0.3, 0.055), "bar": run(0.3, 0.055)} + with pytest.raises(KeyError, match="understeer"): + trade.compare(results, [trade.MetricSpec("SteadyStateEval", "understeer_grad")]) + + +def test_a_name_reported_per_group_must_be_asked_for_by_group(tmp_path: Path) -> None: + csv_path = tmp_path / "metrics.csv" + csv_path.write_text( + "standard,group,metric,value,units,description\n" + "T,step,yaw_gain_dc,2.10,(rad/s)/rad,\n" + "T,frequency,yaw_gain_dc,2.30,(rad/s)/rad,\n" + "T,step,yaw_overshoot_pct,16.6,%,\n" + "T,step,yaw_overshoot_pct,16.6,%,\n" # TransientEval really does write this twice + ) + metrics = standards.read_metrics(csv_path) + assert "yaw_gain_dc" not in metrics, "the bare name would silently mean the last row" + assert metrics["step.yaw_gain_dc"][0] == 2.10 and metrics["frequency.yaw_gain_dc"][0] == 2.30 + assert metrics["yaw_overshoot_pct"] == (16.6, "%"), "an identical repeat is one metric" + + results = {"baseline": {"TransientEval": {**WHOLE, **metrics}}, "bar": {"TransientEval": {**WHOLE, **metrics}}} + with pytest.raises(KeyError, match=r"name one of: \['frequency.yaw_gain_dc', 'step.yaw_gain_dc'\]"): + trade.compare(results, [trade.MetricSpec("TransientEval", "yaw_gain_dc")]) + + +def test_a_metric_repeated_with_conflicting_values_is_an_error(tmp_path: Path) -> None: + csv_path = tmp_path / "metrics.csv" + csv_path.write_text("metric,value\nroll_gain,0.04\nroll_gain,0.05\n") + with pytest.raises(ValueError, match="twice with different values"): + standards.read_metrics(csv_path) + + +def test_every_registered_standard_exists_and_shares_the_vehicle_sim_executable() -> None: + import yaml + + for standard in standards.STANDARDS.values(): + assert standard.sim_path.is_file(), standard.sim_path + config = yaml.safe_load(standard.config_path.read_text()) + assert config["simulation"]["exec_name"] == standards.VEHICLE_SIM_MODEL, ( + f"{standard.name} runs a different model, so one compile cannot serve it" + ) + with pytest.raises(KeyError, match="FourPostSim"): + standards.get_standard("FourPostEval") + + +def test_a_standards_config_is_written_beside_the_variant_not_over_the_shared_one(tmp_path: Path) -> None: + standard = standards.get_standard("TransientEval") + before = standard.config_path.read_bytes() + config_path, metrics_csv = standards.write_config( + standard, variant_dir=tmp_path, build_dir=tmp_path / "build", render_report=False, max_workers=6 + ) + import yaml + + written = yaml.safe_load(config_path.read_text()) + assert standard.config_path.read_bytes() == before + assert written["simulation"]["build_dir"] == str(tmp_path / "build") + assert written["report"]["enabled"] is False and written["execution"]["max_workers"] == 6 + assert metrics_csv == tmp_path / "results/TransientEval/transient_eval_report_metrics.csv" + + +def test_study_candidates_are_validated_before_anything_is_compiled() -> None: + variables = {"rear.bar": {"range": [300.0, 900.0]}} + good = trade_study.load_candidates({"candidates": {"stiff": {"rear.bar": 800}}}, variables) + assert good == {"stiff": {"rear.bar": 800.0}} + with pytest.raises(KeyError, match="not variables in vehicle_architecture.yaml"): + trade_study.load_candidates({"candidates": {"x": {"wheelbase": 1.6}}}, variables) + with pytest.raises(ValueError, match="always included"): + trade_study.load_candidates({"candidates": {"baseline": {"rear.bar": 800}}}, variables) + with pytest.raises(ValueError, match="changes nothing"): + trade_study.load_candidates({"candidates": {"same": {}}}, variables) + + +def test_the_example_study_names_real_standards_and_is_loadable() -> None: + import yaml + + study = yaml.safe_load(trade_study.DEFAULT_STUDY.read_text()) + specs = trade_study.load_specs(study) + assert {s.standard for s in specs} <= set(standards.STANDARDS) + assert all(s.resolution and s.resolution > 0 for s in specs) + + +def test_cpus_go_to_concurrent_runs_until_each_would_be_starved() -> None: + assert split_cpus(9, 12) == (3, 4) + assert split_cpus(1, 12) == (1, 12), "a lone run gets every CPU" + assert split_cpus(2, 12) == (2, 6) + assert split_cpus(5, 2) == (1, 2), "never zero concurrent runs" + + +def test_float_noise_does_not_split_one_vehicle_into_two_cache_entries() -> None: + assert variant_key({"b": 0.1 + 0.2, "a": 1.0}) == variant_key({"a": 1.0, "b": 0.3}) + assert variant_key({}) != variant_key({"a": 1.0}) + + +def test_the_solver_refuses_an_evaluation_that_lost_cases() -> None: + """One definition of "not whole", shared by the trade table and the solver. + + A star point that lost a case still returns finite gradients, fitted through + fewer points. Accepted silently it bends the surrogate, and because + evaluations are cached it would bend every later solve as well. + """ + from StandardSens.pipeline.evaluator import require_whole + + whole = {"n_cases": 8.0, "n_successful_cases": 8.0, "understeer": 0.3} + assert standards.case_loss(whole) is None + require_whole({"rear.bar": 700.0}, whole) + + lossy = {**whole, "n_successful_cases": 7.0} + assert standards.case_loss(lossy) == "1 of 8 cases failed" + with pytest.raises(RuntimeError, match="1 of 8 cases failed.*target_ays"): + require_whole({"rear.bar": 700.0}, lossy) + + assert standards.case_loss({"understeer": 0.3}) == "reported no case counts"