From 423b2e01a20e908670b1eb4dcac4fddcf9127351 Mon Sep 17 00:00:00 2001 From: Arjun Rao Date: Sun, 20 Sep 2026 21:52:20 -0500 Subject: [PATCH 1/2] Compare named vehicles across the standard sims with opt-trade OptSim could sample a design space and invert for a setup. It could not answer the question that a design review asks: what does this specific change give, and what does it cost, across more than one study. opt-trade reads a YAML of named candidates and the metrics to compare. It compiles each candidate once and runs each requested standard against that one executable. Then it writes a comparison table. SteadyStateEval, RampSteerEval and TransientEval all run the same compiled model, and _3_StandardSim already builds it once for all three. They also share one interface. So pipeline/standards.py is a registry of three entries and a generic runner. The sweep's batch step now uses the registry. Before this change, a second standard in compiler_config.yaml would have run SteadyStateEval's report against the wrong executable. opt-trade compiles each candidate and does not use an override. opt-solve makes the opposite choice. A trade study must be able to change mass, CG, toe and camber, and an override of those silently does nothing. The solver's evaluator had a private cache of compiled vehicles. That cache is now pipeline/variants.py, and both tools share it. The cache key is the content, so two studies build and simulate a shared vehicle only once. The report has no score and no ranking. It makes sure that each difference is worth reading: - It shows a delta below the metric's stated resolution, but marks it. - It does not compare a run that lost simulation cases. - If one candidate is exactly two others combined, it reports the interaction. So "we'll do both" is not assumed to be the sum. TransientEval reports some metrics once per group under one name. You must ask for those by group. A metric that repeats with different values is an error. Real run on the example study, four vehicles by two standards: 1144 s cold, 3 s cached. --- .../StandardSens/configs/trade_study.yaml | 49 ++++ _4_OptSim/StandardSens/pipeline/batch.py | 9 +- _4_OptSim/StandardSens/pipeline/compiler.py | 8 + _4_OptSim/StandardSens/pipeline/evaluator.py | 189 +++------------ _4_OptSim/StandardSens/pipeline/standards.py | 209 +++++++++++++++++ .../pipeline/steady_state_eval_report.py | 175 +++----------- _4_OptSim/StandardSens/pipeline/trade.py | 218 ++++++++++++++++++ _4_OptSim/StandardSens/pipeline/variants.py | 150 ++++++++++++ _4_OptSim/StandardSens/trade_study.py | 198 ++++++++++++++++ docs/doe-reverse-engineering.md | 94 ++++++++ makefile | 11 +- tests/test_optsim_overrides.py | 10 +- tests/test_optsim_trade.py | 186 +++++++++++++++ 13 files changed, 1198 insertions(+), 308 deletions(-) create mode 100644 _4_OptSim/StandardSens/configs/trade_study.yaml create mode 100644 _4_OptSim/StandardSens/pipeline/standards.py create mode 100644 _4_OptSim/StandardSens/pipeline/trade.py create mode 100644 _4_OptSim/StandardSens/pipeline/variants.py create mode 100644 _4_OptSim/StandardSens/trade_study.py create mode 100644 tests/test_optsim_trade.py diff --git a/_4_OptSim/StandardSens/configs/trade_study.yaml b/_4_OptSim/StandardSens/configs/trade_study.yaml new file mode 100644 index 0000000..9d6218a --- /dev/null +++ b/_4_OptSim/StandardSens/configs/trade_study.yaml @@ -0,0 +1,49 @@ +# A trade study: named vehicles compared across the standard sims. +# +# make opt-trade +# make opt-trade STUDY=path/to/another_study.yaml +# +# See docs/doe-reverse-engineering.md. Copy this file for each study; results land +# in _4_OptSim/results/trade/.md and .csv. +name: rear_roll_stiffness + +# Each candidate is the baseline vehicle with these variables changed. Any +# variable declared in vehicle_architecture.yaml may be used, mass and CG +# included, because every candidate is compiled rather than overridden. The +# baseline is always included and needs no entry. +# +# `both` makes exactly the changes of the other two, so the report also says +# whether those two changes stack or interact. +candidates: + stiff_rear_bar: + rear.stabar.rate_n_m_per_rad: 961.495352 + soft_front_spring: + front.actuation.spring_rate_n_per_m: 21015.2202 + both: + rear.stabar.rate_n_m_per_rad: 961.495352 + front.actuation.spring_rate_n_per_m: 21015.2202 + +# What to compare, per standard: SteadyStateEval, RampSteerEval or TransientEval. +# All three run against the one executable compiled for each candidate. +# +# `resolution` is the smallest change worth acting on. A smaller delta is still +# reported, marked `~`. Leave it off and the delta is reported unjudged. There is +# no score and no ranking: what a tenth of a degree of understeer is worth in +# settling time is your call, not the tool's. +metrics: + SteadyStateEval: + understeer_gradient_deg_per_g: {resolution: 0.02} + roll_gradient_deg_per_g: {resolution: 0.02} + peak_handwheel_torque_Nm: {resolution: 0.5} + TransientEval: + yaw_rise_time_s: {resolution: 0.005} + yaw_overshoot_pct: {resolution: 1.0} + roll_overshoot_pct: {resolution: 1.0} + settling_time_s: {resolution: 0.05} + +# Also render each standard's PDF report per vehicle. Worth having when the study +# goes in front of a design review; costs roughly ten seconds per run. +render_reports: true + +# CPUs available to simulations (the Docker container's count, not the host's). +cpus: 12 diff --git a/_4_OptSim/StandardSens/pipeline/batch.py b/_4_OptSim/StandardSens/pipeline/batch.py index be45e00..162d462 100644 --- a/_4_OptSim/StandardSens/pipeline/batch.py +++ b/_4_OptSim/StandardSens/pipeline/batch.py @@ -1,4 +1,4 @@ -"""batch.py — Run the SteadyStateEval report wrapper for all variants. +"""batch.py — Run each configured standard for all variants. For each variant_XXXX/build// that has a compiled executable: 1. Skip if metrics.csv already exists and is valid (correct row count) @@ -20,7 +20,7 @@ import pandas as pd import yaml -from StandardSens.pipeline.steady_state_eval_report import run_report +from StandardSens.pipeline.standards import get_standard, run_standard # --------------------------------------------------------------------------- # Config @@ -83,7 +83,10 @@ def run_variant( results_dir.mkdir(parents=True, exist_ok=True) try: - metrics_csv = run_report( + # Dispatch on the standard's name: running one standard's report against + # another's executable fails late and unhelpfully. + metrics_csv = run_standard( + get_standard(standard), variant_dir=variant_dir, build_dir=build_dir, exec_name=standard_cfg["model"], diff --git a/_4_OptSim/StandardSens/pipeline/compiler.py b/_4_OptSim/StandardSens/pipeline/compiler.py index 8b58ebd..6d3fa06 100644 --- a/_4_OptSim/StandardSens/pipeline/compiler.py +++ b/_4_OptSim/StandardSens/pipeline/compiler.py @@ -17,6 +17,7 @@ from __future__ import annotations +from collections.abc import Collection import platform import subprocess import sys @@ -55,6 +56,7 @@ # list would keep caches its neighbour had already declared stale. PIPELINE_TOOLING_INPUTS = ( DEFAULT_REPORT_WRAPPER, + STANDARD_DIR / "pipeline/standards.py", DEFAULT_STEADY_STATE_SIM, DEFAULT_STEADY_STATE_CONFIG, DEFAULT_MODELICA_RUNNER, @@ -269,9 +271,13 @@ def compile_all( template_path: Path = DEFAULT_MOS_TEMPLATE, doe_config_path: Path = DEFAULT_DOE_CONFIG, architecture_config_path: Path = DEFAULT_ARCHITECTURE_CONFIG, + only_standards: Collection[str] | None = None, ) -> dict[str, list[Path]]: """Compile all variants in population_dir for all standards in config. + `only_standards` narrows that to the named ones, for callers that run several + standards against one executable and must not pay for a build per standard. + Skips variants that are already compiled and whose inputs haven't changed. Runs variants in parallel using ProcessPoolExecutor. Returns dict mapping standard -> list of successful exe paths. @@ -279,6 +285,8 @@ def compile_all( """ cfg = load_compiler_config(compiler_config_path) standards: dict[str, dict] = cfg["standards"] + if only_standards is not None: + standards = {name: standards[name] for name in only_standards} max_workers: int = cfg.get("max_workers", 2) # Resolve boblib_path relative to the config file diff --git a/_4_OptSim/StandardSens/pipeline/evaluator.py b/_4_OptSim/StandardSens/pipeline/evaluator.py index 9ba5437..218bc1d 100644 --- a/_4_OptSim/StandardSens/pipeline/evaluator.py +++ b/_4_OptSim/StandardSens/pipeline/evaluator.py @@ -7,10 +7,10 @@ value. That default is the slow one on purpose: a compile is always correct, and a wrong override is silent. See `overrides.py` for why the list is what it is. -Executables and evaluations are cached under the solve directory, and discarded -when BobLib, the vehicle or the SteadyStateEval tooling changes. The star design -does not depend on the targets, so a second solve against new targets reuses -every star evaluation and pays only for verification. +Executables live in a `VariantStore` and evaluations beside it; both are +discarded when BobLib, the vehicle or the SteadyStateEval tooling changes. The +star design does not depend on the targets, so a second solve against new targets +reuses every star evaluation and pays only for verification. """ from __future__ import annotations @@ -25,39 +25,18 @@ import time from typing import Any -import yaml - from StandardSens.pipeline import overrides as ov -from StandardSens.pipeline._pipeline_hash import check_pipeline_hash -from StandardSens.pipeline.compiler import ( - PIPELINE_TOOLING_INPUTS, - compile_all, - find_exe, - load_compiler_config, -) -from StandardSens.pipeline.generate_configs import SWEEP_SCOPE_ALL, refresh_doe_config -from StandardSens.pipeline.generator import build_context, generate_variants, read_metrics_csv -from StandardSens.pipeline.orchestration import ( - ARCHITECTURE_CONFIG, - COMPILER_CONFIG, - DOE_CONFIG, - STANDARD_BUILD_DIR, -) +from StandardSens.pipeline.generator import build_context, read_metrics_csv +from StandardSens.pipeline.orchestration import STANDARD_BUILD_DIR from StandardSens.pipeline.sampler import read_baseline from StandardSens.pipeline.steady_state_eval_report import Isoline, run_report +from StandardSens.pipeline.variants import BUILD_STANDARD, VariantStore, split_cpus, variant_key -STANDARD = "SteadyStateEval" SOLVE_DIR = STANDARD_BUILD_DIR / "solve" EVALUATION_TIMEOUT_S = 1800 -# Simulation throughput on a 12-CPU container was 1.9x higher with every CPU busy -# than with four, so concurrent evaluations are preferred until each would drop -# below this many case workers. -MIN_CASE_WORKERS = 4 - Variant = dict[str, float] Metrics = dict[str, float] -CompileKey = tuple[tuple[str, float], ...] class SteadyStateEvaluator: @@ -70,62 +49,28 @@ def __init__(self, *, isoline: Isoline, cpus: int) -> None: float(isoline.velocity_mps), tuple(sorted(map(float, isoline.target_ays), reverse=True)) ) self.cpus = max(1, int(cpus)) - self.exes_dir = SOLVE_DIR / "exes" + + self.store = VariantStore(SOLVE_DIR) self.evals_dir = SOLVE_DIR / "evals" + if self.store.was_reset: + shutil.rmtree(self.evals_dir, ignore_errors=True) - cfg = self._write_own_doe_config() + cfg = self.store.doe_config self.variables: dict[str, dict[str, Any]] = {v["path"]: v for v in cfg["variables"]} - self.context = build_context(cfg, self.doe_config_path) + self.context = build_context(cfg, self.store.doe_config_path) self.baseline: Variant = read_baseline(Path(cfg["baseline_mo"]), cfg["variables"]) - - compiler_cfg = load_compiler_config(COMPILER_CONFIG) - self.standard_cfg: dict[str, Any] = compiler_cfg["standards"][STANDARD] - boblib_path = (COMPILER_CONFIG.resolve().parent / compiler_cfg["boblib_path"]).resolve() - - self._drop_caches_if_inputs_changed(boblib_path) - self._index: dict[str, int] = ( - json.loads(self._index_path.read_text()) if self._index_path.exists() else {} - ) self._init_cache: dict[Path, dict[str, ov.InitParameter]] = {} - def _write_own_doe_config(self) -> dict[str, Any]: - """Generate the solver's DOE config beside its caches, not over the sweep's. - - The sweep's `_doe_config.yaml` is committed and records the scope its - population was built at, which `opt-search` relies on. The solver needs - every variable's spec whatever that scope was, so it keeps its own copy. - The generated file names the baseline record and vehicle template - relative to the sweep's config directory; here they are made absolute so - they resolve from anywhere. - """ - self.doe_config_path = (SOLVE_DIR / "_doe_config.yaml").resolve() - self.doe_config_path.parent.mkdir(parents=True, exist_ok=True) - cfg = refresh_doe_config( - architecture_config_path=ARCHITECTURE_CONFIG, - compiler_config_path=COMPILER_CONFIG, - doe_config_path=self.doe_config_path, - scope=SWEEP_SCOPE_ALL, - ) - config_dir = DOE_CONFIG.resolve().parent - cfg["baseline_mo"] = (config_dir / cfg["baseline_mo"]).resolve().as_posix() - cfg["architecture"]["template"] = ( - (config_dir.parent / cfg["architecture"]["template"]).resolve().as_posix() - ) - self.doe_config_path.write_text(yaml.safe_dump(cfg, sort_keys=False)) - return cfg - - # ---------------------------------------------------------------- evaluation - def __call__(self, variants: list[Variant]) -> list[Metrics]: pending = [v for v in variants if not self._metrics_path(v).exists()] if pending: - keys = {compile_key(v, self.baseline) for v in pending} - self._ensure_executables(keys) - for key in keys: # parse each init XML once, before the threads race to - self._init_parameters(self._build_dir(key)) + compiled = [compiled_part(v, self.baseline) for v in pending] + self.store.ensure_compiled(compiled) + for vehicle in compiled: # parse each init XML once, before the threads race to + self._init_parameters(self.store.build_dir(vehicle)) - concurrent = min(len(pending), max(1, self.cpus // MIN_CASE_WORKERS)) - workers = min(len(self.isoline.target_ays), max(1, self.cpus // concurrent)) + concurrent, workers = split_cpus(len(pending), self.cpus) + workers = min(workers, len(self.isoline.target_ays)) print( f"Simulating {len(pending)} setup(s), {concurrent} at a time with {workers} " f"case workers each ({len(variants) - len(pending)} cached)...", @@ -140,7 +85,7 @@ def __call__(self, variants: list[Variant]) -> list[Metrics]: return [read_metrics_csv(self._metrics_path(v)) for v in variants] def _simulate(self, variant: Variant, *, workers: int) -> None: - build_dir = self._build_dir(compile_key(variant, self.baseline)) + build_dir = self.store.build_dir(compiled_part(variant, self.baseline)) runtime = {p: v for p, v in variant.items() if p in ov.RUNTIME_SAFE_PATHS} eval_dir = self._eval_dir(variant) eval_dir.mkdir(parents=True, exist_ok=True) @@ -148,7 +93,7 @@ def _simulate(self, variant: Variant, *, workers: int) -> None: run_report( variant_dir=eval_dir, build_dir=build_dir, - exec_name=self.standard_cfg["model"], + exec_name=self.store.standard_cfg["model"], timeout=EVALUATION_TIMEOUT_S, init_parameters=ov.variant_overrides( runtime, self.variables, self.context, self._init_parameters(build_dir) @@ -158,100 +103,30 @@ def _simulate(self, variant: Variant, *, workers: int) -> None: render_report=False, # the solver reads the metrics CSV, never the PDF ) - # --------------------------------------------------------------- executables - - def _ensure_executables(self, keys: set[CompileKey]) -> None: - new = [k for k in sorted(keys) if _key_json(k) not in self._index] - if not new and all(find_exe(self._build_dir(k), self.standard_cfg) for k in keys): - return - - for key in new: - self._index[_key_json(key)] = len(self._index) - self.exes_dir.mkdir(parents=True, exist_ok=True) - ordered = sorted(self._index, key=self._index.__getitem__) - generate_variants( - self.doe_config_path, [json.loads(entry) for entry in ordered], self.exes_dir - ) - self._index_path.write_text(json.dumps(self._index, indent=1)) - compile_all(self.exes_dir, COMPILER_CONFIG, doe_config_path=self.doe_config_path) - - for key in keys: - build_dir = self._build_dir(key) - if find_exe(build_dir, self.standard_cfg) is None: - log = build_dir.parents[1] / f"compile_error_{STANDARD}.log" - detail = log.read_text()[-2000:] if log.exists() else "no compile log written" - raise RuntimeError(f"Compile failed for {dict(key) or 'baseline'}:\n{detail}") - - def _build_dir(self, key: CompileKey) -> Path: - return self.exes_dir / f"variant_{self._index[_key_json(key)]:04d}" / "build" / STANDARD - def _init_parameters(self, build_dir: Path) -> dict[str, ov.InitParameter]: if build_dir not in self._init_cache: - init_xml = ov.find_init_xml(build_dir, self.standard_cfg["model"]) + init_xml = ov.find_init_xml(build_dir, self.store.standard_cfg["model"]) self._init_cache[build_dir] = ov.load_init_parameters(init_xml) return self._init_cache[build_dir] - @property - def _index_path(self) -> Path: - return self.exes_dir / "index.json" - - # --------------------------------------------------------------------- cache - def _eval_dir(self, variant: Variant) -> Path: - payload = json.dumps( - {"variant": {p: _round(v) for p, v in sorted(variant.items())}, "isoline": self.isoline}, - sort_keys=True, - ) + payload = json.dumps({"variant": variant_key(variant), "isoline": self.isoline}) return self.evals_dir / hashlib.sha1(payload.encode()).hexdigest()[:16] def _metrics_path(self, variant: Variant) -> Path: - return self._eval_dir(variant) / "results" / STANDARD / "metrics.csv" - - def _drop_caches_if_inputs_changed(self, boblib_path: Path) -> None: - """A cache must never outlive the inputs it was built from.""" - if not self.exes_dir.exists(): - return - try: - # The same call compile_all makes, so the two cannot disagree about - # what counts as stale. - check_pipeline_hash( - self.exes_dir, - self.doe_config_path, - COMPILER_CONFIG, - boblib_path, - ARCHITECTURE_CONFIG, - PIPELINE_TOOLING_INPUTS, - ) - except RuntimeError: - print( - "Solver cache predates a change to BobLib, the vehicle, or the " - "SteadyStateEval tooling; discarding it." - ) - shutil.rmtree(self.exes_dir) - shutil.rmtree(self.evals_dir, ignore_errors=True) + return self._eval_dir(variant) / "results" / BUILD_STANDARD / "metrics.csv" -def compile_key(variant: Variant, baseline: Variant) -> CompileKey: - """The compile-only values that decide which executable a variant needs. +def compiled_part(variant: Variant, baseline: Variant) -> Variant: + """The changes that decide which executable a variant needs. Runtime-safe knobs never appear, so every setting of them shares one executable. A compile-only knob left at its baseline does not appear either, so it shares the baseline executable instead of forcing an identical rebuild. """ - return tuple( - sorted( - (path, _round(value)) - for path, value in variant.items() - if path not in ov.RUNTIME_SAFE_PATHS - and not math.isclose(value, baseline[path], rel_tol=1e-12, abs_tol=1e-12) - ) - ) - - -def _round(value: float) -> float: - """Collapse float noise so equal settings share a cache entry.""" - return float(f"{float(value):.12g}") - - -def _key_json(key: CompileKey) -> str: - return json.dumps(dict(key), sort_keys=True) + return { + path: value + for path, value in variant.items() + if path not in ov.RUNTIME_SAFE_PATHS + and not math.isclose(value, baseline[path], rel_tol=1e-12, abs_tol=1e-12) + } diff --git a/_4_OptSim/StandardSens/pipeline/standards.py b/_4_OptSim/StandardSens/pipeline/standards.py new file mode 100644 index 0000000..06af10a --- /dev/null +++ b/_4_OptSim/StandardSens/pipeline/standards.py @@ -0,0 +1,209 @@ +"""standards.py — Run any VehicleSim standard against a compiled variant. + +SteadyStateEval, RampSteerEval and TransientEval are three questions asked of the +same compiled model, `BobLib.Experiments.Standards.VehicleSim`. `_3_StandardSim` +already builds that model once and points all three configs at it. They also +share one interface: `python -m `, a `simulation` block +naming the executable, and a `report` block whose `output_path` decides where the +PDF and its `_metrics.csv` land. So one compile per vehicle serves every +standard here, and adding a standard is one entry in `STANDARDS`. + +FourPostEval is not listed: it runs a different model (`FourPostSim`), which +means a second compile per vehicle and its own config conventions. +""" + +from __future__ import annotations + +from collections.abc import Callable +import csv +from dataclasses import dataclass +import hashlib +from pathlib import Path +import shutil +import subprocess +import sys +from typing import Any + +import yaml + +REPO_ROOT = Path(__file__).resolve().parents[3] +VEHICLE_SIM_MODEL = "BobLib.Experiments.Standards.VehicleSim" + +ConfigEdit = Callable[[dict[str, Any]], None] + + +@dataclass(frozen=True) +class Standard: + name: str + package: str # directory under _3_StandardSim, also the python package + stem: str # "_sim.py", "_config.yml", "_report.pdf" + + @property + def module(self) -> str: + return f"_3_StandardSim.{self.package}.{self.stem}_sim" + + @property + def sim_path(self) -> Path: + return REPO_ROOT / "_3_StandardSim" / self.package / f"{self.stem}_sim.py" + + @property + def config_path(self) -> Path: + return REPO_ROOT / "_3_StandardSim" / self.package / f"{self.stem}_config.yml" + + +STANDARDS: dict[str, Standard] = { + s.name: s + for s in ( + Standard("SteadyStateEval", "SteadyStateEval", "steady_state_eval"), + Standard("RampSteerEval", "RampSteerEval", "ramp_steer_eval"), + Standard("TransientEval", "TransientEval", "transient_eval"), + ) +} + + +def input_digest(standard: Standard) -> str: + """Fingerprint of what decides a standard's numbers besides the vehicle. + + The pipeline hash covers SteadyStateEval's tooling only, so results cached + for another standard are stamped with this and rerun when it changes. + """ + digest = hashlib.sha256() + for path in (standard.sim_path, standard.config_path): + digest.update(path.read_bytes()) + return digest.hexdigest() + + +def get_standard(name: str) -> Standard: + try: + return STANDARDS[name] + except KeyError: + raise KeyError( + f"{name!r} is not a VehicleSim standard OptSim can run. Known: {sorted(STANDARDS)}. " + "A standard on a different model (FourPostEval runs FourPostSim) needs its own " + "compile and is not wired in." + ) from None + + +def write_config( + standard: Standard, + *, + variant_dir: Path, + build_dir: Path, + exec_name: str = VEHICLE_SIM_MODEL, + render_report: bool = True, + max_workers: int | None = None, + edit: ConfigEdit | None = None, +) -> tuple[Path, Path]: + """Write the standard's config for one variant; return (config, metrics CSV). + + The shared standard's own config file is read, never written. `edit` is the + hook for the few callers that need more than the executable swapped. + """ + with standard.config_path.open(encoding="utf-8") as handle: + config = yaml.safe_load(handle) + if not isinstance(config, dict): + raise TypeError(f"Expected a mapping at top level: {standard.config_path}") + + simulation = config.setdefault("simulation", {}) + simulation["build_dir"] = str(build_dir) + simulation["exec_name"] = exec_name + + # Otherwise execution settings stay exactly as the standard's config has them. + if max_workers is not None: + execution = config.setdefault("execution", {}) + execution["parallel"] = True + execution["max_workers"] = int(max_workers) + + # The metrics CSV path is derived from output_path whether or not the PDF is + # rendered, so anchoring it to the variant keeps both beside its results. + results_dir = variant_dir / "results" / standard.name + report = config.setdefault("report", {}) + report["enabled"] = render_report + report["output_path"] = str(results_dir / f"{standard.stem}_report.pdf") + + if edit is not None: + edit(config) + + config_path = results_dir / f"{standard.stem}_config.generated.yml" + config_path.parent.mkdir(parents=True, exist_ok=True) + with config_path.open("w", encoding="utf-8") as handle: + yaml.safe_dump(config, handle, sort_keys=False) + return config_path, results_dir / f"{standard.stem}_report_metrics.csv" + + +def run_standard( + standard: Standard, + *, + variant_dir: Path, + build_dir: Path, + exec_name: str = VEHICLE_SIM_MODEL, + timeout: int | None = None, + render_report: bool = True, + max_workers: int | None = None, + edit: ConfigEdit | None = None, +) -> Path: + """Run one standard for one variant and return its metrics CSV.""" + config_path, metrics_csv = write_config( + standard, + variant_dir=variant_dir, + build_dir=build_dir, + exec_name=exec_name, + render_report=render_report, + max_workers=max_workers, + edit=edit, + ) + completed = subprocess.run( + [sys.executable, "-m", standard.module, str(config_path)], + cwd=str(REPO_ROOT), + capture_output=True, + text=True, + timeout=timeout, + ) + if completed.returncode != 0: + raise RuntimeError( + f"{standard.name} failed.\nConfig: {config_path}\n" + f"Stdout:\n{completed.stdout}\nStderr:\n{completed.stderr}" + ) + if not metrics_csv.exists(): + raise FileNotFoundError(f"{standard.name} produced no metrics CSV: {metrics_csv}") + + # A stable name beside the report-style one, which is what the sweep's + # batch runner and aggregator look for. + canonical = metrics_csv.with_name("metrics.csv") + shutil.copyfile(metrics_csv, canonical) + return canonical + + +def read_metrics(path: Path) -> dict[str, tuple[float, str]]: + """Read a standard's metrics CSV as {name: (value, units)}. + + TransientEval reports some metrics once per group under the same name + (`ay_gain_dc` for the step and again for the frequency sweep). Keyed by bare + name the later row would silently replace the earlier, so a name that spans + groups is exposed only in its qualified form, `group.metric`. A row repeated + with the same value is one metric; repeated with a different value it is an + ambiguity nothing here can resolve, and an error. + """ + with path.open(newline="", encoding="utf-8") as handle: + rows = [row for row in csv.DictReader(handle) if row.get("metric")] + groups: dict[str, set[str]] = {} + for row in rows: + groups.setdefault(row["metric"], set()).add(row.get("group") or "") + + metrics: dict[str, tuple[float, str]] = {} + for row in rows: + name = row["metric"] + if len(groups[name]) > 1: + name = f"{row.get('group') or '?'}.{name}" + try: + value = float(row.get("value") or "nan") + except ValueError: + value = float("nan") + if name in metrics and not _same(metrics[name][0], value): + raise ValueError(f"{path} reports {name!r} twice with different values") + metrics[name] = (value, row.get("units") or "") + return metrics + + +def _same(a: float, b: float) -> bool: + return a == b or (a != a and b != b) # NaN equals NaN here diff --git a/_4_OptSim/StandardSens/pipeline/steady_state_eval_report.py b/_4_OptSim/StandardSens/pipeline/steady_state_eval_report.py index a55c608..bef3089 100644 --- a/_4_OptSim/StandardSens/pipeline/steady_state_eval_report.py +++ b/_4_OptSim/StandardSens/pipeline/steady_state_eval_report.py @@ -1,22 +1,16 @@ -"""steady_state_eval_report.py — Run the SteadyStateEval postprocessor for one DOE variant. +"""steady_state_eval_report.py — SteadyStateEval's own options on the standards runner. -This reuses the existing SteadyStateEval Python wrapper to generate the report-style -metrics CSV from the variant-specific compiled executable. +`standards.run_standard` runs any VehicleSim standard for a variant. This adds the +two things only SteadyStateEval callers ask for: Modelica parameter overrides +applied to every case, and a single isoline in place of the standard's four. """ from __future__ import annotations -import shutil -import subprocess -import sys from pathlib import Path from typing import Any, NamedTuple -import yaml - -STANDARD_DIR = Path(__file__).resolve().parents[1] -REPO_ROOT = STANDARD_DIR.parent.parent -BASE_CONFIG = REPO_ROOT / "_3_StandardSim/SteadyStateEval/steady_state_eval_config.yml" +from StandardSens.pipeline.standards import VEHICLE_SIM_MODEL, get_standard, run_standard class Isoline(NamedTuple): @@ -26,149 +20,46 @@ class Isoline(NamedTuple): target_ays: tuple[float, ...] -def _load_yaml(path: Path) -> dict[str, Any]: - with path.open("r", encoding="utf-8") as f: - data = yaml.safe_load(f) - if not isinstance(data, dict): - raise TypeError(f"Expected mapping at top level: {path}") - return data - - -def _dump_yaml(path: Path, data: dict[str, Any]) -> None: - path.parent.mkdir(parents=True, exist_ok=True) - with path.open("w", encoding="utf-8") as f: - yaml.safe_dump(data, f, sort_keys=False) - - -def build_report_config( - *, - variant_dir: Path, - build_dir: Path, - exec_name: str, - base_config_path: Path = BASE_CONFIG, - init_parameters: dict[str, float] | None = None, - isoline: Isoline | None = None, - max_workers: int | None = None, - render_report: bool = True, -) -> tuple[Path, Path]: - """Write a temporary SteadyStateEval config for one DOE variant. - - `init_parameters` are Modelica parameter overrides applied to every case, so - one compiled executable can stand in for a vehicle it was not compiled as. - `isoline` swaps the standard's four-isoline matrix for a single one, without - touching the shared standard's own config or its regression baselines. - `render_report=False` skips the PDF; the metrics CSV is written either way. - - Returns: - (config_path, canonical_metrics_csv_path) - """ - config = _load_yaml(base_config_path) - - simulation = config.setdefault("simulation", {}) - if not isinstance(simulation, dict): - raise TypeError("SteadyStateEval config simulation block must be a mapping") - - report = config.setdefault("report", {}) - if not isinstance(report, dict): - raise TypeError("SteadyStateEval config report block must be a mapping") - - execution = config.setdefault("execution", {}) - if not isinstance(execution, dict): - raise TypeError("SteadyStateEval config execution block must be a mapping") - - simulation["build_dir"] = str(build_dir) - simulation["exec_name"] = exec_name - - if init_parameters: - merged = dict(simulation.get("init_parameters") or {}) - merged.update({name: float(value) for name, value in init_parameters.items()}) - simulation["init_parameters"] = merged - - if isoline is not None: - # These three move together: the cap and the exported-metric velocity - # must name the one isoline being run, or the report selects nothing. - sweep = config.setdefault("sweep", {}) - sweep["testVels"] = [isoline.velocity_mps] - sweep["targetAys"] = list(isoline.target_ays) - sweep["maxAyByVelocity"] = {isoline.velocity_mps: max(isoline.target_ays)} - report["metric_target_velocity_mps"] = isoline.velocity_mps - - # Otherwise execution settings stay exactly as the standard-sim config has - # them; that config controls whether cases run serially or in parallel. - if max_workers is not None: - execution["parallel"] = True - execution["max_workers"] = int(max_workers) - - # Keep the report output location anchored to the variant so the generated - # PDF and metrics CSV live beside the DOE result artifacts. The CSV path is - # derived from output_path whether or not the PDF is rendered. - report["enabled"] = render_report - report["output_path"] = str( - variant_dir / "results" / "SteadyStateEval" / "steady_state_eval_report.pdf" - ) - - results_dir = variant_dir / "results" / "SteadyStateEval" - config_path = results_dir / "steady_state_eval_config.generated.yml" - metrics_csv = results_dir / "steady_state_eval_report_metrics.csv" - - _dump_yaml(config_path, config) - return config_path, metrics_csv - - def run_report( *, variant_dir: Path, build_dir: Path, - exec_name: str, + exec_name: str = VEHICLE_SIM_MODEL, timeout: int | None = None, - base_config_path: Path = BASE_CONFIG, init_parameters: dict[str, float] | None = None, isoline: Isoline | None = None, max_workers: int | None = None, render_report: bool = True, ) -> Path: - """Run the SteadyStateEval report wrapper and return the metrics CSV path.""" - config_path, metrics_csv = build_report_config( + """Run SteadyStateEval for one variant and return its metrics CSV. + + `init_parameters` lets one compiled executable stand in for a vehicle it was + not compiled as. `isoline` swaps the standard's four-isoline matrix for one, + without touching the shared standard's own config or regression baselines. + """ + + def edit(config: dict[str, Any]) -> None: + if init_parameters: + simulation = config["simulation"] + merged = dict(simulation.get("init_parameters") or {}) + merged.update({name: float(value) for name, value in init_parameters.items()}) + simulation["init_parameters"] = merged + if isoline is not None: + # These move together: the cap and the exported-metric velocity must + # name the one isoline being run, or the report selects nothing. + sweep = config.setdefault("sweep", {}) + sweep["testVels"] = [isoline.velocity_mps] + sweep["targetAys"] = list(isoline.target_ays) + sweep["maxAyByVelocity"] = {isoline.velocity_mps: max(isoline.target_ays)} + config["report"]["metric_target_velocity_mps"] = isoline.velocity_mps + + return run_standard( + get_standard("SteadyStateEval"), variant_dir=variant_dir, build_dir=build_dir, exec_name=exec_name, - base_config_path=base_config_path, - init_parameters=init_parameters, - isoline=isoline, - max_workers=max_workers, - render_report=render_report, - ) - - cmd = [ - sys.executable, - "-m", - "_3_StandardSim.SteadyStateEval.steady_state_eval_sim", - str(config_path), - ] - - completed = subprocess.run( - cmd, - cwd=str(REPO_ROOT), - capture_output=True, - text=True, timeout=timeout, + render_report=render_report, + max_workers=max_workers, + edit=edit, ) - - if completed.returncode != 0: - raise RuntimeError( - "SteadyStateEval report generation failed.\n" - f"Config: {config_path}\n" - f"Stdout:\n{completed.stdout}\n" - f"Stderr:\n{completed.stderr}" - ) - - if not metrics_csv.exists(): - raise FileNotFoundError( - f"Metrics CSV not produced by SteadyStateEval report: {metrics_csv}" - ) - - # Keep a stable, pipeline-friendly name alongside the report-style CSV. - canonical_metrics_csv = metrics_csv.with_name("metrics.csv") - shutil.copyfile(metrics_csv, canonical_metrics_csv) - - return canonical_metrics_csv diff --git a/_4_OptSim/StandardSens/pipeline/trade.py b/_4_OptSim/StandardSens/pipeline/trade.py new file mode 100644 index 0000000..af41032 --- /dev/null +++ b/_4_OptSim/StandardSens/pipeline/trade.py @@ -0,0 +1,218 @@ +"""trade.py — Compare named vehicles across standard-sim metrics. + +A trade study asks "what does this change buy, and what does it cost", across +several metrics at once. This module turns simulated metrics into that +comparison and nothing more. There is deliberately no score and no ranking: how +much understeer is worth how much roll is the engineer's call, and a weighted sum +would hide that call inside a number. + +What it does insist on is that a difference be worth reading: + +- Every metric may carry a `resolution`, the smallest change worth acting on. A + smaller delta is still shown, marked as not resolved. +- A vehicle whose simulation lost cases is not comparable. Its fits run through + fewer points than the baseline's, so a delta would mix the design change with + the missing data. +- Where one candidate is exactly two others combined, the interaction is + reported: whether the two changes stack, or the combination does something + neither predicts. + +Nothing here touches the filesystem or a simulator. +""" + +from __future__ import annotations + +from dataclasses import dataclass +import difflib +from itertools import combinations +import math + +BASELINE = "baseline" + +Variant = dict[str, float] +# candidate -> standard -> metric -> (value, units) +Results = dict[str, dict[str, dict[str, tuple[float, str]]]] + +RESOLVED = "resolved" +BELOW_RESOLUTION = "below resolution" +NOT_COMPARABLE = "not comparable" + + +@dataclass(frozen=True) +class MetricSpec: + standard: str + name: str + resolution: float | None = None + + +@dataclass(frozen=True) +class Comparison: + spec: MetricSpec + units: str + candidate: str + baseline: float + value: float + verdict: str # RESOLVED, BELOW_RESOLUTION, NOT_COMPARABLE, or "" with no resolution + + @property + def delta(self) -> float: + return self.value - self.baseline + + +@dataclass(frozen=True) +class Interaction: + """How far a combined change is from the sum of its two parts.""" + + spec: MetricSpec + combined: str + parts: tuple[str, str] + value: float # delta(combined) - delta(part A) - delta(part B) + + @property + def stacks(self) -> bool | None: + if self.spec.resolution is None: + return None + return abs(self.value) < self.spec.resolution + + +def lost_cases(results: Results) -> dict[tuple[str, str], str]: + """Return {(candidate, standard): why} for every run that is not whole.""" + problems: dict[tuple[str, str], str] = {} + for candidate, by_standard in results.items(): + for standard, metrics in by_standard.items(): + total = metrics.get("n_cases", (math.nan, ""))[0] + good = metrics.get("n_successful_cases", (math.nan, ""))[0] + if not (math.isfinite(total) and math.isfinite(good)): + problems[candidate, standard] = "reported no case counts" + elif good < total: + problems[candidate, standard] = f"{int(total - good)} of {int(total)} cases failed" + return problems + + +def compare(results: Results, specs: list[MetricSpec]) -> list[Comparison]: + """Compare every candidate with the baseline on every requested metric.""" + if BASELINE not in results: + raise KeyError(f"results must include {BASELINE!r}") + problems = lost_cases(results) + + comparisons = [] + for spec in specs: + reference, units = _lookup(results, BASELINE, spec) + for candidate in results: + if candidate == BASELINE: + continue + value, _ = _lookup(results, candidate, spec) + whole = (candidate, spec.standard) not in problems and ( + BASELINE, + spec.standard, + ) not in problems + if not (whole and math.isfinite(value) and math.isfinite(reference)): + verdict = NOT_COMPARABLE + elif spec.resolution is None: + verdict = "" + else: + verdict = RESOLVED if abs(value - reference) >= spec.resolution else BELOW_RESOLUTION + comparisons.append(Comparison(spec, units, candidate, reference, value, verdict)) + return comparisons + + +def _lookup(results: Results, candidate: str, spec: MetricSpec) -> tuple[float, str]: + metrics = results[candidate].get(spec.standard) + if metrics is None: + raise KeyError(f"{candidate!r} has no {spec.standard} results") + if spec.name not in metrics: + close = difflib.get_close_matches(spec.name, metrics, n=4, cutoff=0.5) + qualified = sorted(m for m in metrics if m.endswith("." + spec.name)) + hint = ( + f" It is reported once per group, so name one of: {qualified}." + if qualified + else f" Closest: {close}." if close else "" + ) + raise KeyError(f"{spec.standard} has no metric {spec.name!r}.{hint}") + return metrics[spec.name] + + +def interactions(candidates: dict[str, Variant], comparisons: list[Comparison]) -> list[Interaction]: + """Find candidates that are two others combined and measure how they stack. + + `both = a + b` qualifies when `a` and `b` change disjoint variables and `both` + makes exactly their changes. The interaction is then the part of `both`'s + effect that neither `a` nor `b` accounts for. + """ + deltas = { + (c.candidate, c.spec): c.delta for c in comparisons if c.verdict != NOT_COMPARABLE + } + specs = list(dict.fromkeys(c.spec for c in comparisons)) + + found = [] + for combined, changes in candidates.items(): + for a, b in combinations((n for n in candidates if n != combined), 2): + disjoint = not set(candidates[a]) & set(candidates[b]) + if not (disjoint and {**candidates[a], **candidates[b]} == changes): + continue + for spec in specs: + parts = [deltas.get((name, spec)) for name in (combined, a, b)] + if None not in parts: + found.append(Interaction(spec, combined, (a, b), parts[0] - parts[1] - parts[2])) + return found + + +def to_markdown( + name: str, + candidates: dict[str, Variant], + comparisons: list[Comparison], + found: list[Interaction], + problems: dict[tuple[str, str], str], +) -> str: + lines = [f"# Trade study: {name}", "", "| Candidate | Changes from baseline |", "| --- | --- |"] + for candidate, changes in candidates.items(): + listed = ", ".join(f"`{path}` = {value:g}" for path, value in changes.items()) + lines.append(f"| {candidate} | {listed} |") + + if problems: + lines += ["", "**Not comparable** — these runs lost cases, so their fits rest on fewer points:", ""] + lines += [f"- {c} / {s}: {why}" for (c, s), why in sorted(problems.items())] + + names = list(candidates) + for standard in dict.fromkeys(c.spec.standard for c in comparisons): + rows = [c for c in comparisons if c.spec.standard == standard] + lines += ["", f"## {standard}", ""] + lines.append("| Metric | Units | Resolution | baseline | " + " | ".join(names) + " |") + lines.append("| --- | --- | --- | --- | " + " | ".join("---" for _ in names) + " |") + for spec in dict.fromkeys(c.spec for c in rows): + cells = {c.candidate: _cell(c) for c in rows if c.spec == spec} + first = next(c for c in rows if c.spec == spec) + resolution = "" if spec.resolution is None else f"{spec.resolution:g}" + lines.append( + f"| {spec.name} | {first.units} | {resolution} | {first.baseline:.4g} | " + + " | ".join(cells[n] for n in names) + + " |" + ) + + if found: + lines += ["", "## Do the changes stack?", ""] + lines.append("| Combined | Parts | Standard | Metric | Interaction | |") + lines.append("| --- | --- | --- | --- | --- | --- |") + for i in found: + says = {None: "", True: "additive within resolution", False: "**interacts**"}[i.stacks] + lines.append( + f"| {i.combined} | {' + '.join(i.parts)} | {i.spec.standard} | {i.spec.name} | " + f"{i.value:+.4g} | {says} |" + ) + + lines += [ + "", + "Each cell is the simulated value with its change from baseline. `~` marks a change " + "smaller than the metric's resolution: reported, but too small to act on. `n/c` marks " + "a run that is not comparable. The interaction is the combined change minus the sum of " + "its parts. Below the resolution it cannot be told apart from zero, which is a weaker " + "claim than saying the changes add: read the number, not just the label.", + ] + return "\n".join(lines) + "\n" + + +def _cell(comparison: Comparison) -> str: + if comparison.verdict == NOT_COMPARABLE: + return "n/c" + mark = " ~" if comparison.verdict == BELOW_RESOLUTION else "" + return f"{comparison.value:.4g} ({comparison.delta:+.3g}){mark}" diff --git a/_4_OptSim/StandardSens/pipeline/variants.py b/_4_OptSim/StandardSens/pipeline/variants.py new file mode 100644 index 0000000..03c4139 --- /dev/null +++ b/_4_OptSim/StandardSens/pipeline/variants.py @@ -0,0 +1,150 @@ +"""variants.py — A cache of compiled vehicles, addressed by what was changed. + +The sweep numbers its variants in sampling order, which is right for a population +that is generated once. The solver and the trade study instead ask for vehicles +by content — "the baseline with this toe", "the car with the stiff rear bar" — +repeatedly and in no fixed order, and a compile is the most expensive thing either +does. This store maps each distinct set of changes to one compiled variant +directory, compiles whatever is missing in a single parallel batch, and throws +everything away when BobLib, the vehicle or the simulation tooling changes. + +It reuses the sweep's generator and compiler unchanged, so a variant here is +compiled exactly as the same variant in a sweep would be. +""" + +from __future__ import annotations + +import json +from pathlib import Path +import shutil +from typing import Any + +import yaml + +from StandardSens.pipeline._pipeline_hash import check_pipeline_hash +from StandardSens.pipeline.compiler import ( + PIPELINE_TOOLING_INPUTS, + compile_all, + find_exe, + load_compiler_config, +) +from StandardSens.pipeline.generate_configs import SWEEP_SCOPE_ALL, refresh_doe_config +from StandardSens.pipeline.generator import generate_variants +from StandardSens.pipeline.orchestration import ARCHITECTURE_CONFIG, COMPILER_CONFIG, DOE_CONFIG + +# Every VehicleSim standard runs the executable the sweep builds for this one. +BUILD_STANDARD = "SteadyStateEval" + +Variant = dict[str, float] + + +def split_cpus(n_jobs: int, cpus: int, min_workers: int = 4) -> tuple[int, int]: + """Return (jobs to run at once, case workers for each). + + Simulation throughput on a 12-CPU container was 1.9x higher with every CPU + busy than with four, so concurrent jobs are preferred until each would drop + below `min_workers`; a lone job gets every CPU. + """ + concurrent = max(1, min(n_jobs, cpus // min_workers)) + return concurrent, max(1, cpus // concurrent) + + +def variant_key(variant: Variant) -> str: + """Canonical text for a set of changes; float noise must not split an entry.""" + return json.dumps({p: float(f"{float(v):.12g}") for p, v in sorted(variant.items())}) + + +class VariantStore: + def __init__(self, root: Path) -> None: + self.root = root.resolve() + self.root.mkdir(parents=True, exist_ok=True) + self.variants_dir = self.root / "variants" + self.doe_config_path = self.root / "_doe_config.yaml" + self.doe_config = self._write_own_doe_config() + + compiler_cfg = load_compiler_config(COMPILER_CONFIG) + self.standard_cfg: dict[str, Any] = compiler_cfg["standards"][BUILD_STANDARD] + boblib_path = (COMPILER_CONFIG.resolve().parent / compiler_cfg["boblib_path"]).resolve() + + self.was_reset = self._drop_if_inputs_changed(boblib_path) + index_path = self.variants_dir / "index.json" + self._index: dict[str, int] = json.loads(index_path.read_text()) if index_path.exists() else {} + + def _write_own_doe_config(self) -> dict[str, Any]: + """Generate a DOE config here rather than over the sweep's. + + The sweep's `_doe_config.yaml` is committed and records the scope its + population was built at, which `opt-search` relies on. Consumers of this + store need every variable's spec whatever that scope was, so they keep + their own copy. The generated file names the baseline record and vehicle + template relative to the sweep's config directory; here they are made + absolute so they resolve from anywhere. + """ + cfg = refresh_doe_config( + architecture_config_path=ARCHITECTURE_CONFIG, + compiler_config_path=COMPILER_CONFIG, + doe_config_path=self.doe_config_path, + scope=SWEEP_SCOPE_ALL, + ) + config_dir = DOE_CONFIG.resolve().parent + cfg["baseline_mo"] = (config_dir / cfg["baseline_mo"]).resolve().as_posix() + cfg["architecture"]["template"] = ( + (config_dir.parent / cfg["architecture"]["template"]).resolve().as_posix() + ) + self.doe_config_path.write_text(yaml.safe_dump(cfg, sort_keys=False)) + return cfg + + def _drop_if_inputs_changed(self, boblib_path: Path) -> bool: + """A cache must never outlive the inputs it was built from.""" + if not self.variants_dir.exists(): + return False + try: + # The same call compile_all makes, so the two cannot disagree about + # what counts as stale. + check_pipeline_hash( + self.variants_dir, + self.doe_config_path, + COMPILER_CONFIG, + boblib_path, + ARCHITECTURE_CONFIG, + PIPELINE_TOOLING_INPUTS, + ) + except RuntimeError: + print( + f"Cache under {self.root.name}/ predates a change to BobLib, the vehicle, or " + "the simulation tooling; discarding it." + ) + shutil.rmtree(self.variants_dir) + return True + return False + + def variant_dir(self, variant: Variant) -> Path: + return self.variants_dir / f"variant_{self._index[variant_key(variant)]:04d}" + + def build_dir(self, variant: Variant) -> Path: + return self.variant_dir(variant) / "build" / BUILD_STANDARD + + def ensure_compiled(self, variants: list[Variant]) -> None: + """Compile whichever of these vehicles has no executable yet, in one batch.""" + new = sorted({variant_key(v) for v in variants} - set(self._index)) + if not new and all(find_exe(self.build_dir(v), self.standard_cfg) for v in variants): + return + + for key in new: + self._index[key] = len(self._index) + self.variants_dir.mkdir(parents=True, exist_ok=True) + ordered = sorted(self._index, key=self._index.__getitem__) + generate_variants(self.doe_config_path, [json.loads(key) for key in ordered], self.variants_dir) + (self.variants_dir / "index.json").write_text(json.dumps(self._index, indent=1)) + compile_all( + self.variants_dir, + COMPILER_CONFIG, + doe_config_path=self.doe_config_path, + only_standards=(BUILD_STANDARD,), + ) + + for variant in variants: + if find_exe(self.build_dir(variant), self.standard_cfg) is None: + log = self.variant_dir(variant) / f"compile_error_{BUILD_STANDARD}.log" + detail = log.read_text()[-2000:] if log.exists() else "no compile log written" + raise RuntimeError(f"Compile failed for {variant or 'the baseline'}:\n{detail}") diff --git a/_4_OptSim/StandardSens/trade_study.py b/_4_OptSim/StandardSens/trade_study.py new file mode 100644 index 0000000..306f7c7 --- /dev/null +++ b/_4_OptSim/StandardSens/trade_study.py @@ -0,0 +1,198 @@ +"""Compare named vehicles across the standard sims. + + make opt-trade # configs/trade_study.yaml + make opt-trade STUDY=path/to/another_study.yaml + +OptSim asks three different questions of the same compiled vehicles: + +- `opt-standard` samples many vehicles to learn which parameters matter. +- `opt-solve` inverts for the setup that hits target metrics. +- `opt-trade`, this one, compares vehicles you name, on metrics you choose. + +Every candidate is compiled rather than overridden, so any swept variable is fair +game — mass, CG and toe included — and SteadyStateEval, RampSteerEval and +TransientEval all run against that one executable. See `pipeline/trade.py` for +what the comparison insists on and why it offers no score. +""" + +from __future__ import annotations + +import argparse +from concurrent.futures import ThreadPoolExecutor +import csv +from pathlib import Path +import sys +import time +from typing import Any + +import yaml + +from _shared.console import elapsed +from StandardSens.pipeline import trade +from StandardSens.pipeline.orchestration import RESULTS_DIR, STANDARD_BUILD_DIR +from StandardSens.pipeline.standards import ( + Standard, + get_standard, + input_digest, + read_metrics, + run_standard, +) +from StandardSens.pipeline.variants import VariantStore, split_cpus + +DEFAULT_STUDY = Path(__file__).resolve().parent / "configs/trade_study.yaml" +# One store for every study: the baseline, and any candidate two studies share, +# is compiled and simulated once. +TRADE_DIR = STANDARD_BUILD_DIR / "trade" +RUN_TIMEOUT_S = 3600 + +Variant = dict[str, float] + + +def load_candidates(study: dict[str, Any], variables: dict[str, dict[str, Any]]) -> dict[str, Variant]: + candidates: dict[str, Variant] = {} + for name, changes in (study.get("candidates") or {}).items(): + if name == trade.BASELINE: + raise ValueError(f"{trade.BASELINE!r} is always included; name your candidate something else") + if not changes: + raise ValueError(f"Candidate {name!r} changes nothing, so it is the baseline") + unknown = sorted(set(changes) - set(variables)) + if unknown: + raise KeyError( + f"Candidate {name!r} changes {unknown}, which are not variables in " + "vehicle_architecture.yaml. A parameter has to be declared there, with the " + "Modelica record block it maps to, before a study can change it." + ) + candidates[name] = {path: float(value) for path, value in changes.items()} + for path, value in candidates[name].items(): + low, high = variables[path]["range"] + if not low <= value <= high: + print( + f"note: {name}.{path} = {value:g} is outside the sweep range " + f"[{low:g}, {high:g}]. Allowed, but nothing has been simulated out there." + ) + if not candidates: + raise ValueError("The study names no candidates") + return candidates + + +def load_specs(study: dict[str, Any]) -> list[trade.MetricSpec]: + specs = [] + for standard, metrics in (study.get("metrics") or {}).items(): + get_standard(standard) + for name, options in (metrics or {}).items(): + resolution = (options or {}).get("resolution") + specs.append( + trade.MetricSpec(standard, name, None if resolution is None else float(resolution)) + ) + if not specs: + raise ValueError("The study names no metrics") + return specs + + +def _stamp(variant_dir: Path, standard: Standard) -> Path: + return variant_dir / "results" / standard.name / ".input_digest" + + +def is_cached(variant_dir: Path, standard: Standard, render_reports: bool) -> bool: + results_dir = variant_dir / "results" / standard.name + stamp = _stamp(variant_dir, standard) + return ( + (results_dir / "metrics.csv").exists() + and stamp.exists() + and stamp.read_text() == input_digest(standard) + and (not render_reports or (results_dir / f"{standard.stem}_report.pdf").exists()) + ) + + +def run(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--study", type=Path, default=DEFAULT_STUDY) + args = parser.parse_args(argv) + + study = yaml.safe_load(args.study.read_text()) + name = str(study.get("name") or args.study.stem) + specs = load_specs(study) + standards = [get_standard(s) for s in dict.fromkeys(spec.standard for spec in specs)] + render_reports = bool(study.get("render_reports", True)) + + store = VariantStore(TRADE_DIR) + variables = {v["path"]: v for v in store.doe_config["variables"]} + candidates = load_candidates(study, variables) + vehicles: dict[str, Variant] = {trade.BASELINE: {}, **candidates} + + started = time.time() + store.ensure_compiled(list(vehicles.values())) + + jobs = [ + (vehicle, standard) + for vehicle in vehicles.values() + for standard in standards + if not is_cached(store.variant_dir(vehicle), standard, render_reports) + ] + if jobs: + concurrent, workers = split_cpus(len(jobs), int(study.get("cpus", 8))) + total = len(vehicles) * len(standards) + print( + f"Simulating {len(jobs)} of {total} vehicle x standard run(s), {concurrent} at a " + f"time with {workers} case workers each ({total - len(jobs)} cached)...", + flush=True, + ) + + def simulate(job: tuple[Variant, Standard]) -> str: + vehicle, standard = job + variant_dir = store.variant_dir(vehicle) + run_standard( + standard, + variant_dir=variant_dir, + build_dir=store.build_dir(vehicle), + exec_name=store.standard_cfg["model"], + timeout=RUN_TIMEOUT_S, + render_report=render_reports, + max_workers=workers, + ) + _stamp(variant_dir, standard).write_text(input_digest(standard)) + return standard.name + + with ThreadPoolExecutor(max_workers=concurrent) as pool: + for done, finished in enumerate(pool.map(simulate, jobs), start=1): + print(f" [{done}/{len(jobs)}] {finished} {time.time() - started:.0f}s", flush=True) + + results: trade.Results = { + label: { + standard.name: read_metrics( + store.variant_dir(vehicle) / "results" / standard.name / "metrics.csv" + ) + for standard in standards + } + for label, vehicle in vehicles.items() + } + comparisons = trade.compare(results, specs) + problems = trade.lost_cases(results) + report = trade.to_markdown( + name, candidates, comparisons, trade.interactions(candidates, comparisons), problems + ) + + output_dir = RESULTS_DIR / "trade" + output_dir.mkdir(parents=True, exist_ok=True) + (output_dir / f"{name}.md").write_text(report, encoding="utf-8") + with (output_dir / f"{name}.csv").open("w", newline="", encoding="utf-8") as handle: + writer = csv.writer(handle) + writer.writerow( + ["standard", "metric", "units", "resolution", "candidate", "baseline", "value", "delta", "verdict"] + ) + for c in comparisons: + writer.writerow( + [c.spec.standard, c.spec.name, c.units, c.spec.resolution, c.candidate, + c.baseline, c.value, c.delta, c.verdict] + ) + + print("\n" + report) + print(f"Total {elapsed(started)}") + print(f"Report: {output_dir / f'{name}.md'}") + if render_reports: + print(f"Per-vehicle PDF reports: {store.variants_dir}/variant_*/results/") + return 2 if problems else 0 + + +if __name__ == "__main__": + sys.exit(run()) diff --git a/docs/doe-reverse-engineering.md b/docs/doe-reverse-engineering.md index 5305936..2296664 100644 --- a/docs/doe-reverse-engineering.md +++ b/docs/doe-reverse-engineering.md @@ -2,6 +2,22 @@ **TL;DR:** Sweep a population of vehicle variants → simulate them all → aggregate metrics → search backwards. Given target performance numbers, find the car that hits them. Everything here lives under `_4_OptSim/StandardSens/`. +## Three questions, one set of compiled vehicles + +OptSim asks three different things, and they are separate tools rather than +stages of one: + +| Target | Question | Vehicles | Answer | +| --- | --- | --- | --- | +| `make opt-standard` (+ `opt-refined`, `opt-search`) | Which parameters matter, and roughly where is a car with these numbers? | many, sampled | sensitivities, response surfaces, nearest sampled car | +| `make opt-solve` | What do I set on *this* car to hit these numbers? | a star of `2n + 1` around the current car | one setup, simulated | +| `make opt-trade` | What does each of these specific changes buy, and cost? | the ones you name | a comparison table | + +None replaces another. The sweep is still how you learn a design space; the +solver and the trade study are what you reach for once you have a specific +question. All three write their vehicles with the same generator and compile them +with the same compiler, so a variant means the same thing in each. + ## Quick start: a small sweep From a clean checkout, this is the whole thing: @@ -425,6 +441,84 @@ verification), and each later solve against different targets took about 70 seconds. A four-variant sweep of the same car took 26 minutes and cannot interpolate between its samples. +## Trade studies + +A trade study compares vehicles you name, on metrics you choose, across more than +one standard sim: + +```bash +make opt-trade # configs/trade_study.yaml +make opt-trade STUDY=path/to/another_study.yaml +``` + +The study file names candidates as changes from the baseline, and the metrics to +compare per standard: + +```yaml +name: rear_roll_stiffness +candidates: + stiff_rear_bar: {rear.stabar.rate_n_m_per_rad: 961.495352} + soft_front_spring: {front.actuation.spring_rate_n_per_m: 21015.2202} + both: {rear.stabar.rate_n_m_per_rad: 961.495352, + front.actuation.spring_rate_n_per_m: 21015.2202} +metrics: + SteadyStateEval: + understeer_gradient_deg_per_g: {resolution: 0.02} + TransientEval: + yaw_rise_time_s: {resolution: 0.005} +``` + +The report lands in `_4_OptSim/results/trade/.md` and `.csv`: one table per +standard, each cell the simulated value with its change from baseline. + +**Every candidate is compiled, never overridden.** That is the opposite choice +from `opt-solve`, for a reason: a trade study wants to change mass, CG, toe and +camber, which are exactly the parameters an override silently fails to move (see +above). A compile is always correct, costs about 36 s per vehicle when several +build at once, and is cached by content, so the baseline and any candidate two +studies share are built and simulated once across all of them. + +**One compile serves every standard.** SteadyStateEval, RampSteerEval and +TransientEval all run the same model, `BobLib.Experiments.Standards.VehicleSim`; +`_3_StandardSim` itself builds it once and points all three configs at it. They +share one interface too, so a standard is one entry in `STANDARDS` in +`pipeline/standards.py`. FourPostEval is not there: it runs `FourPostSim`, a +different model, so it would need a second compile per vehicle. + +**There is no score and no ranking, on purpose.** How much understeer is worth how +much settling time is an engineering judgement, and a weighted sum would bury it +inside a number. What the report does insist on is that a difference be worth +reading: + +- **Resolution.** Each metric may state the smallest change worth acting on. A + smaller delta is still shown, marked `~`, so a 0.001 s change in rise time is + not read as a finding. Leave it off and the delta is reported unjudged. +- **Lost cases.** A run whose simulation lost cases is marked `n/c` and not + compared. Its fits run through fewer points than the baseline's, so the delta + would mix the design change with the missing data. A baseline that lost cases + makes every comparison on that standard `n/c`. The command exits 2. +- **Stacking.** When one candidate makes exactly the changes of two others, the + report gives the interaction: the combined effect minus the sum of the parts. + Above the metric's resolution the combination does something neither part + predicts, and "we'll do both" is not the sum you budgeted for. Below it the + interaction cannot be told apart from zero, which is weaker than saying the + changes add, so the number is printed beside the label. +- **Ambiguous names.** TransientEval reports some metrics once per group under + one name (`yaw_gain_dc` for the step and again for the frequency sweep). Those + are only available qualified, `step.yaw_gain_dc`, and asking for the bare name + is an error that lists the options. + +**A candidate can only change a declared variable.** The variables are the +`sweep.variables` entries in `configs/vehicle_architecture.yaml`, because each +needs the Modelica record block it maps to. To trade on something new, such as +wheelbase, declare it there first (see *Adding a swept parameter*). A value outside +a variable's sweep range is allowed and noted: the range bounds what the sweep +samples, not what is physically valid. + +Compare like with like: a metric here comes from each standard's own test matrix, +so a gradient from `opt-trade` matches `make standard-eval-steady-state`, not +`opt-solve`, which fits through its own denser isoline. + ## Envelope sensitivities `_4_OptSim/EnvelopeSens/` is the same idea against GGV/YMD envelope outputs diff --git a/makefile b/makefile index 34493d0..0639637 100644 --- a/makefile +++ b/makefile @@ -20,6 +20,7 @@ BUILD_FOUR_POST_MOS := _3_StandardSim/build_four_post_sim.mos SEARCH_TOP ?= 1 TARGETS ?= KNOBS ?= +STUDY ?= REDUCED_DOF ?= 6 REDUCED_KINEMATICS ?= lookup REDUCED_BOBLIB_CSV ?= @@ -137,7 +138,7 @@ CLEAN_DOCKER_IMAGE ?= bobdyn/bobsim:latest lap-validation-visuals \ envelope-ggv envelope-ymd envelope-all \ opt-standard opt-standard-setup opt-standard-architecture \ - opt-envelope opt-refined opt-search opt-solve opt-doe-smoke \ + opt-envelope opt-refined opt-search opt-solve opt-trade opt-doe-smoke \ clean clean-app clean-visual clean-standard clean-envelope clean-opt clean-owned clean-all ifeq ($(OS),Windows_NT) @@ -226,12 +227,14 @@ help: ' opt-refined Run StandardSens refined response surfaces' \ ' opt-search Reverse lookup: target metrics -> vehicle parameters' \ ' opt-solve Solve for the setup that hits target metrics, then simulate it' \ + ' opt-trade Compare named vehicles across the standard sims' \ '' \ ' Search variables:' \ ' METRICS="NAME=VALUE ..." Required target metrics' \ ' SEARCH_TOP= Nearest variants to return. Default: 1' \ ' TARGETS="NAME=VALUE ..." opt-solve targets. Default: configs/solve_config.yaml' \ ' KNOBS="path ..." opt-solve knobs. Default: configs/solve_config.yaml' \ + ' STUDY= opt-trade study. Default: configs/trade_study.yaml' \ '' \ ' DOE sweep variables (default: configs/vehicle_architecture.yaml):' \ ' DOE_METHOD=lhs|interval_splice Sampling method' \ @@ -506,6 +509,12 @@ opt-search: opt-solve: $(FOUR_POST_METRICS) $(RUN) env PYTHONPATH=$(WORKSPACE)/_4_OptSim:$(WORKSPACE) $(PYTHON) -m StandardSens.solve_setup $(if $(TARGETS),--targets $(TARGETS),) $(if $(KNOBS),--knobs $(KNOBS),) +# Compiles each named candidate once and runs every requested standard against +# that one executable. Needs the FourPostEval motion ratios whenever a candidate +# changes a spring rate, for the same reason opt-standard does. +opt-trade: $(FOUR_POST_METRICS) + $(RUN) env PYTHONPATH=$(WORKSPACE)/_4_OptSim:$(WORKSPACE) $(PYTHON) -m StandardSens.trade_study $(if $(STUDY),--study $(STUDY),) + clean: bash -lc 'find $(CLEAN_WORKSPACE) -type d -name "__pycache__" -exec rm -rf {} + 2>/dev/null; \ find $(CLEAN_WORKSPACE) -type f \( -name "*.pyc" -o -name "*.pyo" \) -delete 2>/dev/null; \ diff --git a/tests/test_optsim_overrides.py b/tests/test_optsim_overrides.py index 3a5c633..035609d 100644 --- a/tests/test_optsim_overrides.py +++ b/tests/test_optsim_overrides.py @@ -130,11 +130,11 @@ def test_runtime_safe_knobs_share_one_executable_and_the_rest_do_not() -> None: assert bar in overrides.RUNTIME_SAFE_PATHS assert toe not in overrides.RUNTIME_SAFE_PATHS, "toe is baked into the wheel rotation matrix" baseline = {bar: 535.0, toe: 0.0} - soft = evaluator.compile_key({bar: 400.0, toe: 0.0}, baseline) - stiff = evaluator.compile_key({bar: 900.0, toe: 0.0}, baseline) - toed = evaluator.compile_key({bar: 900.0, toe: 0.1}, baseline) - assert soft == stiff == (), "bar changes must reuse the baseline executable" - assert toed == ((toe, 0.1),), "a toe change must get its own executable" + soft = evaluator.compiled_part({bar: 400.0, toe: 0.0}, baseline) + stiff = evaluator.compiled_part({bar: 900.0, toe: 0.0}, baseline) + toed = evaluator.compiled_part({bar: 900.0, toe: 0.1}, baseline) + assert soft == stiff == {}, "bar changes must reuse the baseline executable" + assert toed == {toe: 0.1}, "a toe change must get its own executable" def test_command_line_targets_replace_the_configured_ones() -> None: diff --git a/tests/test_optsim_trade.py b/tests/test_optsim_trade.py new file mode 100644 index 0000000..20ca347 --- /dev/null +++ b/tests/test_optsim_trade.py @@ -0,0 +1,186 @@ +"""The trade study's comparison logic, and the multi-standard plumbing under it. + +A trade study is only worth running if its table can be trusted, so these pin the +ways a comparison goes quietly wrong: a delta too small to mean anything read as +a finding, a run that lost cases compared as if it were whole, two changes +assumed to add when they do not, and a metric name that means two things. +""" + +from __future__ import annotations + +from pathlib import Path +import sys + +import pytest + +ROOT = Path(__file__).resolve().parents[1] +OPTSIM_DIR = ROOT / "_4_OptSim" +if str(OPTSIM_DIR) not in sys.path: + sys.path.insert(0, str(OPTSIM_DIR)) + +pytest.importorskip("scipy", reason="the OptSim pipeline package imports scipy") + +from StandardSens import trade_study # noqa: E402 +from StandardSens.pipeline import standards, trade # noqa: E402 +from StandardSens.pipeline.variants import split_cpus, variant_key # noqa: E402 + +UNDERSTEER = trade.MetricSpec("SteadyStateEval", "understeer", resolution=0.02) +RISE = trade.MetricSpec("TransientEval", "yaw_rise_time_s", resolution=0.005) +WHOLE = {"n_cases": (8.0, "count"), "n_successful_cases": (8.0, "count")} + + +def run(understeer: float, rise: float, **overrides) -> dict[str, dict[str, tuple[float, str]]]: + return { + "SteadyStateEval": {**WHOLE, "understeer": (understeer, "deg/g"), **overrides}, + "TransientEval": {**WHOLE, "yaw_rise_time_s": (rise, "s")}, + } + + +def test_a_delta_below_the_resolution_is_shown_but_not_called_a_finding() -> None: + results = {"baseline": run(0.300, 0.055), "bar": run(0.345, 0.056)} + by_metric = {c.spec.name: c for c in trade.compare(results, [UNDERSTEER, RISE])} + assert by_metric["understeer"].verdict == trade.RESOLVED + assert by_metric["understeer"].delta == pytest.approx(0.045) + assert by_metric["yaw_rise_time_s"].verdict == trade.BELOW_RESOLUTION + assert by_metric["yaw_rise_time_s"].delta == pytest.approx(0.001), "still reported" + + +def test_a_run_that_lost_cases_is_not_compared() -> None: + """Its fits run through fewer points, so the delta would mix two causes.""" + lossy = run(0.40, 0.055, n_successful_cases=(6.0, "count")) + results = {"baseline": run(0.30, 0.055), "bar": lossy} + assert trade.lost_cases(results) == {("bar", "SteadyStateEval"): "2 of 8 cases failed"} + by_metric = {c.spec.name: c for c in trade.compare(results, [UNDERSTEER, RISE])} + assert by_metric["understeer"].verdict == trade.NOT_COMPARABLE + assert by_metric["yaw_rise_time_s"].verdict != trade.NOT_COMPARABLE, "other standard is fine" + + +def test_a_lossy_baseline_poisons_every_comparison_on_that_standard() -> None: + results = {"baseline": run(0.30, 0.055, n_successful_cases=(7.0, "count")), "bar": run(0.35, 0.040)} + verdicts = {c.spec.name: c.verdict for c in trade.compare(results, [UNDERSTEER, RISE])} + assert verdicts == {"understeer": trade.NOT_COMPARABLE, "yaw_rise_time_s": trade.RESOLVED} + + +def test_interaction_is_found_for_a_candidate_that_is_two_others_combined() -> None: + candidates = {"bar": {"rear.bar": 900.0}, "spring": {"front.spring": 21000.0}, + "both": {"rear.bar": 900.0, "front.spring": 21000.0}, + "unrelated": {"rear.bar": 700.0}} + results = { + "baseline": run(0.300, 0.055), + "bar": run(0.260, 0.055), # -0.040 + "spring": run(0.290, 0.055), # -0.010 + "both": run(0.200, 0.055), # -0.100, not the -0.050 the parts predict + "unrelated": run(0.280, 0.055), + } + found = trade.interactions(candidates, trade.compare(results, [UNDERSTEER])) + assert [(i.combined, i.parts) for i in found] == [("both", ("bar", "spring"))] + assert found[0].value == pytest.approx(-0.050) + assert found[0].stacks is False, "0.05 of unexplained understeer is five resolutions" + + +def test_changes_that_simply_add_are_reported_as_stacking() -> None: + candidates = {"a": {"x": 1.0}, "b": {"y": 1.0}, "ab": {"x": 1.0, "y": 1.0}} + results = {"baseline": run(0.30, 0.055), "a": run(0.26, 0.055), "b": run(0.29, 0.055), + "ab": run(0.251, 0.055)} + (found,) = trade.interactions(candidates, trade.compare(results, [UNDERSTEER])) + assert found.value == pytest.approx(0.001) and found.stacks is True + + +def test_overlapping_changes_are_not_mistaken_for_a_combination() -> None: + """`ab` sets x to a different value than `a` does, so it is not a + b.""" + candidates = {"a": {"x": 1.0}, "b": {"y": 1.0}, "ab": {"x": 2.0, "y": 1.0}} + results = {n: run(0.3, 0.055) for n in ("baseline", "a", "b", "ab")} + assert trade.interactions(candidates, trade.compare(results, [UNDERSTEER])) == [] + + +def test_an_unknown_metric_names_what_was_probably_meant() -> None: + results = {"baseline": run(0.3, 0.055), "bar": run(0.3, 0.055)} + with pytest.raises(KeyError, match="understeer"): + trade.compare(results, [trade.MetricSpec("SteadyStateEval", "understeer_grad")]) + + +def test_a_name_reported_per_group_must_be_asked_for_by_group(tmp_path: Path) -> None: + csv_path = tmp_path / "metrics.csv" + csv_path.write_text( + "standard,group,metric,value,units,description\n" + "T,step,yaw_gain_dc,2.10,(rad/s)/rad,\n" + "T,frequency,yaw_gain_dc,2.30,(rad/s)/rad,\n" + "T,step,yaw_overshoot_pct,16.6,%,\n" + "T,step,yaw_overshoot_pct,16.6,%,\n" # TransientEval really does write this twice + ) + metrics = standards.read_metrics(csv_path) + assert "yaw_gain_dc" not in metrics, "the bare name would silently mean the last row" + assert metrics["step.yaw_gain_dc"][0] == 2.10 and metrics["frequency.yaw_gain_dc"][0] == 2.30 + assert metrics["yaw_overshoot_pct"] == (16.6, "%"), "an identical repeat is one metric" + + results = {"baseline": {"TransientEval": {**WHOLE, **metrics}}, "bar": {"TransientEval": {**WHOLE, **metrics}}} + with pytest.raises(KeyError, match=r"name one of: \['frequency.yaw_gain_dc', 'step.yaw_gain_dc'\]"): + trade.compare(results, [trade.MetricSpec("TransientEval", "yaw_gain_dc")]) + + +def test_a_metric_repeated_with_conflicting_values_is_an_error(tmp_path: Path) -> None: + csv_path = tmp_path / "metrics.csv" + csv_path.write_text("metric,value\nroll_gain,0.04\nroll_gain,0.05\n") + with pytest.raises(ValueError, match="twice with different values"): + standards.read_metrics(csv_path) + + +def test_every_registered_standard_exists_and_shares_the_vehicle_sim_executable() -> None: + import yaml + + for standard in standards.STANDARDS.values(): + assert standard.sim_path.is_file(), standard.sim_path + config = yaml.safe_load(standard.config_path.read_text()) + assert config["simulation"]["exec_name"] == standards.VEHICLE_SIM_MODEL, ( + f"{standard.name} runs a different model, so one compile cannot serve it" + ) + with pytest.raises(KeyError, match="FourPostSim"): + standards.get_standard("FourPostEval") + + +def test_a_standards_config_is_written_beside_the_variant_not_over_the_shared_one(tmp_path: Path) -> None: + standard = standards.get_standard("TransientEval") + before = standard.config_path.read_bytes() + config_path, metrics_csv = standards.write_config( + standard, variant_dir=tmp_path, build_dir=tmp_path / "build", render_report=False, max_workers=6 + ) + import yaml + + written = yaml.safe_load(config_path.read_text()) + assert standard.config_path.read_bytes() == before + assert written["simulation"]["build_dir"] == str(tmp_path / "build") + assert written["report"]["enabled"] is False and written["execution"]["max_workers"] == 6 + assert metrics_csv == tmp_path / "results/TransientEval/transient_eval_report_metrics.csv" + + +def test_study_candidates_are_validated_before_anything_is_compiled() -> None: + variables = {"rear.bar": {"range": [300.0, 900.0]}} + good = trade_study.load_candidates({"candidates": {"stiff": {"rear.bar": 800}}}, variables) + assert good == {"stiff": {"rear.bar": 800.0}} + with pytest.raises(KeyError, match="not variables in vehicle_architecture.yaml"): + trade_study.load_candidates({"candidates": {"x": {"wheelbase": 1.6}}}, variables) + with pytest.raises(ValueError, match="always included"): + trade_study.load_candidates({"candidates": {"baseline": {"rear.bar": 800}}}, variables) + with pytest.raises(ValueError, match="changes nothing"): + trade_study.load_candidates({"candidates": {"same": {}}}, variables) + + +def test_the_example_study_names_real_standards_and_is_loadable() -> None: + import yaml + + study = yaml.safe_load(trade_study.DEFAULT_STUDY.read_text()) + specs = trade_study.load_specs(study) + assert {s.standard for s in specs} <= set(standards.STANDARDS) + assert all(s.resolution and s.resolution > 0 for s in specs) + + +def test_cpus_go_to_concurrent_runs_until_each_would_be_starved() -> None: + assert split_cpus(9, 12) == (3, 4) + assert split_cpus(1, 12) == (1, 12), "a lone run gets every CPU" + assert split_cpus(2, 12) == (2, 6) + assert split_cpus(5, 2) == (1, 2), "never zero concurrent runs" + + +def test_float_noise_does_not_split_one_vehicle_into_two_cache_entries() -> None: + assert variant_key({"b": 0.1 + 0.2, "a": 1.0}) == variant_key({"a": 1.0, "b": 0.3}) + assert variant_key({}) != variant_key({"a": 1.0}) From a36fbf73425e044451d0d8df4ce25cdc63c699bc Mon Sep 17 00:00:00 2001 From: Arjun Rao Date: Sun, 20 Sep 2026 22:47:53 -0500 Subject: [PATCH 2/2] Stop the solver fitting through an evaluation that lost cases The trade study refused to compare a run that lost simulation cases, but the solver did not. A star point that lost a case still returns finite gradients, fitted through fewer points. So it bent the surrogate, and no number looked wrong. Evaluations are cached, so it would also have bent every later solve. Both tools now share one definition, standards.case_loss. The solver stops and reports the variant, the counts and what to change. This commit also replaces the error for tooling that changes during a run. The error came from the sweep's compiler and said to run make clean-opt. That advice is wrong here, because the store discards a stale cache by itself on the next run. A real run hit this error when another process edited hashed files between the star and the first verification compile. The first end-to-end solve over compile-only knobs found that problem. The solve now passes. Front toe, rear toe and the rear bar together built five executables in one batch, and the bar's star points shared the baseline executable. The verification compiled exactly one more. The solve converged on the first try at 0.3578 / 0.8359, against 0.35 / 0.85, in 617 s. --- _4_OptSim/StandardSens/pipeline/evaluator.py | 24 +++++++++++++- _4_OptSim/StandardSens/pipeline/standards.py | 20 ++++++++++- _4_OptSim/StandardSens/pipeline/trade.py | 11 +++--- _4_OptSim/StandardSens/pipeline/variants.py | 35 +++++++++++++------- tests/test_optsim_trade.py | 21 ++++++++++++ 5 files changed, 91 insertions(+), 20 deletions(-) diff --git a/_4_OptSim/StandardSens/pipeline/evaluator.py b/_4_OptSim/StandardSens/pipeline/evaluator.py index 218bc1d..e6177d3 100644 --- a/_4_OptSim/StandardSens/pipeline/evaluator.py +++ b/_4_OptSim/StandardSens/pipeline/evaluator.py @@ -29,6 +29,7 @@ from StandardSens.pipeline.generator import build_context, read_metrics_csv from StandardSens.pipeline.orchestration import STANDARD_BUILD_DIR from StandardSens.pipeline.sampler import read_baseline +from StandardSens.pipeline.standards import case_loss from StandardSens.pipeline.steady_state_eval_report import Isoline, run_report from StandardSens.pipeline.variants import BUILD_STANDARD, VariantStore, split_cpus, variant_key @@ -82,7 +83,10 @@ def __call__(self, variants: list[Variant]) -> list[Metrics]: for done, _ in enumerate(pool.map(simulate, pending), start=1): print(f" [{done}/{len(pending)}] {time.time() - started:.0f}s", flush=True) - return [read_metrics_csv(self._metrics_path(v)) for v in variants] + results = [read_metrics_csv(self._metrics_path(v)) for v in variants] + for variant, metrics in zip(variants, results, strict=True): + require_whole(variant, metrics) + return results def _simulate(self, variant: Variant, *, workers: int) -> None: build_dir = self.store.build_dir(compiled_part(variant, self.baseline)) @@ -117,6 +121,24 @@ def _metrics_path(self, variant: Variant) -> Path: return self._eval_dir(variant) / "results" / BUILD_STANDARD / "metrics.csv" +def require_whole(variant: Variant, metrics: Metrics) -> None: + """Refuse an evaluation that lost simulation cases. + + Its gradients are fitted through fewer points than its neighbours', so it + would bend the surrogate without any number looking wrong, and because + evaluations are cached it would keep bending every later solve too. + """ + why = case_loss(metrics) + if why: + raise RuntimeError( + f"SteadyStateEval was not whole for {variant}: {why}. Its metrics are not " + "comparable with the other evaluations, so the solve stops here rather than " + "fit through them. The usual cause is a target a_y this setup cannot settle " + "at: lower the top of `test_matrix.target_ays` in solve_config.yaml (which " + "re-simulates the star), or inspect the run under Build/StandardSens/solve/." + ) + + def compiled_part(variant: Variant, baseline: Variant) -> Variant: """The changes that decide which executable a variant needs. diff --git a/_4_OptSim/StandardSens/pipeline/standards.py b/_4_OptSim/StandardSens/pipeline/standards.py index 06af10a..d542a85 100644 --- a/_4_OptSim/StandardSens/pipeline/standards.py +++ b/_4_OptSim/StandardSens/pipeline/standards.py @@ -14,10 +14,11 @@ from __future__ import annotations -from collections.abc import Callable +from collections.abc import Callable, Mapping import csv from dataclasses import dataclass import hashlib +import math from pathlib import Path import shutil import subprocess @@ -174,6 +175,23 @@ def run_standard( return canonical +def case_loss(metrics: Mapping[str, float]) -> str | None: + """Say why a run is not whole, or return None when every case settled. + + Every VehicleSim standard exports `n_cases` and `n_successful_cases`. A run + that lost cases fits its gradients through fewer points, so its metrics are + not comparable with a whole run's, and nothing downstream can tell from the + numbers alone. + """ + total = metrics.get("n_cases", math.nan) + good = metrics.get("n_successful_cases", math.nan) + if not (math.isfinite(total) and math.isfinite(good)): + return "reported no case counts" + if good < total: + return f"{int(total - good)} of {int(total)} cases failed" + return None + + def read_metrics(path: Path) -> dict[str, tuple[float, str]]: """Read a standard's metrics CSV as {name: (value, units)}. diff --git a/_4_OptSim/StandardSens/pipeline/trade.py b/_4_OptSim/StandardSens/pipeline/trade.py index af41032..dcbdc34 100644 --- a/_4_OptSim/StandardSens/pipeline/trade.py +++ b/_4_OptSim/StandardSens/pipeline/trade.py @@ -27,6 +27,8 @@ from itertools import combinations import math +from StandardSens.pipeline.standards import case_loss + BASELINE = "baseline" Variant = dict[str, float] @@ -80,12 +82,9 @@ def lost_cases(results: Results) -> dict[tuple[str, str], str]: problems: dict[tuple[str, str], str] = {} for candidate, by_standard in results.items(): for standard, metrics in by_standard.items(): - total = metrics.get("n_cases", (math.nan, ""))[0] - good = metrics.get("n_successful_cases", (math.nan, ""))[0] - if not (math.isfinite(total) and math.isfinite(good)): - problems[candidate, standard] = "reported no case counts" - elif good < total: - problems[candidate, standard] = f"{int(total - good)} of {int(total)} cases failed" + why = case_loss({name: value for name, (value, _units) in metrics.items()}) + if why: + problems[candidate, standard] = why return problems diff --git a/_4_OptSim/StandardSens/pipeline/variants.py b/_4_OptSim/StandardSens/pipeline/variants.py index 03c4139..3601246 100644 --- a/_4_OptSim/StandardSens/pipeline/variants.py +++ b/_4_OptSim/StandardSens/pipeline/variants.py @@ -64,9 +64,18 @@ def __init__(self, root: Path) -> None: compiler_cfg = load_compiler_config(COMPILER_CONFIG) self.standard_cfg: dict[str, Any] = compiler_cfg["standards"][BUILD_STANDARD] - boblib_path = (COMPILER_CONFIG.resolve().parent / compiler_cfg["boblib_path"]).resolve() + self._boblib_path = ( + COMPILER_CONFIG.resolve().parent / compiler_cfg["boblib_path"] + ).resolve() - self.was_reset = self._drop_if_inputs_changed(boblib_path) + # A cache must never outlive the inputs it was built from. + self.was_reset = self.variants_dir.exists() and self._inputs_changed() + if self.was_reset: + print( + f"Cache under {self.root.name}/ predates a change to BobLib, the vehicle, or " + "the simulation tooling; discarding it." + ) + shutil.rmtree(self.variants_dir) index_path = self.variants_dir / "index.json" self._index: dict[str, int] = json.loads(index_path.read_text()) if index_path.exists() else {} @@ -94,10 +103,8 @@ def _write_own_doe_config(self) -> dict[str, Any]: self.doe_config_path.write_text(yaml.safe_dump(cfg, sort_keys=False)) return cfg - def _drop_if_inputs_changed(self, boblib_path: Path) -> bool: - """A cache must never outlive the inputs it was built from.""" - if not self.variants_dir.exists(): - return False + def _inputs_changed(self) -> bool: + """Whether the inputs differ from the ones the compiled variants were built from.""" try: # The same call compile_all makes, so the two cannot disagree about # what counts as stale. @@ -105,16 +112,11 @@ def _drop_if_inputs_changed(self, boblib_path: Path) -> bool: self.variants_dir, self.doe_config_path, COMPILER_CONFIG, - boblib_path, + self._boblib_path, ARCHITECTURE_CONFIG, PIPELINE_TOOLING_INPUTS, ) except RuntimeError: - print( - f"Cache under {self.root.name}/ predates a change to BobLib, the vehicle, or " - "the simulation tooling; discarding it." - ) - shutil.rmtree(self.variants_dir) return True return False @@ -130,6 +132,15 @@ def ensure_compiled(self, variants: list[Variant]) -> None: if not new and all(find_exe(self.build_dir(v), self.standard_cfg) for v in variants): return + if self.variants_dir.exists() and self._inputs_changed(): + # Only reachable mid-run: a stale cache is discarded at construction. + raise RuntimeError( + "BobLib, the vehicle or the simulation tooling changed while this run was in " + "progress, so the vehicles already simulated and the ones still to compile " + "would not be comparable. Nothing on disk is damaged: run it again once the " + "edits have settled, and the cache will be rebuilt against the current inputs." + ) + for key in new: self._index[key] = len(self._index) self.variants_dir.mkdir(parents=True, exist_ok=True) diff --git a/tests/test_optsim_trade.py b/tests/test_optsim_trade.py index 20ca347..f85abdb 100644 --- a/tests/test_optsim_trade.py +++ b/tests/test_optsim_trade.py @@ -184,3 +184,24 @@ def test_cpus_go_to_concurrent_runs_until_each_would_be_starved() -> None: def test_float_noise_does_not_split_one_vehicle_into_two_cache_entries() -> None: assert variant_key({"b": 0.1 + 0.2, "a": 1.0}) == variant_key({"a": 1.0, "b": 0.3}) assert variant_key({}) != variant_key({"a": 1.0}) + + +def test_the_solver_refuses_an_evaluation_that_lost_cases() -> None: + """One definition of "not whole", shared by the trade table and the solver. + + A star point that lost a case still returns finite gradients, fitted through + fewer points. Accepted silently it bends the surrogate, and because + evaluations are cached it would bend every later solve as well. + """ + from StandardSens.pipeline.evaluator import require_whole + + whole = {"n_cases": 8.0, "n_successful_cases": 8.0, "understeer": 0.3} + assert standards.case_loss(whole) is None + require_whole({"rear.bar": 700.0}, whole) + + lossy = {**whole, "n_successful_cases": 7.0} + assert standards.case_loss(lossy) == "1 of 8 cases failed" + with pytest.raises(RuntimeError, match="1 of 8 cases failed.*target_ays"): + require_whole({"rear.bar": 700.0}, lossy) + + assert standards.case_loss({"understeer": 0.3}) == "reported no case counts"